{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/large-language-model/papers/4","list_of":"/task/large-language-model","task":"Large Language Model","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":61,"rows_per_page":100,"rows":[301,400],"of":6097,"counts":{"archive_papers_tagged":6097,"with_a_code_link":2250,"where_syntology_ran_a_sample":801,"not_listed_spam_title":0,"listed":6097,"listed_where_code_ran":801,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":683,"every_run_a_failure_of_syntologys_instrument":118,"listed_with_a_run_with_no_instrument_failure":683,"listed_every_run_a_failure_of_syntologys_instrument":118,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/large-language-model","prev":"/task/large-language-model/papers/3","next":"/task/large-language-model/papers/5","papers":[{"url":"/paper/bridging-the-gap-between-open-source-and","slug":"bridging-the-gap-between-open-source-and","title":"Bridging the Gap Between Open-Source and Proprietary LLMs in Table QA","date":"2025-06-11","arxiv_id":"2506.09657","repositories_listed":1,"syntology":null},{"url":"/paper/v-jepa-2-self-supervised-video-models-enable","slug":"v-jepa-2-self-supervised-video-models-enable","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","date":"2025-06-11","arxiv_id":"2506.09985","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/v-jepa-2-self-supervised-video-models-enable#ran","syntology_url":"https://syntology.ai/paper/2506.09985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09985"}},"official":{"repos":["facebookresearch/vjepa2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/adapting-vision-language-foundation-model-for","slug":"adapting-vision-language-foundation-model-for","title":"Adapting Vision-Language Foundation Model for Next Generation Medical Ultrasound Image Analysis","date":"2025-06-10","arxiv_id":"2506.08849","repositories_listed":1,"syntology":null},{"url":"/paper/consistent-paths-lead-to-truth-self-rewarding","slug":"consistent-paths-lead-to-truth-self-rewarding","title":"Consistent Paths Lead to Truth: Self-Rewarding Reinforcement Learning for LLM Reasoning","date":"2025-06-10","arxiv_id":"2506.08745","repositories_listed":1,"syntology":null},{"url":"/paper/xgraphrag-interactive-visual-analysis-for","slug":"xgraphrag-interactive-visual-analysis-for","title":"XGraphRAG: Interactive Visual Analysis for Graph-based Retrieval-Augmented Generation","date":"2025-06-10","arxiv_id":"2506.13782","repositories_listed":1,"syntology":null},{"url":"/paper/cognitive-weave-synthesizing-abstracted","slug":"cognitive-weave-synthesizing-abstracted","title":"Cognitive Weave: Synthesizing Abstracted Knowledge with a Spatio-Temporal Resonance Graph","date":"2025-06-09","arxiv_id":"2506.08098","repositories_listed":1,"syntology":null},{"url":"/paper/g-memory-tracing-hierarchical-memory-for","slug":"g-memory-tracing-hierarchical-memory-for","title":"G-Memory: Tracing Hierarchical Memory for Multi-Agent Systems","date":"2025-06-09","arxiv_id":"2506.07398","repositories_listed":1,"syntology":null},{"url":"/paper/how-benchmark-prediction-from-fewer-data","slug":"how-benchmark-prediction-from-fewer-data","title":"How Benchmark Prediction from Fewer Data Misses the Mark","date":"2025-06-09","arxiv_id":"2506.07673","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/how-benchmark-prediction-from-fewer-data#ran","syntology_url":"https://syntology.ai/paper/2506.07673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07673"}},"official":{"repos":["socialfoundations/benchmark-prediction"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-what-reinforcement-learning-can-t","slug":"learning-what-reinforcement-learning-can-t","title":"Learning What Reinforcement Learning Can't: Interleaved Online Fine-Tuning for Hardest Questions","date":"2025-06-09","arxiv_id":"2506.07527","repositories_listed":1,"syntology":null},{"url":"/paper/minicpm4-ultra-efficient-llms-on-end-devices","slug":"minicpm4-ultra-efficient-llms-on-end-devices","title":"MiniCPM4: Ultra-Efficient LLMs on End Devices","date":"2025-06-09","arxiv_id":"2506.07900","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minicpm4-ultra-efficient-llms-on-end-devices#ran","syntology_url":"https://syntology.ai/paper/2506.07900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07900"}},"official":{"repos":["openbmb/minicpm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/annodpo-protein-functional-annotation","slug":"annodpo-protein-functional-annotation","title":"AnnoDPO: Protein Functional Annotation Learning with Direct Preference Optimization","date":"2025-06-08","arxiv_id":"2506.07035","repositories_listed":1,"syntology":null},{"url":"/paper/dam-dynamic-attention-mask-for-long-context","slug":"dam-dynamic-attention-mask-for-long-context","title":"DAM: Dynamic Attention Mask for Long-Context Large Language Model Inference Acceleration","date":"2025-06-06","arxiv_id":"2506.11104","repositories_listed":1,"syntology":null},{"url":"/paper/eigenspectrum-analysis-of-neural-networks","slug":"eigenspectrum-analysis-of-neural-networks","title":"Eigenspectrum Analysis of Neural Networks without Aspect Ratio Bias","date":"2025-06-06","arxiv_id":"2506.06280","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eigenspectrum-analysis-of-neural-networks#ran","syntology_url":"https://syntology.ai/paper/2506.06280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.06280"}},"official":null}},{"url":"/paper/heavywater-and-simplexwater-watermarking-low","slug":"heavywater-and-simplexwater-watermarking-low","title":"HeavyWater and SimplexWater: Watermarking Low-Entropy Text Distributions","date":"2025-06-06","arxiv_id":"2506.06409","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/heavywater-and-simplexwater-watermarking-low#ran","syntology_url":"https://syntology.ai/paper/2506.06409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.06409"}},"official":{"repos":["dortsur/heavywater_simplexwater"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/agentomics-ml-autonomous-machine-learning","slug":"agentomics-ml-autonomous-machine-learning","title":"Agentomics-ML: Autonomous Machine Learning Experimentation Agent for Genomic and Transcriptomic Data","date":"2025-06-05","arxiv_id":"2506.05542","repositories_listed":1,"syntology":null},{"url":"/paper/comfyui-copilot-an-intelligent-assistant-for","slug":"comfyui-copilot-an-intelligent-assistant-for","title":"ComfyUI-Copilot: An Intelligent Assistant for Automated Workflow Development","date":"2025-06-05","arxiv_id":"2506.05010","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/comfyui-copilot-an-intelligent-assistant-for#ran","syntology_url":"https://syntology.ai/paper/2506.05010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.05010"}},"official":{"repos":["aidc-ai/comfyui-copilot"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exp4fuse-a-rank-fusion-framework-for-enhanced","slug":"exp4fuse-a-rank-fusion-framework-for-enhanced","title":"Exp4Fuse: A Rank Fusion Framework for Enhanced Sparse Retrieval using Large Language Model-based Query Expansion","date":"2025-06-05","arxiv_id":"2506.04760","repositories_listed":1,"syntology":null},{"url":"/paper/halos-hierarchical-asynchronous-local-sgd","slug":"halos-hierarchical-asynchronous-local-sgd","title":"HALoS: Hierarchical Asynchronous Local SGD over Slow Networks for Geo-Distributed Large Language Model Training","date":"2025-06-05","arxiv_id":"2506.04531","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-channel-allocation-for-ieee-802","slug":"intelligent-channel-allocation-for-ieee-802","title":"Intelligent Channel Allocation for IEEE 802.11be Multi-Link Operation: When MAB Meets LLM","date":"2025-06-05","arxiv_id":"2506.04594","repositories_listed":1,"syntology":null},{"url":"/paper/openmaskdino3d-reasoning-3d-segmentation-via","slug":"openmaskdino3d-reasoning-3d-segmentation-via","title":"OpenMaskDINO3D : Reasoning 3D Segmentation via Large Language Model","date":"2025-06-05","arxiv_id":"2506.04837","repositories_listed":1,"syntology":null},{"url":"/paper/poss-position-specialist-generates-better","slug":"poss-position-specialist-generates-better","title":"POSS: Position Specialist Generates Better Draft for Speculative Decoding","date":"2025-06-04","arxiv_id":"2506.03566","repositories_listed":1,"syntology":null},{"url":"/paper/rewardanything-generalizable-principle","slug":"rewardanything-generalizable-principle","title":"RewardAnything: Generalizable Principle-Following Reward Models","date":"2025-06-04","arxiv_id":"2506.03637","repositories_listed":1,"syntology":null},{"url":"/paper/seed-coder-let-the-code-model-curate-data-for","slug":"seed-coder-let-the-code-model-curate-data-for","title":"Seed-Coder: Let the Code Model Curate Data for Itself","date":"2025-06-04","arxiv_id":"2506.03524","repositories_listed":1,"syntology":null},{"url":"/paper/think-like-a-person-before-responding-a-multi","slug":"think-like-a-person-before-responding-a-multi","title":"Think Like a Person Before Responding: A Multi-Faceted Evaluation of Persona-Guided LLMs for Countering Hate","date":"2025-06-04","arxiv_id":"2506.04043","repositories_listed":1,"syntology":null},{"url":"/paper/a-smart-multimodal-healthcare-copilot-with","slug":"a-smart-multimodal-healthcare-copilot-with","title":"A Smart Multimodal Healthcare Copilot with Powerful LLM Reasoning","date":"2025-06-03","arxiv_id":"2506.02470","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-graph-pruning-for-multi-agent","slug":"adaptive-graph-pruning-for-multi-agent","title":"Adaptive Graph Pruning for Multi-Agent Communication","date":"2025-06-03","arxiv_id":"2506.02951","repositories_listed":1,"syntology":null},{"url":"/paper/cybergym-evaluating-ai-agents-cybersecurity","slug":"cybergym-evaluating-ai-agents-cybersecurity","title":"CyberGym: Evaluating AI Agents' Cybersecurity Capabilities with Real-World Vulnerabilities at Scale","date":"2025-06-03","arxiv_id":"2506.02548","repositories_listed":1,"syntology":null},{"url":"/paper/agentcpm-gui-building-mobile-use-agents-with","slug":"agentcpm-gui-building-mobile-use-agents-with","title":"AgentCPM-GUI: Building Mobile-Use Agents with Reinforcement Fine-Tuning","date":"2025-06-02","arxiv_id":"2506.01391","repositories_listed":1,"syntology":null},{"url":"/paper/compiler-optimization-via-llm-reasoning-for","slug":"compiler-optimization-via-llm-reasoning-for","title":"Compiler Optimization via LLM Reasoning for Efficient Model Serving","date":"2025-06-02","arxiv_id":"2506.01374","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/compiler-optimization-via-llm-reasoning-for#ran","syntology_url":"https://syntology.ai/paper/2506.01374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01374"}},"official":null}},{"url":"/paper/parameter-efficient-fine-tuning-llama-3-1-for","slug":"parameter-efficient-fine-tuning-llama-3-1-for","title":"Parameter Efficient Fine Tuning Llama 3.1 for Answering Arabic Legal Questions: A Case Study on Jordanian Laws","date":"2025-06-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-table-exploring-reinforcement","slug":"reasoning-table-exploring-reinforcement","title":"Reasoning-Table: Exploring Reinforcement Learning for Table Reasoning","date":"2025-06-02","arxiv_id":"2506.01710","repositories_listed":1,"syntology":null},{"url":"/paper/shapellm-omni-a-native-multimodal-llm-for-3d","slug":"shapellm-omni-a-native-multimodal-llm-for-3d","title":"ShapeLLM-Omni: A Native Multimodal LLM for 3D Generation and Understanding","date":"2025-06-02","arxiv_id":"2506.01853","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/shapellm-omni-a-native-multimodal-llm-for-3d#ran","syntology_url":"https://syntology.ai/paper/2506.01853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01853"}},"official":{"repos":["jamesyjl/shapellm-omni"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fusionaudio-1-2m-towards-fine-grained-audio","slug":"fusionaudio-1-2m-towards-fine-grained-audio","title":"FusionAudio-1.2M: Towards Fine-grained Audio Captioning with Multimodal Contextual Fusion","date":"2025-06-01","arxiv_id":"2506.01111","repositories_listed":1,"syntology":null},{"url":"/paper/defenderbench-a-toolkit-for-evaluating","slug":"defenderbench-a-toolkit-for-evaluating","title":"DefenderBench: A Toolkit for Evaluating Language Agents in Cybersecurity Environments","date":"2025-05-31","arxiv_id":"2506.00739","repositories_listed":1,"syntology":null},{"url":"/paper/goal-aware-identification-and-rectification","slug":"goal-aware-identification-and-rectification","title":"Goal-Aware Identification and Rectification of Misinformation in Multi-Agent Systems","date":"2025-05-31","arxiv_id":"2506.00509","repositories_listed":1,"syntology":null},{"url":"/paper/translate-with-care-addressing-gender-bias","slug":"translate-with-care-addressing-gender-bias","title":"Translate With Care: Addressing Gender Bias, Neutrality, and Reasoning in Large Language Model Translations","date":"2025-05-31","arxiv_id":"2506.00748","repositories_listed":1,"syntology":null},{"url":"/paper/fable-a-novel-data-flow-analysis-benchmark-on","slug":"fable-a-novel-data-flow-analysis-benchmark-on","title":"FABLE: A Novel Data-Flow Analysis Benchmark on Procedural Text for Large Language Model Evaluation","date":"2025-05-30","arxiv_id":"2505.24258","repositories_listed":1,"syntology":null},{"url":"/paper/geovision-labeler-zero-shot-geospatial","slug":"geovision-labeler-zero-shot-geospatial","title":"GeoVision Labeler: Zero-Shot Geospatial Classification with Vision and Language Models","date":"2025-05-30","arxiv_id":"2505.24340","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-videos-for-3d-world-enhancing","slug":"learning-from-videos-for-3d-world-enhancing","title":"Learning from Videos for 3D World: Enhancing MLLMs with 3D Vision Geometry Priors","date":"2025-05-30","arxiv_id":"2505.24625","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learning-from-videos-for-3d-world-enhancing#ran","syntology_url":"https://syntology.ai/paper/2505.24625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24625"}},"official":null}},{"url":"/paper/period-llm-extending-the-periodic-capability","slug":"period-llm-extending-the-periodic-capability","title":"Period-LLM: Extending the Periodic Capability of Multimodal Large Language Model","date":"2025-05-30","arxiv_id":"2505.24476","repositories_listed":1,"syntology":null},{"url":"/paper/sale-low-bit-estimation-for-efficient-sparse","slug":"sale-low-bit-estimation-for-efficient-sparse","title":"SALE : Low-bit Estimation for Efficient Sparse Attention in Long-context LLM Prefilling","date":"2025-05-30","arxiv_id":"2505.24179","repositories_listed":1,"syntology":null},{"url":"/paper/trident-enhancing-large-language-model-safety","slug":"trident-enhancing-large-language-model-safety","title":"TRIDENT: Enhancing Large Language Model Safety with Tri-Dimensional Diversified Red-Teaming Data Synthesis","date":"2025-05-30","arxiv_id":"2505.24672","repositories_listed":1,"syntology":null},{"url":"/paper/un-2-clip-improving-clip-s-visual-detail","slug":"un-2-clip-improving-clip-s-visual-detail","title":"un$^2$CLIP: Improving CLIP's Visual Detail Capturing Ability via Inverting unCLIP","date":"2025-05-30","arxiv_id":"2505.24517","repositories_listed":1,"syntology":null},{"url":"/paper/augment-or-not-a-comparative-study-of-pure","slug":"augment-or-not-a-comparative-study-of-pure","title":"Augment or Not? A Comparative Study of Pure and Augmented Large Language Model Recommenders","date":"2025-05-29","arxiv_id":"2505.23053","repositories_listed":1,"syntology":null},{"url":"/paper/bioreason-incentivizing-multimodal-biological","slug":"bioreason-incentivizing-multimodal-biological","title":"BioReason: Incentivizing Multimodal Biological Reasoning within a DNA-LLM Model","date":"2025-05-29","arxiv_id":"2505.23579","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bioreason-incentivizing-multimodal-biological#ran","syntology_url":"https://syntology.ai/paper/2505.23579","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23579"}},"official":{"repos":["bowang-lab/bioreason"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cdr-agent-intelligent-selection-and-execution","slug":"cdr-agent-intelligent-selection-and-execution","title":"CDR-Agent: Intelligent Selection and Execution of Clinical Decision Rules Using Large Language Model Agents","date":"2025-05-29","arxiv_id":"2505.23055","repositories_listed":1,"syntology":null},{"url":"/paper/flat-llm-fine-grained-low-rank-activation","slug":"flat-llm-fine-grained-low-rank-activation","title":"FLAT-LLM: Fine-grained Low-rank Activation Space Transformation for Large Language Model Compression","date":"2025-05-29","arxiv_id":"2505.23966","repositories_listed":1,"syntology":null},{"url":"/paper/ml-agent-reinforcing-llm-agents-for","slug":"ml-agent-reinforcing-llm-agents-for","title":"ML-Agent: Reinforcing LLM Agents for Autonomous Machine Learning Engineering","date":"2025-05-29","arxiv_id":"2505.23723","repositories_listed":1,"syntology":null},{"url":"/paper/on-policy-rl-with-optimal-reward-baseline","slug":"on-policy-rl-with-optimal-reward-baseline","title":"On-Policy RL with Optimal Reward Baseline","date":"2025-05-29","arxiv_id":"2505.23585","repositories_listed":1,"syntology":null},{"url":"/paper/owl-optimized-workforce-learning-for-general","slug":"owl-optimized-workforce-learning-for-general","title":"OWL: Optimized Workforce Learning for General Multi-Agent Assistance in Real-World Task Automation","date":"2025-05-29","arxiv_id":"2505.23885","repositories_listed":1,"syntology":null},{"url":"/paper/safescientist-toward-risk-aware-scientific","slug":"safescientist-toward-risk-aware-scientific","title":"SafeScientist: Toward Risk-Aware Scientific Discoveries by LLM Agents","date":"2025-05-29","arxiv_id":"2505.23559","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safescientist-toward-risk-aware-scientific#ran","syntology_url":"https://syntology.ai/paper/2505.23559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23559"}},"official":{"repos":["ulab-uiuc/safescientist"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-the-information-propagation","slug":"understanding-the-information-propagation","title":"Understanding the Information Propagation Effects of Communication Topologies in LLM-based Multi-Agent Systems","date":"2025-05-29","arxiv_id":"2505.23352","repositories_listed":1,"syntology":null},{"url":"/paper/vcapsbench-a-large-scale-fine-grained","slug":"vcapsbench-a-large-scale-fine-grained","title":"VCapsBench: A Large-scale Fine-grained Benchmark for Video Caption Quality Evaluation","date":"2025-05-29","arxiv_id":"2505.23484","repositories_listed":1,"syntology":null},{"url":"/paper/a-tool-for-generating-exceptional-behavior","slug":"a-tool-for-generating-exceptional-behavior","title":"A Tool for Generating Exceptional Behavior Tests With Large Language Models","date":"2025-05-28","arxiv_id":"2505.22818","repositories_listed":1,"syntology":null},{"url":"/paper/cadrille-multi-modal-cad-reconstruction-with","slug":"cadrille-multi-modal-cad-reconstruction-with","title":"cadrille: Multi-modal CAD Reconstruction with Online Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.22914","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/cadrille-multi-modal-cad-reconstruction-with#ran","syntology_url":"https://syntology.ai/paper/2505.22914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22914"}},"official":null}},{"url":"/paper/chatcfd-an-end-to-end-cfd-agent-with-domain","slug":"chatcfd-an-end-to-end-cfd-agent-with-domain","title":"ChatCFD: an End-to-End CFD Agent with Domain-specific Structured Thinking","date":"2025-05-28","arxiv_id":"2506.02019","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-rag-sub-dimensional-retrieval","slug":"cross-modal-rag-sub-dimensional-retrieval","title":"Cross-modal RAG: Sub-dimensional Retrieval-Augmented Text-to-Image Generation","date":"2025-05-28","arxiv_id":"2505.21956","repositories_listed":1,"syntology":null},{"url":"/paper/gatenlp-at-semeval-2025-task-10-hierarchical","slug":"gatenlp-at-semeval-2025-task-10-hierarchical","title":"GateNLP at SemEval-2025 Task 10: Hierarchical Three-Step Prompting for Multilingual Narrative Classification","date":"2025-05-28","arxiv_id":"2505.22867","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-vision-encoder-grafting-via-llm","slug":"zero-shot-vision-encoder-grafting-via-llm","title":"Zero-Shot Vision Encoder Grafting via LLM Surrogates","date":"2025-05-28","arxiv_id":"2505.22664","repositories_listed":1,"syntology":null},{"url":"/paper/automated-privacy-information-annotation-in","slug":"automated-privacy-information-annotation-in","title":"Automated Privacy Information Annotation in Large Language Model Interactions","date":"2025-05-27","arxiv_id":"2505.20910","repositories_listed":1,"syntology":null},{"url":"/paper/cognibench-a-legal-inspired-framework-and","slug":"cognibench-a-legal-inspired-framework-and","title":"CogniBench: A Legal-inspired Framework and Dataset for Assessing Cognitive Faithfulness of Large Language Models","date":"2025-05-27","arxiv_id":"2505.20767","repositories_listed":1,"syntology":null},{"url":"/paper/cross-from-left-to-right-brain-adaptive-text","slug":"cross-from-left-to-right-brain-adaptive-text","title":"Cross from Left to Right Brain: Adaptive Text Dreamer for Vision-and-Language Navigation","date":"2025-05-27","arxiv_id":"2505.20897","repositories_listed":1,"syntology":null},{"url":"/paper/geollava-8k-scaling-remote-sensing-multimodal","slug":"geollava-8k-scaling-remote-sensing-multimodal","title":"GeoLLaVA-8K: Scaling Remote-Sensing Multimodal Large Language Models to 8K Resolution","date":"2025-05-27","arxiv_id":"2505.21375","repositories_listed":1,"syntology":null},{"url":"/paper/let-me-think-a-long-chain-of-thought-can-be","slug":"let-me-think-a-long-chain-of-thought-can-be","title":"Let Me Think! A Long Chain-of-Thought Can Be Worth Exponentially Many Short Ones","date":"2025-05-27","arxiv_id":"2505.21825","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/let-me-think-a-long-chain-of-thought-can-be#ran","syntology_url":"https://syntology.ai/paper/2505.21825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21825"}},"official":{"repos":["seyedparsa/let-me-think"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/real-prover-retrieval-augmented-lean-prover","slug":"real-prover-retrieval-augmented-lean-prover","title":"REAL-Prover: Retrieval Augmented Lean Prover for Mathematical Reasoning","date":"2025-05-27","arxiv_id":"2505.20613","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-llm-guided-semantic-correction-in","slug":"multimodal-llm-guided-semantic-correction-in","title":"Multimodal LLM-Guided Semantic Correction in Text-to-Image Diffusion","date":"2025-05-26","arxiv_id":"2505.20053","repositories_listed":1,"syntology":null},{"url":"/paper/neusym-rag-hybrid-neural-symbolic-retrieval","slug":"neusym-rag-hybrid-neural-symbolic-retrieval","title":"NeuSym-RAG: Hybrid Neural Symbolic Retrieval with Multiview Structuring for PDF Question Answering","date":"2025-05-26","arxiv_id":"2505.19754","repositories_listed":1,"syntology":null},{"url":"/paper/rearank-reasoning-re-ranking-agent-via","slug":"rearank-reasoning-re-ranking-agent-via","title":"REARANK: Reasoning Re-ranking Agent via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.20046","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/rearank-reasoning-re-ranking-agent-via#ran","syntology_url":"https://syntology.ai/paper/2505.20046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20046"}},"official":{"repos":["lezhang7/rearank"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-multimodal-large-language-model","slug":"unifying-multimodal-large-language-model","title":"Unifying Multimodal Large Language Model Capabilities and Modalities via Model Merging","date":"2025-05-26","arxiv_id":"2505.19892","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2505.19892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19892"}},"official":{"repos":["walkerworldpeace/mllmerging"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wina-weight-informed-neuron-activation-for","slug":"wina-weight-informed-neuron-activation-for","title":"WINA: Weight Informed Neuron Activation for Accelerating Large Language Model Inference","date":"2025-05-26","arxiv_id":"2505.19427","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wina-weight-informed-neuron-activation-for#ran","syntology_url":"https://syntology.ai/paper/2505.19427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19427"}},"official":{"repos":["microsoft/wina"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-grafting-of-large-language-models","slug":"knowledge-grafting-of-large-language-models","title":"Knowledge Grafting of Large Language Models","date":"2025-05-24","arxiv_id":"2505.18502","repositories_listed":1,"syntology":null},{"url":"/paper/llm-qfl-distilling-large-language-model-for","slug":"llm-qfl-distilling-large-language-model-for","title":"LLM-QFL: Distilling Large Language Model for Quantum Federated Learning","date":"2025-05-24","arxiv_id":"2505.18656","repositories_listed":1,"syntology":null},{"url":"/paper/tulun-transparent-and-adaptable-low-resource","slug":"tulun-transparent-and-adaptable-low-resource","title":"TULUN: Transparent and Adaptable Low-resource Machine Translation","date":"2025-05-24","arxiv_id":"2505.18683","repositories_listed":1,"syntology":null},{"url":"/paper/vision-meets-language-a-rag-augmented-yolov8","slug":"vision-meets-language-a-rag-augmented-yolov8","title":"Vision Meets Language: A RAG-Augmented YOLOv8 Framework for Coffee Disease Diagnosis and Farmer Assistance","date":"2025-05-24","arxiv_id":"2505.21544","repositories_listed":1,"syntology":null},{"url":"/paper/runaway-is-ashamed-but-helpful-on-the-early","slug":"runaway-is-ashamed-but-helpful-on-the-early","title":"Runaway is Ashamed, But Helpful: On the Early-Exit Behavior of Large Language Model-based Agents in Embodied Environments","date":"2025-05-23","arxiv_id":"2505.17616","repositories_listed":1,"syntology":null},{"url":"/paper/think-or-not-exploring-thinking-efficiency-in","slug":"think-or-not-exploring-thinking-efficiency-in","title":"Think or Not? Exploring Thinking Efficiency in Large Reasoning Models via an Information-Theoretic Lens","date":"2025-05-23","arxiv_id":"2505.18237","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-evaluation-of-contemporary-ml","slug":"a-comprehensive-evaluation-of-contemporary-ml","title":"A Comprehensive Evaluation of Contemporary ML-Based Solvers for Combinatorial Optimization","date":"2025-05-22","arxiv_id":"2505.16952","repositories_listed":1,"syntology":null},{"url":"/paper/adams-momentum-itself-can-be-a-normalizer-for","slug":"adams-momentum-itself-can-be-a-normalizer-for","title":"AdamS: Momentum Itself Can Be A Normalizer for LLM Pretraining and Post-training","date":"2025-05-22","arxiv_id":"2505.16363","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adams-momentum-itself-can-be-a-normalizer-for#ran","syntology_url":"https://syntology.ai/paper/2505.16363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16363"}},"official":{"repos":["pku-huzhang/AdamS"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/castillo-characterizing-response-length","slug":"castillo-characterizing-response-length","title":"CASTILLO: Characterizing Response Length Distributions of Large Language Models","date":"2025-05-22","arxiv_id":"2505.16881","repositories_listed":1,"syntology":null},{"url":"/paper/chemmllm-chemical-multimodal-large-language","slug":"chemmllm-chemical-multimodal-large-language","title":"ChemMLLM: Chemical Multimodal Large Language Model","date":"2025-05-22","arxiv_id":"2505.16326","repositories_listed":1,"syntology":null},{"url":"/paper/conciserl-conciseness-guided-reinforcement","slug":"conciserl-conciseness-guided-reinforcement","title":"ConciseRL: Conciseness-Guided Reinforcement Learning for Efficient Reasoning Models","date":"2025-05-22","arxiv_id":"2505.17250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conciserl-conciseness-guided-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2505.17250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17250"}},"official":{"repos":["razvandu/conciserl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dimple-discrete-diffusion-multimodal-large","slug":"dimple-discrete-diffusion-multimodal-large","title":"Dimple: Discrete Diffusion Multimodal Large Language Model with Parallel Decoding","date":"2025-05-22","arxiv_id":"2505.16990","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dimple-discrete-diffusion-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2505.16990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16990"}},"official":{"repos":["yu-rp/dimple"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/emulate-a-multi-agent-framework-for","slug":"emulate-a-multi-agent-framework-for","title":"EMULATE: A Multi-Agent Framework for Determining the Veracity of Atomic Claims by Emulating Human Actions","date":"2025-05-22","arxiv_id":"2505.16576","repositories_listed":1,"syntology":null},{"url":"/paper/how-do-scaling-laws-apply-to-knowledge-graph","slug":"how-do-scaling-laws-apply-to-knowledge-graph","title":"How do Scaling Laws Apply to Knowledge Graph Engineering Tasks? The Impact of Model Size on Large Language Model Performance","date":"2025-05-22","arxiv_id":"2505.16276","repositories_listed":1,"syntology":null},{"url":"/paper/power-law-decay-loss-for-large-language-model","slug":"power-law-decay-loss-for-large-language-model","title":"Power-Law Decay Loss for Large Language Model Finetuning: Focusing on Information Sparsity to Enhance Generation Quality","date":"2025-05-22","arxiv_id":"2505.16900","repositories_listed":1,"syntology":null},{"url":"/paper/autodata-a-multi-agent-system-for-open-web","slug":"autodata-a-multi-agent-system-for-open-web","title":"AutoData: A Multi-Agent System for Open Web Data Collection","date":"2025-05-21","arxiv_id":"2505.15859","repositories_listed":1,"syntology":null},{"url":"/paper/clicksight-interpreting-student-clickstreams","slug":"clicksight-interpreting-student-clickstreams","title":"ClickSight: Interpreting Student Clickstreams to Reveal Insights on Learning Strategies via LLMs","date":"2025-05-21","arxiv_id":"2505.15410","repositories_listed":1,"syntology":null},{"url":"/paper/craken-cybersecurity-llm-agent-with-knowledge","slug":"craken-cybersecurity-llm-agent-with-knowledge","title":"CRAKEN: Cybersecurity LLM Agent with Knowledge-Based Execution","date":"2025-05-21","arxiv_id":"2505.17107","repositories_listed":1,"syntology":null},{"url":"/paper/how-memory-management-impacts-llm-agents-an","slug":"how-memory-management-impacts-llm-agents-an","title":"How Memory Management Impacts LLM Agents: An Empirical Study of Experience-Following Behavior","date":"2025-05-21","arxiv_id":"2505.16067","repositories_listed":1,"syntology":null},{"url":"/paper/keep-security-benchmarking-security-policy","slug":"keep-security-benchmarking-security-policy","title":"Keep Security! Benchmarking Security Policy Preservation in Large Language Model Contexts Against Indirect Attacks in Question Answering","date":"2025-05-21","arxiv_id":"2505.15805","repositories_listed":1,"syntology":null},{"url":"/paper/lmgame-bench-how-good-are-llms-at-playing","slug":"lmgame-bench-how-good-are-llms-at-playing","title":"lmgame-Bench: How Good are LLMs at Playing Games?","date":"2025-05-21","arxiv_id":"2505.15146","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lmgame-bench-how-good-are-llms-at-playing#ran","syntology_url":"https://syntology.ai/paper/2505.15146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15146"}},"official":{"repos":["lmgame-org/gamingagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lost-in-benchmarks-rethinking-large-language","slug":"lost-in-benchmarks-rethinking-large-language","title":"Lost in Benchmarks? Rethinking Large Language Model Benchmarking with Item Response Theory","date":"2025-05-21","arxiv_id":"2505.15055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lost-in-benchmarks-rethinking-large-language#ran","syntology_url":"https://syntology.ai/paper/2505.15055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15055"}},"official":{"repos":["Joe-Hall-Lee/PSN-IRT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lyaplock-bounded-knowledge-preservation-in","slug":"lyaplock-bounded-knowledge-preservation-in","title":"LyapLock: Bounded Knowledge Preservation in Sequential Large Language Model Editing","date":"2025-05-21","arxiv_id":"2505.15702","repositories_listed":1,"syntology":null},{"url":"/paper/piflow-principle-aware-scientific-discovery","slug":"piflow-principle-aware-scientific-discovery","title":"PiFlow: Principle-aware Scientific Discovery with Multi-Agent Collaboration","date":"2025-05-21","arxiv_id":"2505.15047","repositories_listed":1,"syntology":null},{"url":"/paper/privacy-preserving-conformal-prediction-under","slug":"privacy-preserving-conformal-prediction-under","title":"Privacy-Preserving Conformal Prediction Under Local Differential Privacy","date":"2025-05-21","arxiv_id":"2505.15721","repositories_listed":1,"syntology":null},{"url":"/paper/web-shepherd-advancing-prms-for-reinforcing","slug":"web-shepherd-advancing-prms-for-reinforcing","title":"Web-Shepherd: Advancing PRMs for Reinforcing Web Agents","date":"2025-05-21","arxiv_id":"2505.15277","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/web-shepherd-advancing-prms-for-reinforcing#ran","syntology_url":"https://syntology.ai/paper/2505.15277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15277"}},"official":{"repos":["kyle8581/Web-Shepherd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/x-webagentbench-a-multilingual-interactive","slug":"x-webagentbench-a-multilingual-interactive","title":"X-WebAgentBench: A Multilingual Interactive Web Benchmark for Evaluating Global Agentic System","date":"2025-05-21","arxiv_id":"2505.15372","repositories_listed":1,"syntology":null},{"url":"/paper/bar-a-backward-reasoning-based-agent-for","slug":"bar-a-backward-reasoning-based-agent-for","title":"BAR: A Backward Reasoning based Agent for Complex Minecraft Tasks","date":"2025-05-20","arxiv_id":"2505.14079","repositories_listed":1,"syntology":null},{"url":"/paper/guarded-query-routing-for-large-language","slug":"guarded-query-routing-for-large-language","title":"Guarded Query Routing for Large Language Models","date":"2025-05-20","arxiv_id":"2505.14524","repositories_listed":1,"syntology":null},{"url":"/paper/polar-sparsity-high-throughput-batched-llm","slug":"polar-sparsity-high-throughput-batched-llm","title":"Polar Sparsity: High Throughput Batched LLM Inferencing with Scalable Contextual Sparsity","date":"2025-05-20","arxiv_id":"2505.14884","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polar-sparsity-high-throughput-batched-llm#ran","syntology_url":"https://syntology.ai/paper/2505.14884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14884"}},"official":{"repos":["susavlsh10/polar-sparsity"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"d0d3b760e19fe479868263ddc3b4ac6c6fe76070326e4fb8c805e8a9fd069e7e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}