{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modelling/papers/14","list_of":"/task/language-modelling","task":"Language Modelling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":14,"pages_in_order":177,"rows_per_page":100,"rows":[1301,1400],"of":17610,"counts":{"archive_papers_tagged":17610,"with_a_code_link":7012,"where_syntology_ran_a_sample":2428,"not_listed_spam_title":0,"listed":17610,"listed_where_code_ran":2428,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2027,"every_run_a_failure_of_syntologys_instrument":401,"listed_with_a_run_with_no_instrument_failure":2027,"listed_every_run_a_failure_of_syntologys_instrument":401,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modelling","prev":"/task/language-modelling/papers/13","next":"/task/language-modelling/papers/15","papers":[{"url":"/paper/augment-or-not-a-comparative-study-of-pure","slug":"augment-or-not-a-comparative-study-of-pure","title":"Augment or Not? A Comparative Study of Pure and Augmented Large Language Model Recommenders","date":"2025-05-29","arxiv_id":"2505.23053","repositories_listed":1,"syntology":null},{"url":"/paper/cdr-agent-intelligent-selection-and-execution","slug":"cdr-agent-intelligent-selection-and-execution","title":"CDR-Agent: Intelligent Selection and Execution of Clinical Decision Rules Using Large Language Model Agents","date":"2025-05-29","arxiv_id":"2505.23055","repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-policy-optimization-for-token","slug":"discriminative-policy-optimization-for-token","title":"Discriminative Policy Optimization for Token-Level Reward Models","date":"2025-05-29","arxiv_id":"2505.23363","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/discriminative-policy-optimization-for-token#ran","syntology_url":"https://syntology.ai/paper/2505.23363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23363"}},"official":{"repos":["homzer/q-rm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/flat-llm-fine-grained-low-rank-activation","slug":"flat-llm-fine-grained-low-rank-activation","title":"FLAT-LLM: Fine-grained Low-rank Activation Space Transformation for Large Language Model Compression","date":"2025-05-29","arxiv_id":"2505.23966","repositories_listed":1,"syntology":null},{"url":"/paper/learning-parametric-distributions-from","slug":"learning-parametric-distributions-from","title":"Learning Parametric Distributions from Samples and Preferences","date":"2025-05-29","arxiv_id":"2505.23557","repositories_listed":1,"syntology":null},{"url":"/paper/spoken-language-modeling-with-duration","slug":"spoken-language-modeling-with-duration","title":"Spoken Language Modeling with Duration-Penalized Self-Supervised Units","date":"2025-05-29","arxiv_id":"2505.23494","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-the-information-propagation","slug":"understanding-the-information-propagation","title":"Understanding the Information Propagation Effects of Communication Topologies in LLM-based Multi-Agent Systems","date":"2025-05-29","arxiv_id":"2505.23352","repositories_listed":1,"syntology":null},{"url":"/paper/uni-mumer-unified-multi-task-fine-tuning-of","slug":"uni-mumer-unified-multi-task-fine-tuning-of","title":"Uni-MuMER: Unified Multi-Task Fine-Tuning of Vision-Language Model for Handwritten Mathematical Expression Recognition","date":"2025-05-29","arxiv_id":"2505.23566","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":1,"n_ran_checked":3,"n_instrument":4,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uni-mumer-unified-multi-task-fine-tuning-of#ran","syntology_url":"https://syntology.ai/paper/2505.23566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23566"}},"official":{"repos":["bflameswift/uni-mumer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-word-level-quality-estimation","slug":"unsupervised-word-level-quality-estimation","title":"Unsupervised Word-level Quality Estimation for Machine Translation Through the Lens of Annotators (Dis)agreement","date":"2025-05-29","arxiv_id":"2505.23183","repositories_listed":1,"syntology":null},{"url":"/paper/vcapsbench-a-large-scale-fine-grained","slug":"vcapsbench-a-large-scale-fine-grained","title":"VCapsBench: A Large-scale Fine-grained Benchmark for Video Caption Quality Evaluation","date":"2025-05-29","arxiv_id":"2505.23484","repositories_listed":1,"syntology":null},{"url":"/paper/a-tool-for-generating-exceptional-behavior","slug":"a-tool-for-generating-exceptional-behavior","title":"A Tool for Generating Exceptional Behavior Tests With Large Language Models","date":"2025-05-28","arxiv_id":"2505.22818","repositories_listed":1,"syntology":null},{"url":"/paper/cfp-gen-combinatorial-functional-protein","slug":"cfp-gen-combinatorial-functional-protein","title":"CFP-Gen: Combinatorial Functional Protein Generation via Diffusion Language Models","date":"2025-05-28","arxiv_id":"2505.22869","repositories_listed":1,"syntology":null},{"url":"/paper/chatcfd-an-end-to-end-cfd-agent-with-domain","slug":"chatcfd-an-end-to-end-cfd-agent-with-domain","title":"ChatCFD: an End-to-End CFD Agent with Domain-specific Structured Thinking","date":"2025-05-28","arxiv_id":"2506.02019","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-rag-sub-dimensional-retrieval","slug":"cross-modal-rag-sub-dimensional-retrieval","title":"Cross-modal RAG: Sub-dimensional Retrieval-Augmented Text-to-Image Generation","date":"2025-05-28","arxiv_id":"2505.21956","repositories_listed":1,"syntology":null},{"url":"/paper/gatenlp-at-semeval-2025-task-10-hierarchical","slug":"gatenlp-at-semeval-2025-task-10-hierarchical","title":"GateNLP at SemEval-2025 Task 10: Hierarchical Three-Step Prompting for Multilingual Narrative Classification","date":"2025-05-28","arxiv_id":"2505.22867","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-vision-encoder-grafting-via-llm","slug":"zero-shot-vision-encoder-grafting-via-llm","title":"Zero-Shot Vision Encoder Grafting via LLM Surrogates","date":"2025-05-28","arxiv_id":"2505.22664","repositories_listed":1,"syntology":null},{"url":"/paper/automated-privacy-information-annotation-in","slug":"automated-privacy-information-annotation-in","title":"Automated Privacy Information Annotation in Large Language Model Interactions","date":"2025-05-27","arxiv_id":"2505.20910","repositories_listed":1,"syntology":null},{"url":"/paper/cognibench-a-legal-inspired-framework-and","slug":"cognibench-a-legal-inspired-framework-and","title":"CogniBench: A Legal-inspired Framework and Dataset for Assessing Cognitive Faithfulness of Large Language Models","date":"2025-05-27","arxiv_id":"2505.20767","repositories_listed":1,"syntology":null},{"url":"/paper/improved-representation-steering-for-language","slug":"improved-representation-steering-for-language","title":"Improved Representation Steering for Language Models","date":"2025-05-27","arxiv_id":"2505.20809","repositories_listed":1,"syntology":null},{"url":"/paper/let-me-think-a-long-chain-of-thought-can-be","slug":"let-me-think-a-long-chain-of-thought-can-be","title":"Let Me Think! A Long Chain-of-Thought Can Be Worth Exponentially Many Short Ones","date":"2025-05-27","arxiv_id":"2505.21825","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/let-me-think-a-long-chain-of-thought-can-be#ran","syntology_url":"https://syntology.ai/paper/2505.21825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21825"}},"official":{"repos":["seyedparsa/let-me-think"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pretraining-language-models-to-ponder-in","slug":"pretraining-language-models-to-ponder-in","title":"Pretraining Language Models to Ponder in Continuous Space","date":"2025-05-27","arxiv_id":"2505.20674","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pretraining-language-models-to-ponder-in#ran","syntology_url":"https://syntology.ai/paper/2505.20674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20674"}},"official":{"repos":["lumia-group/ponderinglm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/real-prover-retrieval-augmented-lean-prover","slug":"real-prover-retrieval-augmented-lean-prover","title":"REAL-Prover: Retrieval Augmented Lean Prover for Mathematical Reasoning","date":"2025-05-27","arxiv_id":"2505.20613","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-classifier-free-guidance-via-dynamic","slug":"adaptive-classifier-free-guidance-via-dynamic","title":"Adaptive Classifier-Free Guidance via Dynamic Low-Confidence Masking","date":"2025-05-26","arxiv_id":"2505.20199","repositories_listed":1,"syntology":null},{"url":"/paper/can-compressed-llms-truly-act-an-empirical","slug":"can-compressed-llms-truly-act-an-empirical","title":"Can Compressed LLMs Truly Act? An Empirical Evaluation of Agentic Capabilities in LLM Compression","date":"2025-05-26","arxiv_id":"2505.19433","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-compressed-llms-truly-act-an-empirical#ran","syntology_url":"https://syntology.ai/paper/2505.19433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19433"}},"official":{"repos":["pprp/acbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/causal-llava-causal-disentanglement-for","slug":"causal-llava-causal-disentanglement-for","title":"Causal-LLaVA: Causal Disentanglement for Mitigating Hallucination in Multimodal Large Language Models","date":"2025-05-26","arxiv_id":"2505.19474","repositories_listed":1,"syntology":{"n":19,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":11,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/causal-llava-causal-disentanglement-for#ran","syntology_url":"https://syntology.ai/paper/2505.19474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19474"}},"official":{"repos":["ignisavium/causal-llava"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":11,"ran_from_kinds":["official"]}}},{"url":"/paper/rearank-reasoning-re-ranking-agent-via","slug":"rearank-reasoning-re-ranking-agent-via","title":"REARANK: Reasoning Re-ranking Agent via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.20046","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/rearank-reasoning-re-ranking-agent-via#ran","syntology_url":"https://syntology.ai/paper/2505.20046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20046"}},"official":{"repos":["lezhang7/rearank"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/style2code-a-style-controllable-code","slug":"style2code-a-style-controllable-code","title":"Style2Code: A Style-Controllable Code Generation Framework with Dual-Modal Contrastive Representation Learning","date":"2025-05-26","arxiv_id":"2505.19442","repositories_listed":1,"syntology":null},{"url":"/paper/trojanstego-your-language-model-can-secretly","slug":"trojanstego-your-language-model-can-secretly","title":"TrojanStego: Your Language Model Can Secretly Be A Steganographic Privacy Leaking Agent","date":"2025-05-26","arxiv_id":"2505.20118","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-multimodal-large-language-model","slug":"unifying-multimodal-large-language-model","title":"Unifying Multimodal Large Language Model Capabilities and Modalities via Model Merging","date":"2025-05-26","arxiv_id":"2505.19892","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2505.19892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19892"}},"official":{"repos":["walkerworldpeace/mllmerging"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voicestar-robust-zero-shot-autoregressive-tts","slug":"voicestar-robust-zero-shot-autoregressive-tts","title":"VoiceStar: Robust Zero-Shot Autoregressive TTS with Duration Control and Extrapolation","date":"2025-05-26","arxiv_id":"2505.19462","repositories_listed":1,"syntology":null},{"url":"/paper/vscbench-bridging-the-gap-in-vision-language","slug":"vscbench-bridging-the-gap-in-vision-language","title":"VSCBench: Bridging the Gap in Vision-Language Model Safety Calibration","date":"2025-05-26","arxiv_id":"2505.20362","repositories_listed":1,"syntology":null},{"url":"/paper/wina-weight-informed-neuron-activation-for","slug":"wina-weight-informed-neuron-activation-for","title":"WINA: Weight Informed Neuron Activation for Accelerating Large Language Model Inference","date":"2025-05-26","arxiv_id":"2505.19427","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wina-weight-informed-neuron-activation-for#ran","syntology_url":"https://syntology.ai/paper/2505.19427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19427"}},"official":{"repos":["microsoft/wina"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/screenexplorer-training-a-vision-language","slug":"screenexplorer-training-a-vision-language","title":"ScreenExplorer: Training a Vision-Language Model for Diverse Exploration in Open GUI World","date":"2025-05-25","arxiv_id":"2505.19095","repositories_listed":1,"syntology":null},{"url":"/paper/llm-qfl-distilling-large-language-model-for","slug":"llm-qfl-distilling-large-language-model-for","title":"LLM-QFL: Distilling Large Language Model for Quantum Federated Learning","date":"2025-05-24","arxiv_id":"2505.18656","repositories_listed":1,"syntology":null},{"url":"/paper/partition-generative-modeling-masked-modeling","slug":"partition-generative-modeling-masked-modeling","title":"Partition Generative Modeling: Masked Modeling Without Masks","date":"2025-05-24","arxiv_id":"2505.18883","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/partition-generative-modeling-masked-modeling#ran","syntology_url":"https://syntology.ai/paper/2505.18883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18883"}},"official":{"repos":["kuleshov-group/mdlm"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tulun-transparent-and-adaptable-low-resource","slug":"tulun-transparent-and-adaptable-low-resource","title":"TULUN: Transparent and Adaptable Low-resource Machine Translation","date":"2025-05-24","arxiv_id":"2505.18683","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-prompt-engineering-robust-behavior","slug":"beyond-prompt-engineering-robust-behavior","title":"Beyond Prompt Engineering: Robust Behavior Control in LLMs via Steering Target Atoms","date":"2025-05-23","arxiv_id":"2505.20322","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-prompt-engineering-robust-behavior#ran","syntology_url":"https://syntology.ai/paper/2505.20322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20322"}},"official":{"repos":["zjunlp/steer-target-atoms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/daily-omni-towards-audio-visual-reasoning","slug":"daily-omni-towards-audio-visual-reasoning","title":"Daily-Omni: Towards Audio-Visual Reasoning with Temporal Alignment across Modalities","date":"2025-05-23","arxiv_id":"2505.17862","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/daily-omni-towards-audio-visual-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.17862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17862"}},"official":{"repos":["lliar-liar/daily-omni"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/danmakutppbench-a-multi-modal-benchmark-for","slug":"danmakutppbench-a-multi-modal-benchmark-for","title":"DanmakuTPPBench: A Multi-modal Benchmark for Temporal Point Process Modeling and Understanding","date":"2025-05-23","arxiv_id":"2505.18411","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/danmakutppbench-a-multi-modal-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2505.18411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18411"}},"official":{"repos":["frenkie-chiang/danmakutppbench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/decoupled-visual-interpretation-and","slug":"decoupled-visual-interpretation-and","title":"Decoupled Visual Interpretation and Linguistic Reasoning for Math Problem Solving","date":"2025-05-23","arxiv_id":"2505.17609","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoupled-visual-interpretation-and#ran","syntology_url":"https://syntology.ai/paper/2505.17609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17609"}},"official":{"repos":["guozix/dvlr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inference-time-decomposition-of-activations","slug":"inference-time-decomposition-of-activations","title":"Inference-Time Decomposition of Activations (ITDA): A Scalable Approach to Interpreting Large Language Models","date":"2025-05-23","arxiv_id":"2505.17769","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inference-time-decomposition-of-activations#ran","syntology_url":"https://syntology.ai/paper/2505.17769","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17769"}},"official":{"repos":["pleask/itda"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/keepitsimple-at-semeval-2025-task-3-llm","slug":"keepitsimple-at-semeval-2025-task-3-llm","title":"keepitsimple at SemEval-2025 Task 3: LLM-Uncertainty based Approach for Multilingual Hallucination Span Detection","date":"2025-05-23","arxiv_id":"2505.17485","repositories_listed":1,"syntology":null},{"url":"/paper/reprompt-reasoning-augmented-reprompting-for","slug":"reprompt-reasoning-augmented-reprompting-for","title":"RePrompt: Reasoning-Augmented Reprompting for Text-to-Image Generation via Reinforcement Learning","date":"2025-05-23","arxiv_id":"2505.17540","repositories_listed":1,"syntology":null},{"url":"/paper/runaway-is-ashamed-but-helpful-on-the-early","slug":"runaway-is-ashamed-but-helpful-on-the-early","title":"Runaway is Ashamed, But Helpful: On the Early-Exit Behavior of Large Language Model-based Agents in Embodied Environments","date":"2025-05-23","arxiv_id":"2505.17616","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-up-biomedical-vision-language-models","slug":"scaling-up-biomedical-vision-language-models","title":"Scaling Up Biomedical Vision-Language Models: Fine-Tuning, Instruction Tuning, and Multi-Modal Learning","date":"2025-05-23","arxiv_id":"2505.17436","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-evaluation-of-contemporary-ml","slug":"a-comprehensive-evaluation-of-contemporary-ml","title":"A Comprehensive Evaluation of Contemporary ML-Based Solvers for Combinatorial Optimization","date":"2025-05-22","arxiv_id":"2505.16952","repositories_listed":1,"syntology":null},{"url":"/paper/a-japanese-language-model-and-three-new","slug":"a-japanese-language-model-and-three-new","title":"A Japanese Language Model and Three New Evaluation Benchmarks for Pharmaceutical NLP","date":"2025-05-22","arxiv_id":"2505.16661","repositories_listed":1,"syntology":null},{"url":"/paper/castillo-characterizing-response-length","slug":"castillo-characterizing-response-length","title":"CASTILLO: Characterizing Response Length Distributions of Large Language Models","date":"2025-05-22","arxiv_id":"2505.16881","repositories_listed":1,"syntology":null},{"url":"/paper/chemmllm-chemical-multimodal-large-language","slug":"chemmllm-chemical-multimodal-large-language","title":"ChemMLLM: Chemical Multimodal Large Language Model","date":"2025-05-22","arxiv_id":"2505.16326","repositories_listed":1,"syntology":null},{"url":"/paper/dimple-discrete-diffusion-multimodal-large","slug":"dimple-discrete-diffusion-multimodal-large","title":"Dimple: Discrete Diffusion Multimodal Large Language Model with Parallel Decoding","date":"2025-05-22","arxiv_id":"2505.16990","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dimple-discrete-diffusion-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2505.16990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16990"}},"official":{"repos":["yu-rp/dimple"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/emulate-a-multi-agent-framework-for","slug":"emulate-a-multi-agent-framework-for","title":"EMULATE: A Multi-Agent Framework for Determining the Veracity of Atomic Claims by Emulating Human Actions","date":"2025-05-22","arxiv_id":"2505.16576","repositories_listed":1,"syntology":null},{"url":"/paper/how-do-scaling-laws-apply-to-knowledge-graph","slug":"how-do-scaling-laws-apply-to-knowledge-graph","title":"How do Scaling Laws Apply to Knowledge Graph Engineering Tasks? The Impact of Model Size on Large Language Model Performance","date":"2025-05-22","arxiv_id":"2505.16276","repositories_listed":1,"syntology":null},{"url":"/paper/lavida-a-large-diffusion-language-model-for","slug":"lavida-a-large-diffusion-language-model-for","title":"LaViDa: A Large Diffusion Language Model for Multimodal Understanding","date":"2025-05-22","arxiv_id":"2505.16839","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/lavida-a-large-diffusion-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2505.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16839"}},"official":{"repos":["jacklishufan/lavida"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/power-law-decay-loss-for-large-language-model","slug":"power-law-decay-loss-for-large-language-model","title":"Power-Law Decay Loss for Large Language Model Finetuning: Focusing on Information Sparsity to Enhance Generation Quality","date":"2025-05-22","arxiv_id":"2505.16900","repositories_listed":1,"syntology":null},{"url":"/paper/saturn-sat-based-reinforcement-learning-to","slug":"saturn-sat-based-reinforcement-learning-to","title":"SATURN: SAT-based Reinforcement Learning to Unleash Language Model Reasoning","date":"2025-05-22","arxiv_id":"2505.16368","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/saturn-sat-based-reinforcement-learning-to#ran","syntology_url":"https://syntology.ai/paper/2505.16368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16368"}},"official":{"repos":["gtxygyzb/saturn-code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/structure-aligned-protein-language-model","slug":"structure-aligned-protein-language-model","title":"Structure-Aligned Protein Language Model","date":"2025-05-22","arxiv_id":"2505.16896","repositories_listed":1,"syntology":null},{"url":"/paper/clicksight-interpreting-student-clickstreams","slug":"clicksight-interpreting-student-clickstreams","title":"ClickSight: Interpreting Student Clickstreams to Reveal Insights on Learning Strategies via LLMs","date":"2025-05-21","arxiv_id":"2505.15410","repositories_listed":1,"syntology":null},{"url":"/paper/diagnosing-our-datasets-how-does-my-language","slug":"diagnosing-our-datasets-how-does-my-language","title":"Diagnosing our datasets: How does my language model learn clinical information?","date":"2025-05-21","arxiv_id":"2505.15024","repositories_listed":1,"syntology":null},{"url":"/paper/human-in-the-loop-adaptive-optimization-for","slug":"human-in-the-loop-adaptive-optimization-for","title":"Human in the Loop Adaptive Optimization for Improved Time Series Forecasting","date":"2025-05-21","arxiv_id":"2505.15354","repositories_listed":1,"syntology":null},{"url":"/paper/keep-security-benchmarking-security-policy","slug":"keep-security-benchmarking-security-policy","title":"Keep Security! Benchmarking Security Policy Preservation in Large Language Model Contexts Against Indirect Attacks in Question Answering","date":"2025-05-21","arxiv_id":"2505.15805","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-online-data-to-enhance-medical","slug":"leveraging-online-data-to-enhance-medical","title":"Leveraging Online Data to Enhance Medical Knowledge in a Small Persian Language Model","date":"2025-05-21","arxiv_id":"2505.16000","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-unit-language-guidance-to-advance","slug":"leveraging-unit-language-guidance-to-advance","title":"Leveraging Unit Language Guidance to Advance Speech Modeling in Textless Speech-to-Speech Translation","date":"2025-05-21","arxiv_id":"2505.15333","repositories_listed":1,"syntology":null},{"url":"/paper/lmgame-bench-how-good-are-llms-at-playing","slug":"lmgame-bench-how-good-are-llms-at-playing","title":"lmgame-Bench: How Good are LLMs at Playing Games?","date":"2025-05-21","arxiv_id":"2505.15146","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lmgame-bench-how-good-are-llms-at-playing#ran","syntology_url":"https://syntology.ai/paper/2505.15146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15146"}},"official":{"repos":["lmgame-org/gamingagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lost-in-benchmarks-rethinking-large-language","slug":"lost-in-benchmarks-rethinking-large-language","title":"Lost in Benchmarks? Rethinking Large Language Model Benchmarking with Item Response Theory","date":"2025-05-21","arxiv_id":"2505.15055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lost-in-benchmarks-rethinking-large-language#ran","syntology_url":"https://syntology.ai/paper/2505.15055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15055"}},"official":{"repos":["Joe-Hall-Lee/PSN-IRT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lyaplock-bounded-knowledge-preservation-in","slug":"lyaplock-bounded-knowledge-preservation-in","title":"LyapLock: Bounded Knowledge Preservation in Sequential Large Language Model Editing","date":"2025-05-21","arxiv_id":"2505.15702","repositories_listed":1,"syntology":null},{"url":"/paper/viqagent-zero-shot-video-question-answering","slug":"viqagent-zero-shot-video-question-answering","title":"ViQAgent: Zero-Shot Video Question Answering via Agent with Open-Vocabulary Grounding Validation","date":"2025-05-21","arxiv_id":"2505.15928","repositories_listed":1,"syntology":null},{"url":"/paper/x-webagentbench-a-multilingual-interactive","slug":"x-webagentbench-a-multilingual-interactive","title":"X-WebAgentBench: A Multilingual Interactive Web Benchmark for Evaluating Global Agentic System","date":"2025-05-21","arxiv_id":"2505.15372","repositories_listed":1,"syntology":null},{"url":"/paper/cad-coder-an-open-source-vision-language","slug":"cad-coder-an-open-source-vision-language","title":"CAD-Coder: An Open-Source Vision-Language Model for Computer-Aided Design Code Generation","date":"2025-05-20","arxiv_id":"2505.14646","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cad-coder-an-open-source-vision-language#ran","syntology_url":"https://syntology.ai/paper/2505.14646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14646"}},"official":{"repos":["anniedoris/cad-coder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-graph-representations-of-logical","slug":"exploring-graph-representations-of-logical","title":"Exploring Graph Representations of Logical Forms for Language Modeling","date":"2025-05-20","arxiv_id":"2505.14523","repositories_listed":1,"syntology":null},{"url":"/paper/improve-language-model-and-brain-alignment","slug":"improve-language-model-and-brain-alignment","title":"Improve Language Model and Brain Alignment via Associative Memory","date":"2025-05-20","arxiv_id":"2505.13844","repositories_listed":1,"syntology":null},{"url":"/paper/multihal-multilingual-dataset-for-knowledge","slug":"multihal-multilingual-dataset-for-knowledge","title":"MultiHal: Multilingual Dataset for Knowledge-Graph Grounded Evaluation of LLM Hallucinations","date":"2025-05-20","arxiv_id":"2505.14101","repositories_listed":1,"syntology":null},{"url":"/paper/rank-k-test-time-reasoning-for-listwise","slug":"rank-k-test-time-reasoning-for-listwise","title":"Rank-K: Test-Time Reasoning for Listwise Reranking","date":"2025-05-20","arxiv_id":"2505.14432","repositories_listed":1,"syntology":null},{"url":"/paper/speculative-decoding-reimagined-for","slug":"speculative-decoding-reimagined-for","title":"Speculative Decoding Reimagined for Multimodal Large Language Models","date":"2025-05-20","arxiv_id":"2505.14260","repositories_listed":1,"syntology":null},{"url":"/paper/too-long-didn-t-model-decomposing-llm-long","slug":"too-long-didn-t-model-decomposing-llm-long","title":"Too Long, Didn't Model: Decomposing LLM Long-Context Understanding With Novels","date":"2025-05-20","arxiv_id":"2505.14925","repositories_listed":1,"syntology":null},{"url":"/paper/u-sam-an-audio-language-model-for-unified","slug":"u-sam-an-audio-language-model-for-unified","title":"U-SAM: An audio language Model for Unified Speech, Audio, and Music Understanding","date":"2025-05-20","arxiv_id":"2505.13880","repositories_listed":1,"syntology":null},{"url":"/paper/3d-visual-illusion-depth-estimation","slug":"3d-visual-illusion-depth-estimation","title":"3D Visual Illusion Depth Estimation","date":"2025-05-19","arxiv_id":"2505.13061","repositories_listed":1,"syntology":null},{"url":"/paper/cie-controlling-language-model-text","slug":"cie-controlling-language-model-text","title":"CIE: Controlling Language Model Text Generations Using Continuous Signals","date":"2025-05-19","arxiv_id":"2505.13448","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-speech-language-modeling-via-energy","slug":"efficient-speech-language-modeling-via-energy","title":"Efficient Speech Language Modeling via Energy Distance in Continuous Latent Space","date":"2025-05-19","arxiv_id":"2505.13181","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-speech-language-modeling-via-energy#ran","syntology_url":"https://syntology.ai/paper/2505.13181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13181"}},"official":{"repos":["ictnlp/sled-tts"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/g1-bootstrapping-perception-and-reasoning","slug":"g1-bootstrapping-perception-and-reasoning","title":"G1: Bootstrapping Perception and Reasoning Abilities of Vision-Language Model via Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.13426","repositories_listed":1,"syntology":null},{"url":"/paper/mindomni-unleashing-reasoning-generation-in","slug":"mindomni-unleashing-reasoning-generation-in","title":"MindOmni: Unleashing Reasoning Generation in Vision Language Models with RGPO","date":"2025-05-19","arxiv_id":"2505.13031","repositories_listed":1,"syntology":null},{"url":"/paper/r3-robust-rubric-agnostic-reward-models","slug":"r3-robust-rubric-agnostic-reward-models","title":"R3: Robust Rubric-Agnostic Reward Models","date":"2025-05-19","arxiv_id":"2505.13388","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r3-robust-rubric-agnostic-reward-models#ran","syntology_url":"https://syntology.ai/paper/2505.13388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13388"}},"official":{"repos":["rubricreward/r3"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-oriented-recipe-for-transferring","slug":"temporal-oriented-recipe-for-transferring","title":"Temporal-Oriented Recipe for Transferring Large Vision-Language Model to Video Understanding","date":"2025-05-19","arxiv_id":"2505.12605","repositories_listed":1,"syntology":null},{"url":"/paper/the-traitors-deception-and-trust-in-multi","slug":"the-traitors-deception-and-trust-in-multi","title":"The Traitors: Deception and Trust in Multi-Agent Language Model Simulations","date":"2025-05-19","arxiv_id":"2505.12923","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-traitors-deception-and-trust-in-multi#ran","syntology_url":"https://syntology.ai/paper/2505.12923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12923"}},"official":{"repos":["pedrocurvo/thetraitors"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-generative-and-discriminative","slug":"bridging-generative-and-discriminative","title":"Bridging Generative and Discriminative Learning: Few-Shot Relation Extraction via Two-Stage Knowledge-Guided Pre-training","date":"2025-05-18","arxiv_id":"2505.12236","repositories_listed":1,"syntology":null},{"url":"/paper/slot-sample-specific-language-model","slug":"slot-sample-specific-language-model","title":"SLOT: Sample-specific Language Model Optimization at Test-time","date":"2025-05-18","arxiv_id":"2505.12392","repositories_listed":1,"syntology":null},{"url":"/paper/towards-ds-ner-unveiling-and-addressing","slug":"towards-ds-ner-unveiling-and-addressing","title":"Towards DS-NER: Unveiling and Addressing Latent Noise in Distant Annotations","date":"2025-05-18","arxiv_id":"2505.12454","repositories_listed":1,"syntology":null},{"url":"/paper/demystifying-and-enhancing-the-efficiency-of","slug":"demystifying-and-enhancing-the-efficiency-of","title":"Demystifying and Enhancing the Efficiency of Large Language Model Based Search Agents","date":"2025-05-17","arxiv_id":"2505.12065","repositories_listed":1,"syntology":null},{"url":"/paper/internal-causal-mechanisms-robustly-predict","slug":"internal-causal-mechanisms-robustly-predict","title":"Internal Causal Mechanisms Robustly Predict Language Model Out-of-Distribution Behaviors","date":"2025-05-17","arxiv_id":"2505.11770","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internal-causal-mechanisms-robustly-predict#ran","syntology_url":"https://syntology.ai/paper/2505.11770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11770"}},"official":{"repos":["explanare/ood-prediction"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lifelongagentbench-evaluating-llm-agents-as","slug":"lifelongagentbench-evaluating-llm-agents-as","title":"LifelongAgentBench: Evaluating LLM Agents as Lifelong Learners","date":"2025-05-17","arxiv_id":"2505.11942","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-large-language-model-errors-arise","slug":"reasoning-large-language-model-errors-arise","title":"Reasoning Large Language Model Errors Arise from Hallucinating Critical Problem Features","date":"2025-05-17","arxiv_id":"2505.12151","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10769","slug":"2505-10769","title":"Unifying Segment Anything in Microscopy with Multimodal Large Language Model","date":"2025-05-16","arxiv_id":"2505.10769","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10861","slug":"2505-10861","title":"Improving the Data-efficiency of Reinforcement Learning by Warm-starting with LLM","date":"2025-05-16","arxiv_id":"2505.10861","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2505-10861#ran","syntology_url":"https://syntology.ai/paper/2505.10861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10861"}},"official":{"repos":["duongnhatthang/llamagym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-11040","slug":"2505-11040","title":"Efficient Attention via Pre-Scoring: Prioritizing Informative Keys in Transformers","date":"2025-05-16","arxiv_id":"2505.11040","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11177","slug":"2505-11177","title":"Low-Resource Language Processing: An OCR-Driven Summarization and Translation Pipeline","date":"2025-05-16","arxiv_id":"2505.11177","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11221","slug":"2505-11221","title":"Sample Efficient Reinforcement Learning via Large Vision Language Model Distillation","date":"2025-05-16","arxiv_id":"2505.11221","repositories_listed":1,"syntology":null},{"url":"/paper/an-agentic-system-with-reinforcement-learned","slug":"an-agentic-system-with-reinforcement-learned","title":"An agentic system with reinforcement-learned subsystem improvements for parsing form-like documents","date":"2025-05-16","arxiv_id":"2505.13504","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10719","slug":"2505-10719","title":"Tracr-Injection: Distilling Algorithms into Pre-trained Language Models","date":"2025-05-15","arxiv_id":"2505.10719","repositories_listed":1,"syntology":null},{"url":"/paper/complexformer-disruptively-advancing","slug":"complexformer-disruptively-advancing","title":"ComplexFormer: Disruptively Advancing Transformer Inference Ability via Head-Specific Complex Vector Attention","date":"2025-05-15","arxiv_id":"2505.10222","repositories_listed":1,"syntology":null},{"url":"/paper/imaginebench-evaluating-reinforcement","slug":"imaginebench-evaluating-reinforcement","title":"ImagineBench: Evaluating Reinforcement Learning with Large Language Model Rollouts","date":"2025-05-15","arxiv_id":"2505.10010","repositories_listed":1,"syntology":null},{"url":"/paper/multi-token-prediction-needs-registers","slug":"multi-token-prediction-needs-registers","title":"Multi-Token Prediction Needs Registers","date":"2025-05-15","arxiv_id":"2505.10518","repositories_listed":1,"syntology":{"n":20,"n_ran":18,"n_constructed":0,"n_ran_checked":11,"n_instrument":7,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 7 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-token-prediction-needs-registers#ran","syntology_url":"https://syntology.ai/paper/2505.10518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10518"}},"official":{"repos":["nasosger/mutor"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"f0b37307628c4b05e47ce7235d1ec53235619e345dc7a7306cd1df578d979efd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}