{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/large-language-model/papers/ran/1","list_of":"/task/large-language-model","task":"Large Language Model","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":9,"rows_per_page":100,"rows":[1,100],"of":801,"counts":{"archive_papers_tagged":6097,"with_a_code_link":2250,"where_syntology_ran_a_sample":801,"not_listed_spam_title":0,"listed":6097,"listed_where_code_ran":801,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":683,"every_run_a_failure_of_syntologys_instrument":118,"listed_with_a_run_with_no_instrument_failure":683,"listed_every_run_a_failure_of_syntologys_instrument":118,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/large-language-model/papers/ran/1","prev":null,"next":"/task/large-language-model/papers/ran/2","papers":[{"url":"/paper/seq-vs-seq-an-open-suite-of-paired-encoders","slug":"seq-vs-seq-an-open-suite-of-paired-encoders","title":"Seq vs Seq: An Open Suite of Paired Encoders and Decoders","date":"2025-07-15","arxiv_id":"2507.11412","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/seq-vs-seq-an-open-suite-of-paired-encoders#ran","syntology_url":"https://syntology.ai/paper/2507.11412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.11412"}},"official":{"repos":["jhu-clsp/ettin-encoder-vs-decoder"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/drafterbench-benchmarking-large-language","slug":"drafterbench-benchmarking-large-language","title":"DrafterBench: Benchmarking Large Language Models for Tasks Automation in Civil Engineering","date":"2025-07-15","arxiv_id":"2507.11527","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/drafterbench-benchmarking-large-language#ran","syntology_url":"https://syntology.ai/paper/2507.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.11527"}},"official":{"repos":["eason-li-ais/drafterbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/open-source-planning-control-system-with","slug":"open-source-planning-control-system-with","title":"Open Source Planning & Control System with Language Agents for Autonomous Scientific Discovery","date":"2025-07-09","arxiv_id":"2507.07257","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-source-planning-control-system-with#ran","syntology_url":"https://syntology.ai/paper/2507.07257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.07257"}},"official":{"repos":["cmbagents/cmbagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/llava-sp-enhancing-visual-representation-with","slug":"llava-sp-enhancing-visual-representation-with","title":"LLaVA-SP: Enhancing Visual Representation with Visual Spatial Tokens for MLLMs","date":"2025-07-01","arxiv_id":"2507.00505","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-sp-enhancing-visual-representation-with#ran","syntology_url":"https://syntology.ai/paper/2507.00505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.00505"}},"official":{"repos":["cnfaker/llava-sp"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dataset-distillation-via-vision-language","slug":"dataset-distillation-via-vision-language","title":"Dataset Distillation via Vision-Language Category Prototype","date":"2025-06-30","arxiv_id":"2506.23580","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dataset-distillation-via-vision-language#ran","syntology_url":"https://syntology.ai/paper/2506.23580","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.23580"}},"official":{"repos":["Guang000/Awesome-Dataset-Distillation","zou-yawen/dataset-distillation-via-vision-language-category-prototype"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/agentstealth-reinforcing-large-language-model","slug":"agentstealth-reinforcing-large-language-model","title":"AgentStealth: Reinforcing Large Language Model for Anonymizing User-generated Text","date":"2025-06-26","arxiv_id":"2506.22508","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agentstealth-reinforcing-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2506.22508","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.22508"}},"official":{"repos":["tsinghua-fib-lab/agentstealth"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-community-driven-agents-for-machine","slug":"towards-community-driven-agents-for-machine","title":"Towards Community-Driven Agents for Machine Learning Engineering","date":"2025-06-25","arxiv_id":"2506.20640","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-community-driven-agents-for-machine#ran","syntology_url":"https://syntology.ai/paper/2506.20640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20640"}},"official":{"repos":["comind-ml/comind"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evolving-prompts-in-context-an-open-ended","slug":"evolving-prompts-in-context-an-open-ended","title":"Evolving Prompts In-Context: An Open-ended, Self-replicating Perspective","date":"2025-06-22","arxiv_id":"2506.17930","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evolving-prompts-in-context-an-open-ended#ran","syntology_url":"https://syntology.ai/paper/2506.17930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.17930"}},"official":{"repos":["jianyu-cs/promptquine"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sharegpt-4o-image-aligning-multimodal-models","slug":"sharegpt-4o-image-aligning-multimodal-models","title":"ShareGPT-4o-Image: Aligning Multimodal Models with GPT-4o-Level Image Generation","date":"2025-06-22","arxiv_id":"2506.18095","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sharegpt-4o-image-aligning-multimodal-models#ran","syntology_url":"https://syntology.ai/paper/2506.18095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.18095"}},"official":{"repos":["freedomintelligence/sharegpt-4o-image"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/lmr-bench-evaluating-llm-agent-s-ability-on","slug":"lmr-bench-evaluating-llm-agent-s-ability-on","title":"LMR-BENCH: Evaluating LLM Agent's Ability on Reproducing Language Modeling Research","date":"2025-06-19","arxiv_id":"2506.17335","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lmr-bench-evaluating-llm-agent-s-ability-on#ran","syntology_url":"https://syntology.ai/paper/2506.17335","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.17335"}},"official":{"repos":["du-nlp-lab/lmr-bench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/video-salmonn-2-captioning-enhanced-audio","slug":"video-salmonn-2-captioning-enhanced-audio","title":"video-SALMONN 2: Captioning-Enhanced Audio-Visual Large Language Models","date":"2025-06-18","arxiv_id":"2506.15220","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-salmonn-2-captioning-enhanced-audio#ran","syntology_url":"https://syntology.ai/paper/2506.15220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.15220"}},"official":{"repos":["bytedance/video-salmonn-2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ras-eval-a-comprehensive-benchmark-for","slug":"ras-eval-a-comprehensive-benchmark-for","title":"RAS-Eval: A Comprehensive Benchmark for Security Evaluation of LLM Agents in Real-World Environments","date":"2025-06-18","arxiv_id":"2506.15253","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ras-eval-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2506.15253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.15253"}},"official":{"repos":["lanzer-tree/ras-eval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vis-shepherd-constructing-critic-for-llm","slug":"vis-shepherd-constructing-critic-for-llm","title":"VIS-Shepherd: Constructing Critic for LLM-based Data Visualization Generation","date":"2025-06-16","arxiv_id":"2506.13326","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vis-shepherd-constructing-critic-for-llm#ran","syntology_url":"https://syntology.ai/paper/2506.13326","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13326"}},"official":{"repos":["bopan3/vis-shepherd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/v-jepa-2-self-supervised-video-models-enable","slug":"v-jepa-2-self-supervised-video-models-enable","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","date":"2025-06-11","arxiv_id":"2506.09985","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/v-jepa-2-self-supervised-video-models-enable#ran","syntology_url":"https://syntology.ai/paper/2506.09985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09985"}},"official":{"repos":["facebookresearch/vjepa2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2506-08762","slug":"2506-08762","title":"EDINET-Bench: Evaluating LLMs on Complex Financial Tasks using Japanese Financial Statements","date":"2025-06-10","arxiv_id":"2506.08762","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2506-08762#ran","syntology_url":"https://syntology.ai/paper/2506.08762","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08762"}},"official":{"repos":["sakanaai/edinet-bench","sakanaai/edinet2dataset"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/how-benchmark-prediction-from-fewer-data","slug":"how-benchmark-prediction-from-fewer-data","title":"How Benchmark Prediction from Fewer Data Misses the Mark","date":"2025-06-09","arxiv_id":"2506.07673","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/how-benchmark-prediction-from-fewer-data#ran","syntology_url":"https://syntology.ai/paper/2506.07673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07673"}},"official":{"repos":["socialfoundations/benchmark-prediction"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/minicpm4-ultra-efficient-llms-on-end-devices","slug":"minicpm4-ultra-efficient-llms-on-end-devices","title":"MiniCPM4: Ultra-Efficient LLMs on End Devices","date":"2025-06-09","arxiv_id":"2506.07900","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minicpm4-ultra-efficient-llms-on-end-devices#ran","syntology_url":"https://syntology.ai/paper/2506.07900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07900"}},"official":{"repos":["openbmb/minicpm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robopara-dual-arm-robot-planning-with","slug":"robopara-dual-arm-robot-planning-with","title":"RoboPARA: Dual-Arm Robot Planning with Parallel Allocation and Recomposition Across Tasks","date":"2025-06-07","arxiv_id":"2506.06683","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robopara-dual-arm-robot-planning-with#ran","syntology_url":"https://syntology.ai/paper/2506.06683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.06683"}},"official":null}},{"url":"/paper/eigenspectrum-analysis-of-neural-networks","slug":"eigenspectrum-analysis-of-neural-networks","title":"Eigenspectrum Analysis of Neural Networks without Aspect Ratio Bias","date":"2025-06-06","arxiv_id":"2506.06280","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eigenspectrum-analysis-of-neural-networks#ran","syntology_url":"https://syntology.ai/paper/2506.06280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.06280"}},"official":null}},{"url":"/paper/heavywater-and-simplexwater-watermarking-low","slug":"heavywater-and-simplexwater-watermarking-low","title":"HeavyWater and SimplexWater: Watermarking Low-Entropy Text Distributions","date":"2025-06-06","arxiv_id":"2506.06409","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/heavywater-and-simplexwater-watermarking-low#ran","syntology_url":"https://syntology.ai/paper/2506.06409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.06409"}},"official":{"repos":["dortsur/heavywater_simplexwater"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/comfyui-copilot-an-intelligent-assistant-for","slug":"comfyui-copilot-an-intelligent-assistant-for","title":"ComfyUI-Copilot: An Intelligent Assistant for Automated Workflow Development","date":"2025-06-05","arxiv_id":"2506.05010","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/comfyui-copilot-an-intelligent-assistant-for#ran","syntology_url":"https://syntology.ai/paper/2506.05010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.05010"}},"official":{"repos":["aidc-ai/comfyui-copilot"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/compiler-optimization-via-llm-reasoning-for","slug":"compiler-optimization-via-llm-reasoning-for","title":"Compiler Optimization via LLM Reasoning for Efficient Model Serving","date":"2025-06-02","arxiv_id":"2506.01374","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/compiler-optimization-via-llm-reasoning-for#ran","syntology_url":"https://syntology.ai/paper/2506.01374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01374"}},"official":null}},{"url":"/paper/shapellm-omni-a-native-multimodal-llm-for-3d","slug":"shapellm-omni-a-native-multimodal-llm-for-3d","title":"ShapeLLM-Omni: A Native Multimodal LLM for 3D Generation and Understanding","date":"2025-06-02","arxiv_id":"2506.01853","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/shapellm-omni-a-native-multimodal-llm-for-3d#ran","syntology_url":"https://syntology.ai/paper/2506.01853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01853"}},"official":{"repos":["jamesyjl/shapellm-omni"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-videos-for-3d-world-enhancing","slug":"learning-from-videos-for-3d-world-enhancing","title":"Learning from Videos for 3D World: Enhancing MLLMs with 3D Vision Geometry Priors","date":"2025-05-30","arxiv_id":"2505.24625","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learning-from-videos-for-3d-world-enhancing#ran","syntology_url":"https://syntology.ai/paper/2505.24625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24625"}},"official":null}},{"url":"/paper/safescientist-toward-risk-aware-scientific","slug":"safescientist-toward-risk-aware-scientific","title":"SafeScientist: Toward Risk-Aware Scientific Discoveries by LLM Agents","date":"2025-05-29","arxiv_id":"2505.23559","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safescientist-toward-risk-aware-scientific#ran","syntology_url":"https://syntology.ai/paper/2505.23559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23559"}},"official":{"repos":["ulab-uiuc/safescientist"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bioreason-incentivizing-multimodal-biological","slug":"bioreason-incentivizing-multimodal-biological","title":"BioReason: Incentivizing Multimodal Biological Reasoning within a DNA-LLM Model","date":"2025-05-29","arxiv_id":"2505.23579","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bioreason-incentivizing-multimodal-biological#ran","syntology_url":"https://syntology.ai/paper/2505.23579","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23579"}},"official":{"repos":["bowang-lab/bioreason"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cadrille-multi-modal-cad-reconstruction-with","slug":"cadrille-multi-modal-cad-reconstruction-with","title":"cadrille: Multi-modal CAD Reconstruction with Online Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.22914","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/cadrille-multi-modal-cad-reconstruction-with#ran","syntology_url":"https://syntology.ai/paper/2505.22914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22914"}},"official":null}},{"url":"/paper/let-me-think-a-long-chain-of-thought-can-be","slug":"let-me-think-a-long-chain-of-thought-can-be","title":"Let Me Think! A Long Chain-of-Thought Can Be Worth Exponentially Many Short Ones","date":"2025-05-27","arxiv_id":"2505.21825","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/let-me-think-a-long-chain-of-thought-can-be#ran","syntology_url":"https://syntology.ai/paper/2505.21825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21825"}},"official":{"repos":["seyedparsa/let-me-think"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/wina-weight-informed-neuron-activation-for","slug":"wina-weight-informed-neuron-activation-for","title":"WINA: Weight Informed Neuron Activation for Accelerating Large Language Model Inference","date":"2025-05-26","arxiv_id":"2505.19427","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wina-weight-informed-neuron-activation-for#ran","syntology_url":"https://syntology.ai/paper/2505.19427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19427"}},"official":{"repos":["microsoft/wina"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-multimodal-large-language-model","slug":"unifying-multimodal-large-language-model","title":"Unifying Multimodal Large Language Model Capabilities and Modalities via Model Merging","date":"2025-05-26","arxiv_id":"2505.19892","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2505.19892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19892"}},"official":{"repos":["walkerworldpeace/mllmerging"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rearank-reasoning-re-ranking-agent-via","slug":"rearank-reasoning-re-ranking-agent-via","title":"REARANK: Reasoning Re-ranking Agent via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.20046","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/rearank-reasoning-re-ranking-agent-via#ran","syntology_url":"https://syntology.ai/paper/2505.20046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20046"}},"official":{"repos":["lezhang7/rearank"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/cosyvoice-3-towards-in-the-wild-speech","slug":"cosyvoice-3-towards-in-the-wild-speech","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","date":"2025-05-23","arxiv_id":"2505.17589","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cosyvoice-3-towards-in-the-wild-speech#ran","syntology_url":"https://syntology.ai/paper/2505.17589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17589"}},"official":{"repos":["funaudiollm/cosyvoice"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/how-can-i-publish-my-llm-benchmark-without","slug":"how-can-i-publish-my-llm-benchmark-without","title":"How Can I Publish My LLM Benchmark Without Giving the True Answers Away?","date":"2025-05-23","arxiv_id":"2505.18102","repositories_listed":0,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-can-i-publish-my-llm-benchmark-without#ran","syntology_url":"https://syntology.ai/paper/2505.18102","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18102"}},"official":null}},{"url":"/paper/adams-momentum-itself-can-be-a-normalizer-for","slug":"adams-momentum-itself-can-be-a-normalizer-for","title":"AdamS: Momentum Itself Can Be A Normalizer for LLM Pretraining and Post-training","date":"2025-05-22","arxiv_id":"2505.16363","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adams-momentum-itself-can-be-a-normalizer-for#ran","syntology_url":"https://syntology.ai/paper/2505.16363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16363"}},"official":{"repos":["pku-huzhang/AdamS"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/dimple-discrete-diffusion-multimodal-large","slug":"dimple-discrete-diffusion-multimodal-large","title":"Dimple: Discrete Diffusion Multimodal Large Language Model with Parallel Decoding","date":"2025-05-22","arxiv_id":"2505.16990","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dimple-discrete-diffusion-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2505.16990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16990"}},"official":{"repos":["yu-rp/dimple"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/conciserl-conciseness-guided-reinforcement","slug":"conciserl-conciseness-guided-reinforcement","title":"ConciseRL: Conciseness-Guided Reinforcement Learning for Efficient Reasoning Models","date":"2025-05-22","arxiv_id":"2505.17250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conciserl-conciseness-guided-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2505.17250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17250"}},"official":{"repos":["razvandu/conciserl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lost-in-benchmarks-rethinking-large-language","slug":"lost-in-benchmarks-rethinking-large-language","title":"Lost in Benchmarks? Rethinking Large Language Model Benchmarking with Item Response Theory","date":"2025-05-21","arxiv_id":"2505.15055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lost-in-benchmarks-rethinking-large-language#ran","syntology_url":"https://syntology.ai/paper/2505.15055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15055"}},"official":{"repos":["Joe-Hall-Lee/PSN-IRT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lmgame-bench-how-good-are-llms-at-playing","slug":"lmgame-bench-how-good-are-llms-at-playing","title":"lmgame-Bench: How Good are LLMs at Playing Games?","date":"2025-05-21","arxiv_id":"2505.15146","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lmgame-bench-how-good-are-llms-at-playing#ran","syntology_url":"https://syntology.ai/paper/2505.15146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15146"}},"official":{"repos":["lmgame-org/gamingagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/web-shepherd-advancing-prms-for-reinforcing","slug":"web-shepherd-advancing-prms-for-reinforcing","title":"Web-Shepherd: Advancing PRMs for Reinforcing Web Agents","date":"2025-05-21","arxiv_id":"2505.15277","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/web-shepherd-advancing-prms-for-reinforcing#ran","syntology_url":"https://syntology.ai/paper/2505.15277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15277"}},"official":{"repos":["kyle8581/Web-Shepherd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/trajectory-bellman-residual-minimization-a","slug":"trajectory-bellman-residual-minimization-a","title":"Trajectory Bellman Residual Minimization: A Simple Value-Based Method for LLM Reasoning","date":"2025-05-21","arxiv_id":"2505.15311","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-bellman-residual-minimization-a#ran","syntology_url":"https://syntology.ai/paper/2505.15311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15311"}},"official":null}},{"url":"/paper/polar-sparsity-high-throughput-batched-llm","slug":"polar-sparsity-high-throughput-batched-llm","title":"Polar Sparsity: High Throughput Batched LLM Inferencing with Scalable Contextual Sparsity","date":"2025-05-20","arxiv_id":"2505.14884","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polar-sparsity-high-throughput-batched-llm#ran","syntology_url":"https://syntology.ai/paper/2505.14884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14884"}},"official":{"repos":["susavlsh10/polar-sparsity"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/busterx-mllm-powered-ai-generated-video","slug":"busterx-mllm-powered-ai-generated-video","title":"BusterX: MLLM-Powered AI-Generated Video Forgery Detection and Explanation","date":"2025-05-19","arxiv_id":"2505.12620","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/busterx-mllm-powered-ai-generated-video#ran","syntology_url":"https://syntology.ai/paper/2505.12620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12620"}},"official":{"repos":["l8cv/busterx"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-traitors-deception-and-trust-in-multi","slug":"the-traitors-deception-and-trust-in-multi","title":"The Traitors: Deception and Trust in Multi-Agent Language Model Simulations","date":"2025-05-19","arxiv_id":"2505.12923","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-traitors-deception-and-trust-in-multi#ran","syntology_url":"https://syntology.ai/paper/2505.12923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12923"}},"official":{"repos":["pedrocurvo/thetraitors"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cpret-a-dataset-benchmark-and-model-for","slug":"cpret-a-dataset-benchmark-and-model-for","title":"CPRet: A Dataset, Benchmark, and Model for Retrieval in Competitive Programming","date":"2025-05-19","arxiv_id":"2505.12925","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cpret-a-dataset-benchmark-and-model-for#ran","syntology_url":"https://syntology.ai/paper/2505.12925","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12925"}},"official":{"repos":["coldchair/cpret"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/2505-10861","slug":"2505-10861","title":"Improving the Data-efficiency of Reinforcement Learning by Warm-starting with LLM","date":"2025-05-16","arxiv_id":"2505.10861","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2505-10861#ran","syntology_url":"https://syntology.ai/paper/2505.10861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10861"}},"official":{"repos":["duongnhatthang/llamagym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-10940","slug":"2505-10940","title":"Who You Are Matters: Bridging Topics and Social Roles via LLM-Enhanced Logical Recommendation","date":"2025-05-16","arxiv_id":"2505.10940","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2505-10940#ran","syntology_url":"https://syntology.ai/paper/2505.10940","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10940"}},"official":null}},{"url":"/paper/token-level-uncertainty-estimation-for-large","slug":"token-level-uncertainty-estimation-for-large","title":"Token-Level Uncertainty Estimation for Large Language Model Reasoning","date":"2025-05-16","arxiv_id":"2505.11737","repositories_listed":0,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/token-level-uncertainty-estimation-for-large#ran","syntology_url":"https://syntology.ai/paper/2505.11737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11737"}},"official":null}},{"url":"/paper/contrastive-cross-course-knowledge-tracing","slug":"contrastive-cross-course-knowledge-tracing","title":"Contrastive Cross-Course Knowledge Tracing via Concept Graph Guided Knowledge Transfer","date":"2025-05-14","arxiv_id":"2505.13489","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":5,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/contrastive-cross-course-knowledge-tracing#ran","syntology_url":"https://syntology.ai/paper/2505.13489","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13489"}},"official":{"repos":["dqyzhwk/transkt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":5,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/measuring-general-intelligence-with-generated","slug":"measuring-general-intelligence-with-generated","title":"Measuring General Intelligence with Generated Games","date":"2025-05-12","arxiv_id":"2505.07215","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/measuring-general-intelligence-with-generated#ran","syntology_url":"https://syntology.ai/paper/2505.07215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07215"}},"official":{"repos":["vivek3141/gg-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/yulan-onesim-towards-the-next-generation-of","slug":"yulan-onesim-towards-the-next-generation-of","title":"YuLan-OneSim: Towards the Next Generation of Social Simulator with Large Language Models","date":"2025-05-12","arxiv_id":"2505.07581","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/yulan-onesim-towards-the-next-generation-of#ran","syntology_url":"https://syntology.ai/paper/2505.07581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07581"}},"official":{"repos":["RUC-GSAI/YuLan-OneSim"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/guidedquant-large-language-model-quantization","slug":"guidedquant-large-language-model-quantization","title":"GuidedQuant: Large Language Model Quantization via Exploiting End Loss Guidance","date":"2025-05-11","arxiv_id":"2505.07004","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guidedquant-large-language-model-quantization#ran","syntology_url":"https://syntology.ai/paper/2505.07004","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07004"}},"official":{"repos":["snu-mllab/guidedquant"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/revealing-weaknesses-in-text-watermarking","slug":"revealing-weaknesses-in-text-watermarking","title":"Revealing Weaknesses in Text Watermarking Through Self-Information Rewrite Attacks","date":"2025-05-08","arxiv_id":"2505.05190","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":2,"n_ran_checked":2,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":10,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/revealing-weaknesses-in-text-watermarking#ran","syntology_url":"https://syntology.ai/paper/2505.05190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05190"}},"official":{"repos":["allencheng97/self-information-rewrite-attack"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/apply-hierarchical-chain-of-generation-to","slug":"apply-hierarchical-chain-of-generation-to","title":"Apply Hierarchical-Chain-of-Generation to Complex Attributes Text-to-3D Generation","date":"2025-05-07","arxiv_id":"2505.05505","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/apply-hierarchical-chain-of-generation-to#ran","syntology_url":"https://syntology.ai/paper/2505.05505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05505"}},"official":{"repos":["wakals/gascol"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/grape-heterogeneous-graph-representation","slug":"grape-heterogeneous-graph-representation","title":"GRAPE: Heterogeneous Graph Representation Learning for Genetic Perturbation with Coding and Non-Coding Biotype","date":"2025-05-06","arxiv_id":"2505.03853","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/grape-heterogeneous-graph-representation#ran","syntology_url":"https://syntology.ai/paper/2505.03853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.03853"}},"official":null}},{"url":"/paper/radio-rate-distortion-optimization-for-large","slug":"radio-rate-distortion-optimization-for-large","title":"Radio: Rate-Distortion Optimization for Large Language Model Compression","date":"2025-05-05","arxiv_id":"2505.03031","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/radio-rate-distortion-optimization-for-large#ran","syntology_url":"https://syntology.ai/paper/2505.03031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.03031"}},"official":null}},{"url":"/paper/mf-llm-simulating-collective-decision","slug":"mf-llm-simulating-collective-decision","title":"MF-LLM: Simulating Population Decision Dynamics via a Mean-Field Large Language Model Framework","date":"2025-04-30","arxiv_id":"2504.21582","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mf-llm-simulating-collective-decision#ran","syntology_url":"https://syntology.ai/paper/2504.21582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.21582"}},"official":{"repos":["Miracle1207/Mean-Field-LLM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/saga-a-security-architecture-for-governing-ai","slug":"saga-a-security-architecture-for-governing-ai","title":"SAGA: A Security Architecture for Governing AI Agentic Systems","date":"2025-04-27","arxiv_id":"2504.21034","repositories_listed":0,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/saga-a-security-architecture-for-governing-ai#ran","syntology_url":"https://syntology.ai/paper/2504.21034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.21034"}},"official":null}},{"url":"/paper/evaluating-judges-as-evaluators-the-jetts","slug":"evaluating-judges-as-evaluators-the-jetts","title":"Evaluating Judges as Evaluators: The JETTS Benchmark of LLM-as-Judges as Test-Time Scaling Evaluators","date":"2025-04-21","arxiv_id":"2504.15253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-judges-as-evaluators-the-jetts#ran","syntology_url":"https://syntology.ai/paper/2504.15253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15253"}},"official":{"repos":["salesforceairesearch/jetts-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/virology-capabilities-test-vct-a-multimodal","slug":"virology-capabilities-test-vct-a-multimodal","title":"Virology Capabilities Test (VCT): A Multimodal Virology Q&A Benchmark","date":"2025-04-21","arxiv_id":"2504.16137","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/virology-capabilities-test-vct-a-multimodal#ran","syntology_url":"https://syntology.ai/paper/2504.16137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16137"}},"official":null}},{"url":"/paper/sotopia-s4-a-user-friendly-system-for","slug":"sotopia-s4-a-user-friendly-system-for","title":"SOTOPIA-S4: a user-friendly system for flexible, customizable, and large-scale social simulation","date":"2025-04-19","arxiv_id":"2504.16122","repositories_listed":0,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sotopia-s4-a-user-friendly-system-for#ran","syntology_url":"https://syntology.ai/paper/2504.16122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16122"}},"official":null}},{"url":"/paper/retrieval-augmented-generation-with-3","slug":"retrieval-augmented-generation-with-3","title":"Retrieval-Augmented Generation with Conflicting Evidence","date":"2025-04-17","arxiv_id":"2504.13079","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/retrieval-augmented-generation-with-3#ran","syntology_url":"https://syntology.ai/paper/2504.13079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13079"}},"official":{"repos":["hannight/ramdocs"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dids-domain-impact-aware-data-sampling-for","slug":"dids-domain-impact-aware-data-sampling-for","title":"DIDS: Domain Impact-aware Data Sampling for Large Language Model Training","date":"2025-04-17","arxiv_id":"2504.13227","repositories_listed":0,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dids-domain-impact-aware-data-sampling-for#ran","syntology_url":"https://syntology.ai/paper/2504.13227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13227"}},"official":null}},{"url":"/paper/kimina-prover-preview-towards-large-formal","slug":"kimina-prover-preview-towards-large-formal","title":"Kimina-Prover Preview: Towards Large Formal Reasoning Models with Reinforcement Learning","date":"2025-04-15","arxiv_id":"2504.11354","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/kimina-prover-preview-towards-large-formal#ran","syntology_url":"https://syntology.ai/paper/2504.11354","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11354"}},"official":{"repos":["moonshotai/kimina-prover-preview"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-unlearning-reveals-a-stronger-than","slug":"llm-unlearning-reveals-a-stronger-than","title":"LLM Unlearning Reveals a Stronger-Than-Expected Coreset Effect in Current Benchmarks","date":"2025-04-14","arxiv_id":"2504.10185","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llm-unlearning-reveals-a-stronger-than#ran","syntology_url":"https://syntology.ai/paper/2504.10185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10185"}},"official":{"repos":["optml-group/mu-coreset"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-scalability-of-simplicity-empirical","slug":"the-scalability-of-simplicity-empirical","title":"The Scalability of Simplicity: Empirical Analysis of Vision-Language Learning with a Single Transformer","date":"2025-04-14","arxiv_id":"2504.10462","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/the-scalability-of-simplicity-empirical#ran","syntology_url":"https://syntology.ai/paper/2504.10462","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10462"}},"official":{"repos":["bytedance/sail"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/internvl3-exploring-advanced-training-and","slug":"internvl3-exploring-advanced-training-and","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","date":"2025-04-14","arxiv_id":"2504.10479","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internvl3-exploring-advanced-training-and#ran","syntology_url":"https://syntology.ai/paper/2504.10479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10479"}},"official":{"repos":["opengvlab/internvl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/segearth-r1-geospatial-pixel-reasoning-via","slug":"segearth-r1-geospatial-pixel-reasoning-via","title":"SegEarth-R1: Geospatial Pixel Reasoning via Large Language Model","date":"2025-04-13","arxiv_id":"2504.09644","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/segearth-r1-geospatial-pixel-reasoning-via#ran","syntology_url":"https://syntology.ai/paper/2504.09644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.09644"}},"official":{"repos":["earth-insights/segearth-r1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/glus-global-local-reasoning-unified-into-a","slug":"glus-global-local-reasoning-unified-into-a","title":"GLUS: Global-Local Reasoning Unified into A Single Large Language Model for Video Segmentation","date":"2025-04-10","arxiv_id":"2504.07962","repositories_listed":1,"syntology":{"n":12,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":12,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/glus-global-local-reasoning-unified-into-a#ran","syntology_url":"https://syntology.ai/paper/2504.07962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07962"}},"official":null}},{"url":"/paper/representation-bending-for-large-language","slug":"representation-bending-for-large-language","title":"Representation Bending for Large Language Model Safety","date":"2025-04-02","arxiv_id":"2504.01550","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/representation-bending-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2504.01550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01550"}},"official":{"repos":["aim-intelligence/repbend"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/astroagents-a-multi-agent-ai-for-hypothesis","slug":"astroagents-a-multi-agent-ai-for-hypothesis","title":"AstroAgents: A Multi-Agent AI for Hypothesis Generation from Mass Spectrometry Data","date":"2025-03-29","arxiv_id":"2503.23170","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/astroagents-a-multi-agent-ai-for-hypothesis#ran","syntology_url":"https://syntology.ai/paper/2503.23170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23170"}},"official":{"repos":["amirgroup-codes/astroagents"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-tokenizer-distillation-via-approximate","slug":"cross-tokenizer-distillation-via-approximate","title":"Cross-Tokenizer Distillation via Approximate Likelihood Matching","date":"2025-03-25","arxiv_id":"2503.20083","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cross-tokenizer-distillation-via-approximate#ran","syntology_url":"https://syntology.ai/paper/2503.20083","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20083"}},"official":{"repos":["bminixhofer/alm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/trajectory-balance-with-asynchrony-decoupling","slug":"trajectory-balance-with-asynchrony-decoupling","title":"Trajectory Balance with Asynchrony: Decoupling Exploration and Learning for Fast, Scalable LLM Post-Training","date":"2025-03-24","arxiv_id":"2503.18929","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trajectory-balance-with-asynchrony-decoupling#ran","syntology_url":"https://syntology.ai/paper/2503.18929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18929"}},"official":null}},{"url":"/paper/does-context-matter-contextualjudgebench-for","slug":"does-context-matter-contextualjudgebench-for","title":"Does Context Matter? ContextualJudgeBench for Evaluating LLM-based Judges in Contextual Settings","date":"2025-03-19","arxiv_id":"2503.15620","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/does-context-matter-contextualjudgebench-for#ran","syntology_url":"https://syntology.ai/paper/2503.15620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.15620"}},"official":{"repos":["salesforceairesearch/contextualjudgebench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/svd-llm-v2-optimizing-singular-value","slug":"svd-llm-v2-optimizing-singular-value","title":"SVD-LLM V2: Optimizing Singular Value Truncation for Large Language Model Compression","date":"2025-03-16","arxiv_id":"2503.12340","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/svd-llm-v2-optimizing-singular-value#ran","syntology_url":"https://syntology.ai/paper/2503.12340","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12340"}},"official":{"repos":["aiot-mlsys-lab/svd-llm"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gumiho-a-hybrid-architecture-to-prioritize","slug":"gumiho-a-hybrid-architecture-to-prioritize","title":"Gumiho: A Hybrid Architecture to Prioritize Early Tokens in Speculative Decoding","date":"2025-03-13","arxiv_id":"2503.10135","repositories_listed":0,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/gumiho-a-hybrid-architecture-to-prioritize#ran","syntology_url":"https://syntology.ai/paper/2503.10135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10135"}},"official":null}},{"url":"/paper/got-unleashing-reasoning-capability-of","slug":"got-unleashing-reasoning-capability-of","title":"GoT: Unleashing Reasoning Capability of Multimodal Large Language Model for Visual Generation and Editing","date":"2025-03-13","arxiv_id":"2503.10639","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/got-unleashing-reasoning-capability-of#ran","syntology_url":"https://syntology.ai/paper/2503.10639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10639"}},"official":{"repos":["rongyaofang/got"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/seedream-2-0-a-native-chinese-english","slug":"seedream-2-0-a-native-chinese-english","title":"Seedream 2.0: A Native Chinese-English Bilingual Image Generation Foundation Model","date":"2025-03-10","arxiv_id":"2503.07703","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/seedream-2-0-a-native-chinese-english#ran","syntology_url":"https://syntology.ai/paper/2503.07703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07703"}},"official":null}},{"url":"/paper/dynamic-updates-for-language-adaptation-in","slug":"dynamic-updates-for-language-adaptation-in","title":"Dynamic Updates for Language Adaptation in Visual-Language Tracking","date":"2025-03-09","arxiv_id":"2503.06621","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dynamic-updates-for-language-adaptation-in#ran","syntology_url":"https://syntology.ai/paper/2503.06621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06621"}},"official":{"repos":["gxnu-zhonglab/dutrack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/qg-sms-enhancing-test-item-analysis-via","slug":"qg-sms-enhancing-test-item-analysis-via","title":"QG-SMS: Enhancing Test Item Analysis via Student Modeling and Simulation","date":"2025-03-07","arxiv_id":"2503.05888","repositories_listed":0,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/qg-sms-enhancing-test-item-analysis-via#ran","syntology_url":"https://syntology.ai/paper/2503.05888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05888"}},"official":null}},{"url":"/paper/detqus-decomposition-enhanced-transformers","slug":"detqus-decomposition-enhanced-transformers","title":"DETQUS: Decomposition-Enhanced Transformers for QUery-focused Summarization","date":"2025-03-07","arxiv_id":"2503.05935","repositories_listed":0,"syntology":{"n":29,"n_ran":19,"n_constructed":0,"n_ran_checked":19,"n_instrument":0,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":0,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/detqus-decomposition-enhanced-transformers#ran","syntology_url":"https://syntology.ai/paper/2503.05935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05935"}},"official":null}},{"url":"/paper/parallelized-planning-acting-for-efficient","slug":"parallelized-planning-acting-for-efficient","title":"Parallelized Planning-Acting for Efficient LLM-based Multi-Agent Systems","date":"2025-03-05","arxiv_id":"2503.03505","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parallelized-planning-acting-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2503.03505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.03505"}},"official":{"repos":["zju-vipa/odyssey"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-fine-grained-hand-object-dynamics","slug":"modeling-fine-grained-hand-object-dynamics","title":"Modeling Fine-Grained Hand-Object Dynamics for Egocentric Video Representation Learning","date":"2025-03-02","arxiv_id":"2503.00986","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/modeling-fine-grained-hand-object-dynamics#ran","syntology_url":"https://syntology.ai/paper/2503.00986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00986"}},"official":{"repos":["openrobotlab/egohod"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2503-00555","slug":"2503-00555","title":"Safety Tax: Safety Alignment Makes Your Large Reasoning Models Less Reasonable","date":"2025-03-01","arxiv_id":"2503.00555","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2503-00555#ran","syntology_url":"https://syntology.ai/paper/2503.00555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00555"}},"official":{"repos":["git-disl/safety-tax"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/udora-a-unified-red-teaming-framework-against","slug":"udora-a-unified-red-teaming-framework-against","title":"UDora: A Unified Red Teaming Framework against LLM Agents by Dynamically Hijacking Their Own Reasoning","date":"2025-02-28","arxiv_id":"2503.01908","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/udora-a-unified-red-teaming-framework-against#ran","syntology_url":"https://syntology.ai/paper/2503.01908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01908"}},"official":{"repos":["ai-secure/udora"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/playing-pokemon-red-via-deep-reinforcement","slug":"playing-pokemon-red-via-deep-reinforcement","title":"Playing Pokémon Red via Deep Reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.19920","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-pokemon-red-via-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2502.19920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19920"}},"official":{"repos":["MarcoMeter/neroRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seismollm-advancing-seismic-monitoring-via","slug":"seismollm-advancing-seismic-monitoring-via","title":"SeisMoLLM: Advancing Seismic Monitoring via Cross-modal Transfer with Pre-trained Large Language Model","date":"2025-02-27","arxiv_id":"2502.19960","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/seismollm-advancing-seismic-monitoring-via#ran","syntology_url":"https://syntology.ai/paper/2502.19960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19960"}},"official":{"repos":["StarMoonWang/SeisMoLLM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/agentsociety-challenge-designing-llm-agents","slug":"agentsociety-challenge-designing-llm-agents","title":"AgentSociety Challenge: Designing LLM Agents for User Modeling and Recommendation on Web Platforms","date":"2025-02-26","arxiv_id":"2502.18754","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/agentsociety-challenge-designing-llm-agents#ran","syntology_url":"https://syntology.ai/paper/2502.18754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18754"}},"official":{"repos":["tsinghua-fib-lab/agentsocietychallenge"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/notagen-advancing-musicality-in-symbolic","slug":"notagen-advancing-musicality-in-symbolic","title":"NotaGen: Advancing Musicality in Symbolic Music Generation with Large Language Model Training Paradigms","date":"2025-02-25","arxiv_id":"2502.18008","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/notagen-advancing-musicality-in-symbolic#ran","syntology_url":"https://syntology.ai/paper/2502.18008","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18008"}},"official":null}},{"url":"/paper/sparc-score-prompting-and-adaptive-fusion-for","slug":"sparc-score-prompting-and-adaptive-fusion-for","title":"SPARC: Score Prompting and Adaptive Fusion for Zero-Shot Multi-Label Recognition in Vision-Language Models","date":"2025-02-24","arxiv_id":"2502.16911","repositories_listed":0,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/sparc-score-prompting-and-adaptive-fusion-for#ran","syntology_url":"https://syntology.ai/paper/2502.16911","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.16911"}},"official":null}},{"url":"/paper/baichuan-audio-a-unified-framework-for-end-to","slug":"baichuan-audio-a-unified-framework-for-end-to","title":"Baichuan-Audio: A Unified Framework for End-to-End Speech Interaction","date":"2025-02-24","arxiv_id":"2502.17239","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/baichuan-audio-a-unified-framework-for-end-to#ran","syntology_url":"https://syntology.ai/paper/2502.17239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17239"}},"official":{"repos":["baichuan-inc/baichuan-audio"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/weakly-supervised-video-scene-graph","slug":"weakly-supervised-video-scene-graph","title":"Weakly Supervised Video Scene Graph Generation via Natural Language Supervision","date":"2025-02-21","arxiv_id":"2502.15370","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/weakly-supervised-video-scene-graph#ran","syntology_url":"https://syntology.ai/paper/2502.15370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15370"}},"official":{"repos":["rlqja1107/NL-VSGG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rapid-word-learning-through-meta-in-context","slug":"rapid-word-learning-through-meta-in-context","title":"Rapid Word Learning Through Meta In-Context Learning","date":"2025-02-20","arxiv_id":"2502.14791","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rapid-word-learning-through-meta-in-context#ran","syntology_url":"https://syntology.ai/paper/2502.14791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14791"}},"official":null}},{"url":"/paper/towards-text-image-interleaved-retrieval","slug":"towards-text-image-interleaved-retrieval","title":"Towards Text-Image Interleaved Retrieval","date":"2025-02-18","arxiv_id":"2502.12799","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/towards-text-image-interleaved-retrieval#ran","syntology_url":"https://syntology.ai/paper/2502.12799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.12799"}},"official":{"repos":["vec-ai/wikihow-tiir"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/smart-self-aware-agent-for-tool-overuse","slug":"smart-self-aware-agent-for-tool-overuse","title":"SMART: Self-Aware Agent for Tool Overuse Mitigation","date":"2025-02-17","arxiv_id":"2502.11435","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/smart-self-aware-agent-for-tool-overuse#ran","syntology_url":"https://syntology.ai/paper/2502.11435","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11435"}},"official":{"repos":["qiancheng0/open-smartagent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-mem-agentic-memory-for-llm-agents","slug":"a-mem-agentic-memory-for-llm-agents","title":"A-MEM: Agentic Memory for LLM Agents","date":"2025-02-17","arxiv_id":"2502.12110","repositories_listed":4,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-mem-agentic-memory-for-llm-agents#ran","syntology_url":"https://syntology.ai/paper/2502.12110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.12110"}},"official":{"repos":["agiresearch/a-mem","wujiangxu/agenticmemory"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/scale-towards-collaborative-content-analysis","slug":"scale-towards-collaborative-content-analysis","title":"SCALE: Towards Collaborative Content Analysis in Social Science with Large Language Model Agents and Human Intervention","date":"2025-02-16","arxiv_id":"2502.10937","repositories_listed":1,"syntology":{"n":8,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/scale-towards-collaborative-content-analysis#ran","syntology_url":"https://syntology.ai/paper/2502.10937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.10937"}},"official":{"repos":["ChengshuaiZhao0/SCALE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/unveiling-environmental-impacts-of-large","slug":"unveiling-environmental-impacts-of-large","title":"Unveiling Environmental Impacts of Large Language Model Serving: A Functional Unit View","date":"2025-02-16","arxiv_id":"2502.11256","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/unveiling-environmental-impacts-of-large#ran","syntology_url":"https://syntology.ai/paper/2502.11256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11256"}},"official":{"repos":["jojacola/fuel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/goedel-prover-a-frontier-model-for-open","slug":"goedel-prover-a-frontier-model-for-open","title":"Goedel-Prover: A Frontier Model for Open-Source Automated Theorem Proving","date":"2025-02-11","arxiv_id":"2502.07640","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/goedel-prover-a-frontier-model-for-open#ran","syntology_url":"https://syntology.ai/paper/2502.07640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07640"}},"official":{"repos":["Goedel-LM/Goedel-Prover"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robotouille-an-asynchronous-planning","slug":"robotouille-an-asynchronous-planning","title":"Robotouille: An Asynchronous Planning Benchmark for LLM Agents","date":"2025-02-06","arxiv_id":"2502.05227","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robotouille-an-asynchronous-planning#ran","syntology_url":"https://syntology.ai/paper/2502.05227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05227"}},"official":{"repos":["portal-cornell/robotouille"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/speculative-prefill-turbocharging-ttft-with","slug":"speculative-prefill-turbocharging-ttft-with","title":"Speculative Prefill: Turbocharging TTFT with Lightweight and Training-Free Token Importance Estimation","date":"2025-02-05","arxiv_id":"2502.02789","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/speculative-prefill-turbocharging-ttft-with#ran","syntology_url":"https://syntology.ai/paper/2502.02789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02789"}},"official":{"repos":["Jingyu6/speculative_prefill"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"e84ed2f2755b9fa07c005ab566a3dbba4c35f9398b8dac0feaa277431fbc576e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}