{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/7","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":56,"rows_per_page":100,"rows":[601,700],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/6","next":"/task/benchmarking/papers/8","papers":[{"url":"/paper/psocr-benchmarking-large-multimodal-models","slug":"psocr-benchmarking-large-multimodal-models","title":"PsOCR: Benchmarking Large Multimodal Models for Optical Character Recognition in Low-resource Pashto Language","date":"2025-05-15","arxiv_id":"2505.10055","repositories_listed":1,"syntology":null},{"url":"/paper/words-that-unite-the-world-a-unified","slug":"words-that-unite-the-world-a-unified","title":"Words That Unite The World: A Unified Framework for Deciphering Central Bank Communications Globally","date":"2025-05-15","arxiv_id":"2505.17048","repositories_listed":1,"syntology":null},{"url":"/paper/biovfm-21m-benchmarking-and-scaling-self","slug":"biovfm-21m-benchmarking-and-scaling-self","title":"BioVFM-21M: Benchmarking and Scaling Self-Supervised Vision Foundation Models for Biomedical Image Analysis","date":"2025-05-14","arxiv_id":"2505.09329","repositories_listed":1,"syntology":null},{"url":"/paper/openlka-an-open-dataset-of-lane-keeping-1","slug":"openlka-an-open-dataset-of-lane-keeping-1","title":"OpenLKA: An Open Dataset of Lane Keeping Assist from Recent Car Models under Real-world Driving Conditions","date":"2025-05-14","arxiv_id":"2505.09092","repositories_listed":1,"syntology":null},{"url":"/paper/towards-scalable-surrogate-models-based-on","slug":"towards-scalable-surrogate-models-based-on","title":"Towards scalable surrogate models based on Neural Fields for large scale aerodynamic simulations","date":"2025-05-14","arxiv_id":"2505.14704","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-ai-scientists-in-omics-data","slug":"benchmarking-ai-scientists-in-omics-data","title":"Benchmarking AI scientists in omics data-driven biological research","date":"2025-05-13","arxiv_id":"2505.08341","repositories_listed":1,"syntology":null},{"url":"/paper/exebench-benchmarking-foundation-models-on","slug":"exebench-benchmarking-foundation-models-on","title":"ExEBench: Benchmarking Foundation Models on Extreme Earth Events","date":"2025-05-13","arxiv_id":"2505.08529","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-synthetic-data-evaluations-of","slug":"grounding-synthetic-data-evaluations-of","title":"Grounding Synthetic Data Evaluations of Language Models in Unsupervised Document Corpora","date":"2025-05-13","arxiv_id":"2505.08905","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-model-psychometrics-a","slug":"large-language-model-psychometrics-a","title":"Large Language Model Psychometrics: A Systematic Review of Evaluation, Validation, and Enhancement","date":"2025-05-13","arxiv_id":"2505.08245","repositories_listed":1,"syntology":null},{"url":"/paper/from-raw-affiliations-to-organization","slug":"from-raw-affiliations-to-organization","title":"From raw affiliations to organization identifiers","date":"2025-05-12","arxiv_id":"2505.07577","repositories_listed":1,"syntology":null},{"url":"/paper/from-knowledge-to-reasoning-evaluating-llms","slug":"from-knowledge-to-reasoning-evaluating-llms","title":"From Knowledge to Reasoning: Evaluating LLMs for Ionic Liquids Research in Chemical and Biological Engineering","date":"2025-05-11","arxiv_id":"2505.06964","repositories_listed":1,"syntology":null},{"url":"/paper/fnbench-benchmarking-robust-federated","slug":"fnbench-benchmarking-robust-federated","title":"FNBench: Benchmarking Robust Federated Learning against Noisy Labels","date":"2025-05-10","arxiv_id":"2505.06684","repositories_listed":1,"syntology":null},{"url":"/paper/jaxrobotarium-training-and-deploying-multi","slug":"jaxrobotarium-training-and-deploying-multi","title":"JaxRobotarium: Training and Deploying Multi-Robot Policies in 10 Minutes","date":"2025-05-10","arxiv_id":"2505.06771","repositories_listed":1,"syntology":null},{"url":"/paper/a-neuro-symbolic-framework-for-sequence","slug":"a-neuro-symbolic-framework-for-sequence","title":"A Neuro-Symbolic Framework for Sequence Classification with Relational and Temporal Knowledge","date":"2025-05-08","arxiv_id":"2505.05106","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-vision-language-action-models-in","slug":"benchmarking-vision-language-action-models-in","title":"Benchmarking Vision, Language, & Action Models in Procedurally Generated, Open Ended Action Environments","date":"2025-05-08","arxiv_id":"2505.05540","repositories_listed":1,"syntology":null},{"url":"/paper/dispbench-benchmarking-disparity-estimation","slug":"dispbench-benchmarking-disparity-estimation","title":"DispBench: Benchmarking Disparity Estimation to Synthetic Corruptions","date":"2025-05-08","arxiv_id":"2505.05091","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-treatment-effect-estimation-via","slug":"enhancing-treatment-effect-estimation-via","title":"Enhancing Treatment Effect Estimation via Active Learning: A Counterfactual Covering Perspective","date":"2025-05-08","arxiv_id":"2505.05242","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-treatment-effect-estimation-via#ran","syntology_url":"https://syntology.ai/paper/2505.05242","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05242"}},"official":{"repos":["uqhwen2/FCCM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pytdc-a-multimodal-machine-learning-training","slug":"pytdc-a-multimodal-machine-learning-training","title":"PyTDC: A multimodal machine learning training, evaluation, and inference platform for biomedical foundation models","date":"2025-05-08","arxiv_id":"2505.05577","repositories_listed":1,"syntology":null},{"url":"/paper/scdrugmap-benchmarking-large-foundation","slug":"scdrugmap-benchmarking-large-foundation","title":"scDrugMap: Benchmarking Large Foundation Models for Drug Response Prediction","date":"2025-05-08","arxiv_id":"2505.05612","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-and-benchmarking-personalized-tool","slug":"advancing-and-benchmarking-personalized-tool","title":"Advancing and Benchmarking Personalized Tool Invocation for LLMs","date":"2025-05-07","arxiv_id":"2505.04072","repositories_listed":1,"syntology":null},{"url":"/paper/are-synthetic-corruptions-a-reliable-proxy","slug":"are-synthetic-corruptions-a-reliable-proxy","title":"Are Synthetic Corruptions A Reliable Proxy For Real-World Corruptions?","date":"2025-05-07","arxiv_id":"2505.04835","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-llm-faithfulness-in-rag-with","slug":"benchmarking-llm-faithfulness-in-rag-with","title":"Benchmarking LLM Faithfulness in RAG with Evolving Leaderboards","date":"2025-05-07","arxiv_id":"2505.04847","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/benchmarking-llm-faithfulness-in-rag-with#ran","syntology_url":"https://syntology.ai/paper/2505.04847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.04847"}},"official":{"repos":["vectara/FaithJudge"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-llms-swarm-intelligence","slug":"benchmarking-llms-swarm-intelligence","title":"Benchmarking LLMs' Swarm intelligence","date":"2025-05-07","arxiv_id":"2505.04364","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-traditional-machine-learning-and","slug":"benchmarking-traditional-machine-learning-and","title":"Benchmarking Traditional Machine Learning and Deep Learning Models for Fault Detection in Power Transformers","date":"2025-05-07","arxiv_id":"2505.06295","repositories_listed":1,"syntology":null},{"url":"/paper/false-promises-in-medical-imaging-ai","slug":"false-promises-in-medical-imaging-ai","title":"False Promises in Medical Imaging AI? Assessing Validity of Outperformance Claims","date":"2025-05-07","arxiv_id":"2505.04720","repositories_listed":1,"syntology":null},{"url":"/paper/rgb-event-fusion-with-self-attention-for","slug":"rgb-event-fusion-with-self-attention-for","title":"RGB-Event Fusion with Self-Attention for Collision Prediction","date":"2025-05-07","arxiv_id":"2505.04258","repositories_listed":1,"syntology":null},{"url":"/paper/combibench-benchmarking-llm-capability-for","slug":"combibench-benchmarking-llm-capability-for","title":"CombiBench: Benchmarking LLM Capability for Combinatorial Mathematics","date":"2025-05-06","arxiv_id":"2505.03171","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/combibench-benchmarking-llm-capability-for#ran","syntology_url":"https://syntology.ai/paper/2505.03171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.03171"}},"official":{"repos":["moonshotai/combibench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/medarabiq-benchmarking-large-language-models","slug":"medarabiq-benchmarking-large-language-models","title":"MedArabiQ: Benchmarking Large Language Models on Arabic Medical Tasks","date":"2025-05-06","arxiv_id":"2505.03427","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-benchmarking-and-recommendation-of","slug":"multimodal-benchmarking-and-recommendation-of","title":"Multimodal Benchmarking and Recommendation of Text-to-Image Generation Models","date":"2025-05-06","arxiv_id":"2505.04650","repositories_listed":1,"syntology":null},{"url":"/paper/towards-efficient-benchmarking-of-foundation","slug":"towards-efficient-benchmarking-of-foundation","title":"Towards Efficient Benchmarking of Foundation Models in Remote Sensing: A Capabilities Encoding Approach","date":"2025-05-06","arxiv_id":"2505.03299","repositories_listed":1,"syntology":null},{"url":"/paper/formalmath-benchmarking-formal-mathematical","slug":"formalmath-benchmarking-formal-mathematical","title":"FormalMATH: Benchmarking Formal Mathematical Reasoning of Large Language Models","date":"2025-05-05","arxiv_id":"2505.02735","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/formalmath-benchmarking-formal-mathematical#ran","syntology_url":"https://syntology.ai/paper/2505.02735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02735"}},"official":{"repos":["sphere-ai-lab/formalmath-bench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-feature-upsampling-methods-for","slug":"benchmarking-feature-upsampling-methods-for","title":"Benchmarking Feature Upsampling Methods for Vision Foundation Models using Interactive Segmentation","date":"2025-05-04","arxiv_id":"2505.02075","repositories_listed":1,"syntology":null},{"url":"/paper/meta-black-box-optimization-through-offline-q","slug":"meta-black-box-optimization-through-offline-q","title":"Meta-Black-Box-Optimization through Offline Q-function Learning","date":"2025-05-04","arxiv_id":"2505.02010","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/meta-black-box-optimization-through-offline-q#ran","syntology_url":"https://syntology.ai/paper/2505.02010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02010"}},"official":{"repos":["metaevo/q-mamba"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/nbbench-benchmarking-language-models-for","slug":"nbbench-benchmarking-language-models-for","title":"NbBench: Benchmarking Language Models for Comprehensive Nanobody Tasks","date":"2025-05-04","arxiv_id":"2505.02022","repositories_listed":1,"syntology":null},{"url":"/paper/rtv-bench-benchmarking-mllm-continuous","slug":"rtv-bench-benchmarking-mllm-continuous","title":"RTV-Bench: Benchmarking MLLM Continuous Perception, Understanding and Reasoning through Real-Time Video","date":"2025-05-04","arxiv_id":"2505.02064","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rtv-bench-benchmarking-mllm-continuous#ran","syntology_url":"https://syntology.ai/paper/2505.02064","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02064"}},"official":{"repos":["ljungang/rtv-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evalxnlp-a-framework-for-benchmarking-post","slug":"evalxnlp-a-framework-for-benchmarking-post","title":"EvalxNLP: A Framework for Benchmarking Post-Hoc Explainability Methods on NLP Models","date":"2025-05-02","arxiv_id":"2505.01238","repositories_listed":1,"syntology":null},{"url":"/paper/parameterized-argumentation-based-reasoning","slug":"parameterized-argumentation-based-reasoning","title":"Parameterized Argumentation-based Reasoning Tasks for Benchmarking Generative Language Models","date":"2025-05-02","arxiv_id":"2505.01539","repositories_listed":1,"syntology":null},{"url":"/paper/minerva-evaluating-complex-video-reasoning","slug":"minerva-evaluating-complex-video-reasoning","title":"MINERVA: Evaluating Complex Video Reasoning","date":"2025-05-01","arxiv_id":"2505.00681","repositories_listed":1,"syntology":null},{"url":"/paper/vision-mamba-in-remote-sensing-a","slug":"vision-mamba-in-remote-sensing-a","title":"Vision Mamba in Remote Sensing: A Comprehensive Survey of Techniques, Applications and Outlook","date":"2025-05-01","arxiv_id":"2505.00630","repositories_listed":1,"syntology":null},{"url":"/paper/2505-00169","slug":"2505-00169","title":"GEOM-Drugs Revisited: Toward More Chemically Accurate Benchmarks for 3D Molecule Generation","date":"2025-04-30","arxiv_id":"2505.00169","repositories_listed":1,"syntology":null},{"url":"/paper/galvatron-an-automatic-distributed-system-for","slug":"galvatron-an-automatic-distributed-system-for","title":"Galvatron: An Automatic Distributed System for Efficient Foundation Model Training","date":"2025-04-30","arxiv_id":"2504.21411","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-generalisation-gap-synthetic","slug":"bridging-the-generalisation-gap-synthetic","title":"Bridging the Generalisation Gap: Synthetic Data Generation for Multi-Site Clinical Model Validation","date":"2025-04-29","arxiv_id":"2504.20635","repositories_listed":1,"syntology":null},{"url":"/paper/osvbench-benchmarking-llms-on-specification","slug":"osvbench-benchmarking-llms-on-specification","title":"OSVBench: Benchmarking LLMs on Specification Generation Tasks for Operating System Verification","date":"2025-04-29","arxiv_id":"2504.20964","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/osvbench-benchmarking-llms-on-specification#ran","syntology_url":"https://syntology.ai/paper/2504.20964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20964"}},"official":{"repos":["lishangyu-hkust/osvbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tf1-en-3m-three-million-synthetic-moral","slug":"tf1-en-3m-three-million-synthetic-moral","title":"TF1-EN-3M: Three Million Synthetic Moral Fables for Training Small, Open Language Models","date":"2025-04-29","arxiv_id":"2504.20605","repositories_listed":1,"syntology":null},{"url":"/paper/truefake-a-real-world-case-dataset-of-last","slug":"truefake-a-real-world-case-dataset-of-last","title":"TrueFake: A Real World Case Dataset of Last Generation Fake Images also Shared on Social Networks","date":"2025-04-29","arxiv_id":"2504.20658","repositories_listed":1,"syntology":null},{"url":"/paper/blade-benchmark-suite-for-llm-driven","slug":"blade-benchmark-suite-for-llm-driven","title":"BLADE: Benchmark suite for LLM-driven Automated Design and Evolution of iterative optimisation heuristics","date":"2025-04-28","arxiv_id":"2504.20183","repositories_listed":1,"syntology":null},{"url":"/paper/bridge-benchmarking-large-language-models-for","slug":"bridge-benchmarking-large-language-models-for","title":"BRIDGE: Benchmarking Large Language Models for Understanding Real-world Clinical Practice Text","date":"2025-04-28","arxiv_id":"2504.19467","repositories_listed":1,"syntology":null},{"url":"/paper/browsecomp-zh-benchmarking-web-browsing","slug":"browsecomp-zh-benchmarking-web-browsing","title":"BrowseComp-ZH: Benchmarking Web Browsing Ability of Large Language Models in Chinese","date":"2025-04-27","arxiv_id":"2504.19314","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/browsecomp-zh-benchmarking-web-browsing#ran","syntology_url":"https://syntology.ai/paper/2504.19314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.19314"}},"official":{"repos":["palin2018/browsecomp-zh"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2504-18589","slug":"2504-18589","title":"Benchmarking Multimodal Mathematical Reasoning with Explicit Visual Dependency","date":"2025-04-24","arxiv_id":"2504.18589","repositories_listed":1,"syntology":null},{"url":"/paper/maya-addressing-inconsistencies-in-generative","slug":"maya-addressing-inconsistencies-in-generative","title":"MAYA: Addressing Inconsistencies in Generative Password Guessing through a Unified Benchmark","date":"2025-04-23","arxiv_id":"2504.16651","repositories_listed":1,"syntology":null},{"url":"/paper/fluorescence-reference-target-quantitative","slug":"fluorescence-reference-target-quantitative","title":"Fluorescence Reference Target Quantitative Analysis Library","date":"2025-04-22","arxiv_id":"2504.15496","repositories_listed":1,"syntology":null},{"url":"/paper/longmamba-enhancing-mamba-s-long-context","slug":"longmamba-enhancing-mamba-s-long-context","title":"LongMamba: Enhancing Mamba's Long Context Capabilities via Training-Free Receptive Field Enlargement","date":"2025-04-22","arxiv_id":"2504.16053","repositories_listed":1,"syntology":null},{"url":"/paper/wasp-benchmarking-web-agent-security-against","slug":"wasp-benchmarking-web-agent-security-against","title":"WASP: Benchmarking Web Agent Security Against Prompt Injection Attacks","date":"2025-04-22","arxiv_id":"2504.18575","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-vision-language-models-on","slug":"benchmarking-large-vision-language-models-on","title":"Benchmarking Large Vision-Language Models on Fine-Grained Image Tasks: A Comprehensive Evaluation","date":"2025-04-21","arxiv_id":"2504.14988","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-vision-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2504.14988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.14988"}},"official":{"repos":["seu-vipgroup/fg-bmk"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/know-me-respond-to-me-benchmarking-llms-for","slug":"know-me-respond-to-me-benchmarking-llms-for","title":"Know Me, Respond to Me: Benchmarking LLMs for Dynamic User Profiling and Personalized Responses at Scale","date":"2025-04-19","arxiv_id":"2504.14225","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/know-me-respond-to-me-benchmarking-llms-for#ran","syntology_url":"https://syntology.ai/paper/2504.14225","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.14225"}},"official":{"repos":["bowen-upenn/personamem"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-llm-based-relevance-judgment","slug":"benchmarking-llm-based-relevance-judgment","title":"Benchmarking LLM-based Relevance Judgment Methods","date":"2025-04-17","arxiv_id":"2504.12558","repositories_listed":1,"syntology":null},{"url":"/paper/causality-enhanced-decision-making-for","slug":"causality-enhanced-decision-making-for","title":"Causality-enhanced Decision-Making for Autonomous Mobile Robots in Dynamic Environments","date":"2025-04-16","arxiv_id":"2504.11901","repositories_listed":1,"syntology":null},{"url":"/paper/continual-learning-strategies-for-3d","slug":"continual-learning-strategies-for-3d","title":"Continual Learning Strategies for 3D Engineering Regression Problems: A Benchmarking Study","date":"2025-04-16","arxiv_id":"2504.12503","repositories_listed":1,"syntology":null},{"url":"/paper/fhbench-towards-efficient-and-personalized","slug":"fhbench-towards-efficient-and-personalized","title":"FHBench: Towards Efficient and Personalized Federated Learning for Multimodal Healthcare","date":"2025-04-15","arxiv_id":"2504.10817","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fhbench-towards-efficient-and-personalized#ran","syntology_url":"https://syntology.ai/paper/2504.10817","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10817"}},"official":{"repos":["wph6/fhbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mamba-based-ensemble-learning-for-white-blood","slug":"mamba-based-ensemble-learning-for-white-blood","title":"Mamba-Based Ensemble learning for White Blood Cell Classification","date":"2025-04-15","arxiv_id":"2504.11438","repositories_listed":1,"syntology":null},{"url":"/paper/tinyversegp-towards-a-modular-cross-domain","slug":"tinyversegp-towards-a-modular-cross-domain","title":"TinyverseGP: Towards a Modular Cross-domain Benchmarking Framework for Genetic Programming","date":"2025-04-14","arxiv_id":"2504.10253","repositories_listed":1,"syntology":null},{"url":"/paper/trade-offs-in-privacy-preserving-eye-tracking","slug":"trade-offs-in-privacy-preserving-eye-tracking","title":"Trade-offs in Privacy-Preserving Eye Tracking through Iris Obfuscation: A Benchmarking Study","date":"2025-04-14","arxiv_id":"2504.10267","repositories_listed":1,"syntology":null},{"url":"/paper/torchfx-a-modern-approach-to-audio-dsp-with","slug":"torchfx-a-modern-approach-to-audio-dsp-with","title":"TorchFX: A modern approach to Audio DSP with PyTorch and GPU acceleration","date":"2025-04-11","arxiv_id":"2504.08624","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-shrinkage-estimation-for","slug":"adaptive-shrinkage-estimation-for","title":"Adaptive Shrinkage Estimation For Personalized Deep Kernel Regression In Modeling Brain Trajectories","date":"2025-04-10","arxiv_id":"2504.08840","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-adversarial-robustness-to-bias","slug":"benchmarking-adversarial-robustness-to-bias","title":"Benchmarking Adversarial Robustness to Bias Elicitation in Large Language Models: Scalable Automated Assessment with LLM-as-a-Judge","date":"2025-04-10","arxiv_id":"2504.07887","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-suite-for-synthetic-aperture","slug":"benchmarking-suite-for-synthetic-aperture","title":"Benchmarking Suite for Synthetic Aperture Radar Imagery Anomaly Detection (SARIAD) Algorithms","date":"2025-04-10","arxiv_id":"2504.08115","repositories_listed":1,"syntology":null},{"url":"/paper/geological-inference-from-textual-data-using","slug":"geological-inference-from-textual-data-using","title":"Geological Inference from Textual Data using Word Embeddings","date":"2025-04-10","arxiv_id":"2504.07490","repositories_listed":1,"syntology":null},{"url":"/paper/noreval-a-norwegian-language-understanding","slug":"noreval-a-norwegian-language-understanding","title":"NorEval: A Norwegian Language Understanding and Generation Evaluation Benchmark","date":"2025-04-10","arxiv_id":"2504.07749","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multimodal-cot-reward-model","slug":"benchmarking-multimodal-cot-reward-model","title":"Benchmarking Multimodal CoT Reward Model Stepwise by Visual Program","date":"2025-04-09","arxiv_id":"2504.06606","repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-generation-of-random-surreal","slug":"evolutionary-generation-of-random-surreal","title":"Evolutionary Generation of Random Surreal Numbers for Benchmarking","date":"2025-04-09","arxiv_id":"2504.07152","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-gpt-4o-image-generation","slug":"an-empirical-study-of-gpt-4o-image-generation","title":"An Empirical Study of GPT-4o Image Generation Capabilities","date":"2025-04-08","arxiv_id":"2504.05979","repositories_listed":1,"syntology":null},{"url":"/paper/v-mage-a-game-evaluation-framework-for","slug":"v-mage-a-game-evaluation-framework-for","title":"V-MAGE: A Game Evaluation Framework for Assessing Vision-Centric Capabilities in Multimodal Large Language Models","date":"2025-04-08","arxiv_id":"2504.06148","repositories_listed":1,"syntology":null},{"url":"/paper/are-you-getting-what-you-pay-for-auditing","slug":"are-you-getting-what-you-pay-for-auditing","title":"Are You Getting What You Pay For? Auditing Model Substitution in LLM APIs","date":"2025-04-07","arxiv_id":"2504.04715","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/are-you-getting-what-you-pay-for-auditing#ran","syntology_url":"https://syntology.ai/paper/2504.04715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.04715"}},"official":{"repos":["sunblaze-ucb/llm-api-audit"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scam-a-real-world-typographic-robustness","slug":"scam-a-real-world-typographic-robustness","title":"SCAM: A Real-World Typographic Robustness Evaluation for Multimodal Foundation Models","date":"2025-04-07","arxiv_id":"2504.04893","repositories_listed":1,"syntology":null},{"url":"/paper/subjective-visual-quality-assessment-for-high","slug":"subjective-visual-quality-assessment-for-high","title":"Subjective Visual Quality Assessment for High-Fidelity Learning-Based Image Compression","date":"2025-04-07","arxiv_id":"2504.06301","repositories_listed":1,"syntology":null},{"url":"/paper/co-bench-benchmarking-language-model-agents","slug":"co-bench-benchmarking-language-model-agents","title":"CO-Bench: Benchmarking Language Model Agents in Algorithm Search for Combinatorial Optimization","date":"2025-04-06","arxiv_id":"2504.04310","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-bench-benchmarking-language-model-agents#ran","syntology_url":"https://syntology.ai/paper/2504.04310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.04310"}},"official":{"repos":["sunnweiwei/co-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-of-pathology-foundation-model","slug":"a-survey-of-pathology-foundation-model","title":"A Survey of Pathology Foundation Model: Progress and Future Directions","date":"2025-04-05","arxiv_id":"2504.04045","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-stereotypes-and-anti-stereotypes","slug":"detecting-stereotypes-and-anti-stereotypes","title":"Detecting Stereotypes and Anti-stereotypes the Correct Way Using Social Psychological Underpinnings","date":"2025-04-04","arxiv_id":"2504.03352","repositories_listed":1,"syntology":null},{"url":"/paper/do-llm-evaluators-prefer-themselves-for-a","slug":"do-llm-evaluators-prefer-themselves-for-a","title":"Do LLM Evaluators Prefer Themselves for a Reason?","date":"2025-04-04","arxiv_id":"2504.03846","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-robustness-a-benchmarking","slug":"quantifying-robustness-a-benchmarking","title":"Quantifying Robustness: A Benchmarking Framework for Deep Learning Forecasting in Cyber-Physical Systems","date":"2025-04-04","arxiv_id":"2504.03494","repositories_listed":1,"syntology":null},{"url":"/paper/envisioning-beyond-the-pixels-benchmarking","slug":"envisioning-beyond-the-pixels-benchmarking","title":"Envisioning Beyond the Pixels: Benchmarking Reasoning-Informed Visual Editing","date":"2025-04-03","arxiv_id":"2504.02826","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/envisioning-beyond-the-pixels-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2504.02826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02826"}},"official":{"repos":["phoenixz810/risebench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-ai-recruitment-sourcing-tools-by","slug":"evaluating-ai-recruitment-sourcing-tools-by","title":"Evaluating AI Recruitment Sourcing Tools by Human Preference","date":"2025-04-03","arxiv_id":"2504.02463","repositories_listed":1,"syntology":null},{"url":"/paper/generative-evaluation-of-complex-reasoning-in","slug":"generative-evaluation-of-complex-reasoning-in","title":"Generative Evaluation of Complex Reasoning in Large Language Models","date":"2025-04-03","arxiv_id":"2504.02810","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-synthetic-tabular-data-a-multi","slug":"benchmarking-synthetic-tabular-data-a-multi","title":"Benchmarking Synthetic Tabular Data: A Multi-Dimensional Evaluation Framework","date":"2025-04-02","arxiv_id":"2504.01908","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-synthetic-tabular-data-a-multi#ran","syntology_url":"https://syntology.ai/paper/2504.01908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01908"}},"official":{"repos":["mostly-ai/mostlyai-qa"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/blendergym-benchmarking-foundational-model","slug":"blendergym-benchmarking-foundational-model","title":"BlenderGym: Benchmarking Foundational Model Systems for Graphics Editing","date":"2025-04-02","arxiv_id":"2504.01786","repositories_listed":1,"syntology":null},{"url":"/paper/can-llms-grasp-implicit-cultural-values","slug":"can-llms-grasp-implicit-cultural-values","title":"Can LLMs Grasp Implicit Cultural Values? Benchmarking LLMs' Metacognitive Cultural Intelligence with CQ-Bench","date":"2025-04-01","arxiv_id":"2504.01127","repositories_listed":1,"syntology":null},{"url":"/paper/loco-epi-leave-one-chromosome-out-loco-as-a","slug":"loco-epi-leave-one-chromosome-out-loco-as-a","title":"LOCO-EPI: Leave-one-chromosome-out (LOCO) as a benchmarking paradigm for deep learning based prediction of enhancer-promoter interactions","date":"2025-04-01","arxiv_id":"2504.00306","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-up-resonate-and-fire-networks-for","slug":"scaling-up-resonate-and-fire-networks-for","title":"Scaling Up Resonate-and-Fire Networks for Fast Deep Learning","date":"2025-04-01","arxiv_id":"2504.00719","repositories_listed":1,"syntology":null},{"url":"/paper/tdbench-benchmarking-vision-language-models","slug":"tdbench-benchmarking-vision-language-models","title":"TDBench: Benchmarking Vision-Language Models in Understanding Top-Down Images","date":"2025-04-01","arxiv_id":"2504.03748","repositories_listed":1,"syntology":null},{"url":"/paper/scireplicate-bench-benchmarking-llms-in-agent","slug":"scireplicate-bench-benchmarking-llms-in-agent","title":"SciReplicate-Bench: Benchmarking LLMs in Agent-driven Algorithmic Reproduction from Research Papers","date":"2025-03-31","arxiv_id":"2504.00255","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-anomaly-detection-in","slug":"unsupervised-anomaly-detection-in","title":"Unsupervised Anomaly Detection in Multivariate Time Series across Heterogeneous Domains","date":"2025-03-29","arxiv_id":"2503.23060","repositories_listed":1,"syntology":null},{"url":"/paper/egotom-benchmarking-theory-of-mind-reasoning","slug":"egotom-benchmarking-theory-of-mind-reasoning","title":"EgoToM: Benchmarking Theory of Mind Reasoning from Egocentric Videos","date":"2025-03-28","arxiv_id":"2503.22152","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/egotom-benchmarking-theory-of-mind-reasoning#ran","syntology_url":"https://syntology.ai/paper/2503.22152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.22152"}},"official":{"repos":["facebookresearch/egotom"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/why-stop-at-one-error-benchmarking-llms-as","slug":"why-stop-at-one-error-benchmarking-llms-as","title":"Why Stop at One Error? Benchmarking LLMs as Data Science Code Debuggers for Multi-Hop and Multi-Bug Errors","date":"2025-03-28","arxiv_id":"2503.22388","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-benchmark-for-rna-3d","slug":"a-comprehensive-benchmark-for-rna-3d","title":"A Comprehensive Benchmark for RNA 3D Structure-Function Modeling","date":"2025-03-27","arxiv_id":"2503.21681","repositories_listed":1,"syntology":null},{"url":"/paper/claimcheck-how-grounded-are-llm-critiques-of","slug":"claimcheck-how-grounded-are-llm-critiques-of","title":"CLAIMCHECK: How Grounded are LLM Critiques of Scientific Papers?","date":"2025-03-27","arxiv_id":"2503.21717","repositories_listed":1,"syntology":null},{"url":"/paper/facebench-a-multi-view-multi-level-facial","slug":"facebench-a-multi-view-multi-level-facial","title":"FaceBench: A Multi-View Multi-Level Facial Attribute VQA Dataset for Benchmarking Face Perception MLLMs","date":"2025-03-27","arxiv_id":"2503.21457","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/facebench-a-multi-view-multi-level-facial#ran","syntology_url":"https://syntology.ai/paper/2503.21457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21457"}},"official":{"repos":["cvi-szu/facebench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-and-optimizing-organism-wide","slug":"benchmarking-and-optimizing-organism-wide","title":"Benchmarking and optimizing organism wide single-cell RNA alignment methods","date":"2025-03-26","arxiv_id":"2503.20730","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-and-optimizing-organism-wide#ran","syntology_url":"https://syntology.ai/paper/2503.20730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20730"}},"official":{"repos":["phenomicai/bascvi"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-geometric-combinatorics-improve-rna","slug":"can-geometric-combinatorics-improve-rna","title":"Can geometric combinatorics improve RNA branching predictions?","date":"2025-03-26","arxiv_id":"2503.20977","repositories_listed":1,"syntology":null},{"url":"/paper/stabletoolbench-mirrorapi-modeling-tool","slug":"stabletoolbench-mirrorapi-modeling-tool","title":"StableToolBench-MirrorAPI: Modeling Tool Environments as Mirrors of 7,000+ Real-World APIs","date":"2025-03-26","arxiv_id":"2503.20527","repositories_listed":1,"syntology":null},{"url":"/paper/terratorch-the-geospatial-foundation-models","slug":"terratorch-the-geospatial-foundation-models","title":"TerraTorch: The Geospatial Foundation Models Toolkit","date":"2025-03-26","arxiv_id":"2503.20563","repositories_listed":1,"syntology":null}],"record_sha256":"30dcec7709e8fce42e1b16438bb54f9c0a089bf45fb32e19f63d69c198527b98","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}