{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/5","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":56,"rows_per_page":100,"rows":[401,500],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/4","next":"/task/benchmarking/papers/6","papers":[{"url":"/paper/light-field-saliency-detection-with-deep","slug":"light-field-saliency-detection-with-deep","title":"Light Field Saliency Detection with Deep Convolutional Networks","date":"2019-06-19","arxiv_id":"1906.08331","repositories_listed":2,"syntology":null},{"url":"/paper/pyrobot-an-open-source-robotics-framework-for","slug":"pyrobot-an-open-source-robotics-framework-for","title":"PyRobot: An Open-source Robotics Framework for Research and Benchmarking","date":"2019-06-19","arxiv_id":"1906.08236","repositories_listed":2,"syntology":null},{"url":"/paper/mnist-c-a-robustness-benchmark-for-computer","slug":"mnist-c-a-robustness-benchmark-for-computer","title":"MNIST-C: A Robustness Benchmark for Computer Vision","date":"2019-06-05","arxiv_id":"1906.02337","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mnist-c-a-robustness-benchmark-for-computer#ran","syntology_url":"https://syntology.ai/paper/1906.02337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.02337"}},"official":{"repos":["google-research/mnist-c"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/scaling-and-benchmarking-self-supervised","slug":"scaling-and-benchmarking-self-supervised","title":"Scaling and Benchmarking Self-Supervised Visual Representation Learning","date":"2019-05-03","arxiv_id":"1905.01235","repositories_listed":2,"syntology":null},{"url":"/paper/coco-the-large-scale-black-box-optimization","slug":"coco-the-large-scale-black-box-optimization","title":"COCO: The Large Scale Black-Box Optimization Benchmarking (bbob-largescale) Test Suite","date":"2019-03-15","arxiv_id":"1903.06396","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-reinforcement-learning","slug":"benchmarking-reinforcement-learning","title":"Benchmarking Reinforcement Learning Algorithms on Real-World Robots","date":"2018-09-20","arxiv_id":"1809.07731","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1809.07731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.07731"}},"official":{"repos":["kindredresearch/SenseAct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/user-guided-deep-anime-line-art-colorization","slug":"user-guided-deep-anime-line-art-colorization","title":"User-Guided Deep Anime Line Art Colorization with Conditional Adversarial Networks","date":"2018-08-09","arxiv_id":"1808.03240","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/user-guided-deep-anime-line-art-colorization#ran","syntology_url":"https://syntology.ai/paper/1808.03240","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.03240"}},"official":{"repos":["orashi/AlacGAN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ann-benchmarks-a-benchmarking-tool-for","slug":"ann-benchmarks-a-benchmarking-tool-for","title":"ANN-Benchmarks: A Benchmarking Tool for Approximate Nearest Neighbor Algorithms","date":"2018-07-15","arxiv_id":"1807.05614","repositories_listed":2,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ann-benchmarks-a-benchmarking-tool-for#ran","syntology_url":"https://syntology.ai/paper/1807.05614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.05614"}},"official":null}},{"url":"/paper/benchmarking-neural-network-robustness-to","slug":"benchmarking-neural-network-robustness-to","title":"Benchmarking Neural Network Robustness to Common Corruptions and Surface Variations","date":"2018-07-04","arxiv_id":"1807.01697","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-neural-network-robustness-to#ran","syntology_url":"https://syntology.ai/paper/1807.01697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.01697"}},"official":null}},{"url":"/paper/pico-element-detection-in-medical-text-via","slug":"pico-element-detection-in-medical-text-via","title":"PICO Element Detection in Medical Text via Long Short-Term Memory Neural Networks","date":"2018-07-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/hyperspectral-image-dataset-for-benchmarking","slug":"hyperspectral-image-dataset-for-benchmarking","title":"Hyperspectral Image Dataset for Benchmarking on Salient Object Detection","date":"2018-06-29","arxiv_id":"1806.11314","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyperspectral-image-dataset-for-benchmarking#ran","syntology_url":"https://syntology.ai/paper/1806.11314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.11314"}},"official":{"repos":["gistairc/HS-SOD"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/real-time-cryo-em-data-pre-processing-with","slug":"real-time-cryo-em-data-pre-processing-with","title":"Real-time cryo-EM data pre-processing with Warp","date":"2018-06-14","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-general-video","slug":"deep-reinforcement-learning-for-general-video","title":"Deep Reinforcement Learning for General Video Game AI","date":"2018-06-06","arxiv_id":"1806.02448","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-general-video#ran","syntology_url":"https://syntology.ai/paper/1806.02448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02448"}},"official":{"repos":["rubenrtorrado/GVGAI_GYM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/resource-interoperability-for-sustainable","slug":"resource-interoperability-for-sustainable","title":"Resource Interoperability for Sustainable Benchmarking: The Case of Events","date":"2018-05-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/revisiting-oxford-and-paris-large-scale-image","slug":"revisiting-oxford-and-paris-large-scale-image","title":"Revisiting Oxford and Paris: Large-Scale Image Retrieval Benchmarking","date":"2018-03-29","arxiv_id":"1803.11285","repositories_listed":2,"syntology":null},{"url":"/paper/rtseg-real-time-semantic-segmentation","slug":"rtseg-real-time-semantic-segmentation","title":"RTSeg: Real-time Semantic Segmentation Comparative Study","date":"2018-03-07","arxiv_id":"1803.02758","repositories_listed":2,"syntology":null},{"url":"/paper/tunability-importance-of-hyperparameters-of","slug":"tunability-importance-of-hyperparameters-of","title":"Tunability: Importance of Hyperparameters of Machine Learning Algorithms","date":"2018-02-26","arxiv_id":"1802.09596","repositories_listed":2,"syntology":null},{"url":"/paper/tap-dlnd-10-a-corpus-for-document-level","slug":"tap-dlnd-10-a-corpus-for-document-level","title":"TAP-DLND 1.0 : A Corpus for Document Level Novelty Detection","date":"2018-02-20","arxiv_id":"1802.06950","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-framework-for-performance","slug":"benchmarking-framework-for-performance","title":"Benchmarking Framework for Performance-Evaluation of Causal Inference Analysis","date":"2018-02-14","arxiv_id":"1802.05046","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-relief-based-feature-selection","slug":"benchmarking-relief-based-feature-selection","title":"Benchmarking Relief-Based Feature Selection Methods for Bioinformatics Data Mining","date":"2017-11-22","arxiv_id":"1711.08477","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-6dof-outdoor-visual-localization","slug":"benchmarking-6dof-outdoor-visual-localization","title":"Benchmarking 6DOF Outdoor Visual Localization in Changing Conditions","date":"2017-07-28","arxiv_id":"1707.09092","repositories_listed":2,"syntology":null},{"url":"/paper/the-biglasso-package-a-memory-and-computation","slug":"the-biglasso-package-a-memory-and-computation","title":"The biglasso Package: A Memory- and Computation-Efficient Solver for Lasso Model Fitting with Big Data in R","date":"2017-01-20","arxiv_id":"1701.05936","repositories_listed":2,"syntology":null},{"url":"/paper/multiple-instance-learning-a-survey-of","slug":"multiple-instance-learning-a-survey-of","title":"Multiple Instance Learning: A Survey of Problem Characteristics and Applications","date":"2016-12-11","arxiv_id":"1612.03365","repositories_listed":2,"syntology":null},{"url":"/paper/the-freiburg-groceries-dataset","slug":"the-freiburg-groceries-dataset","title":"The Freiburg Groceries Dataset","date":"2016-11-17","arxiv_id":"1611.05799","repositories_listed":2,"syntology":null},{"url":"/paper/yum-me-a-personalized-nutrient-based-meal","slug":"yum-me-a-personalized-nutrient-based-meal","title":"Yum-me: A Personalized Nutrient-based Meal Recommender System","date":"2016-05-25","arxiv_id":"1605.07722","repositories_listed":2,"syntology":null},{"url":"/paper/building-a-large-scale-dataset-for-image","slug":"building-a-large-scale-dataset-for-image","title":"Building a Large Scale Dataset for Image Emotion Recognition: The Fine Print and The Benchmark","date":"2016-05-09","arxiv_id":"1605.02677","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/building-a-large-scale-dataset-for-image#ran","syntology_url":"https://syntology.ai/paper/1605.02677","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1605.02677"}},"official":null}},{"url":"/paper/benchmarking-sentiment-analysis-methods-for","slug":"benchmarking-sentiment-analysis-methods-for","title":"Benchmarking sentiment analysis methods for large-scale texts: A case for using continuum-scored words and word shift graphs","date":"2015-12-02","arxiv_id":"1512.00531","repositories_listed":2,"syntology":null},{"url":"/paper/znn-a-fast-and-scalable-algorithm-for","slug":"znn-a-fast-and-scalable-algorithm-for","title":"ZNN - A Fast and Scalable Algorithm for Training 3D Convolutional Networks on Multi-Core and Many-Core Shared Memory Machines","date":"2015-10-22","arxiv_id":"1510.06706","repositories_listed":2,"syntology":null},{"url":"/paper/hyperopt-sklearn-automatic-hyperparameter","slug":"hyperopt-sklearn-automatic-hyperparameter","title":"Hyperopt-Sklearn: Automatic Hyperparameter Configuration for Scikit-Learn","date":"2014-01-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/dvfl-net-a-lightweight-distilled-video-focal","slug":"dvfl-net-a-lightweight-distilled-video-focal","title":"DVFL-Net: A Lightweight Distilled Video Focal Modulation Network for Spatio-Temporal Action Recognition","date":"2025-07-16","arxiv_id":"2507.12426","repositories_listed":1,"syntology":null},{"url":"/paper/dcr-quantifying-data-contamination-in-llms","slug":"dcr-quantifying-data-contamination-in-llms","title":"DCR: Quantifying Data Contamination in LLMs Evaluation","date":"2025-07-15","arxiv_id":"2507.11405","repositories_listed":1,"syntology":null},{"url":"/paper/drafterbench-benchmarking-large-language","slug":"drafterbench-benchmarking-large-language","title":"DrafterBench: Benchmarking Large Language Models for Tasks Automation in Civil Engineering","date":"2025-07-15","arxiv_id":"2507.11527","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/drafterbench-benchmarking-large-language#ran","syntology_url":"https://syntology.ai/paper/2507.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.11527"}},"official":{"repos":["eason-li-ais/drafterbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/flsim-a-modular-and-library-agnostic","slug":"flsim-a-modular-and-library-agnostic","title":"FLsim: A Modular and Library-Agnostic Simulation Framework for Federated Learning","date":"2025-07-15","arxiv_id":"2507.11430","repositories_listed":1,"syntology":null},{"url":"/paper/ref-long-benchmarking-the-long-context","slug":"ref-long-benchmarking-the-long-context","title":"Ref-Long: Benchmarking the Long-context Referencing Capability of Long-context Language Models","date":"2025-07-13","arxiv_id":"2507.09506","repositories_listed":1,"syntology":null},{"url":"/paper/identifying-the-smallest-adversarial-load","slug":"identifying-the-smallest-adversarial-load","title":"Identifying the Smallest Adversarial Load Perturbations that Render DC-OPF Infeasible","date":"2025-07-10","arxiv_id":"2507.07850","repositories_listed":1,"syntology":null},{"url":"/paper/orchestrator-agent-trust-a-modular-agentic-ai","slug":"orchestrator-agent-trust-a-modular-agentic-ai","title":"Orchestrator-Agent Trust: A Modular Agentic AI Visual Classification System with Trust-Aware Orchestration and RAG-Based Reasoning","date":"2025-07-09","arxiv_id":"2507.10571","repositories_listed":1,"syntology":null},{"url":"/paper/senseshift6d-multimodal-rgb-d-benchmarking","slug":"senseshift6d-multimodal-rgb-d-benchmarking","title":"SenseShift6D: Multimodal RGB-D Benchmarking for Robust 6D Pose Estimation across Environment and Sensor Variations","date":"2025-07-08","arxiv_id":"2507.05751","repositories_listed":1,"syntology":null},{"url":"/paper/llmthinkbench-towards-basic-math-reasoning","slug":"llmthinkbench-towards-basic-math-reasoning","title":"LLMThinkBench: Towards Basic Math Reasoning and Overthinking in Large Language Models","date":"2025-07-05","arxiv_id":"2507.04023","repositories_listed":1,"syntology":null},{"url":"/paper/gdgb-a-benchmark-for-generative-dynamic-text","slug":"gdgb-a-benchmark-for-generative-dynamic-text","title":"GDGB: A Benchmark for Generative Dynamic Text-Attributed Graph Learning","date":"2025-07-04","arxiv_id":"2507.03267","repositories_listed":1,"syntology":null},{"url":"/paper/structsense-a-task-agnostic-agentic-framework","slug":"structsense-a-task-agnostic-agentic-framework","title":"STRUCTSENSE: A Task-Agnostic Agentic Framework for Structured Information Extraction with Human-In-The-Loop Evaluation and Benchmarking","date":"2025-07-04","arxiv_id":"2507.03674","repositories_listed":1,"syntology":null},{"url":"/paper/lantern-a-machine-learning-framework-for","slug":"lantern-a-machine-learning-framework-for","title":"LANTERN: A Machine Learning Framework for Lipid Nanoparticle Transfection Efficiency Prediction","date":"2025-07-03","arxiv_id":"2507.03209","repositories_listed":1,"syntology":null},{"url":"/paper/latent-thermodynamic-flows-unified","slug":"latent-thermodynamic-flows-unified","title":"Latent Thermodynamic Flows: Unified Representation Learning and Generative Modeling of Temperature-Dependent Behaviors from Limited Data","date":"2025-07-03","arxiv_id":"2507.03174","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-deep-learning-and-vision","slug":"benchmarking-deep-learning-and-vision","title":"Benchmarking Deep Learning and Vision Foundation Models for Atypical vs. Normal Mitosis Classification with Cross-Dataset Evaluation","date":"2025-06-26","arxiv_id":"2506.21444","repositories_listed":1,"syntology":null},{"url":"/paper/covdocker-benchmarking-covalent-drug-design","slug":"covdocker-benchmarking-covalent-drug-design","title":"CovDocker: Benchmarking Covalent Drug Design with Tasks, Datasets, and Solutions","date":"2025-06-26","arxiv_id":"2506.21085","repositories_listed":1,"syntology":null},{"url":"/paper/mtsbench-benchmarking-multivariate-time","slug":"mtsbench-benchmarking-multivariate-time","title":"mTSBench: Benchmarking Multivariate Time Series Anomaly Detection and Model Selection at Scale","date":"2025-06-26","arxiv_id":"2506.21550","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-unsupervised-strategies-for","slug":"benchmarking-unsupervised-strategies-for","title":"Benchmarking Unsupervised Strategies for Anomaly Detection in Multivariate Time Series","date":"2025-06-25","arxiv_id":"2506.20574","repositories_listed":1,"syntology":null},{"url":"/paper/hribench-benchmarking-vision-language-models","slug":"hribench-benchmarking-vision-language-models","title":"HRIBench: Benchmarking Vision-Language Models for Real-Time Human Perception in Human-Robot Interaction","date":"2025-06-25","arxiv_id":"2506.20566","repositories_listed":1,"syntology":null},{"url":"/paper/inmotifin-a-lightweight-end-to-end-simulation","slug":"inmotifin-a-lightweight-end-to-end-simulation","title":"inMOTIFin: a lightweight end-to-end simulation software for regulatory sequences","date":"2025-06-25","arxiv_id":"2506.20769","repositories_listed":1,"syntology":null},{"url":"/paper/wattsonai-measuring-analyzing-and-visualizing","slug":"wattsonai-measuring-analyzing-and-visualizing","title":"WattsOnAI: Measuring, Analyzing, and Visualizing Energy and Carbon Footprint of AI Workloads","date":"2025-06-25","arxiv_id":"2506.20535","repositories_listed":1,"syntology":null},{"url":"/paper/pocketvina-enables-scalable-and-highly","slug":"pocketvina-enables-scalable-and-highly","title":"PocketVina Enables Scalable and Highly Accurate Physically Valid Docking through Multi-Pocket Conditioning","date":"2025-06-24","arxiv_id":"2506.20043","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-histopathology-foundation-models-1","slug":"benchmarking-histopathology-foundation-models-1","title":"Benchmarking histopathology foundation models in a multi-center dataset for skin cancer subtyping","date":"2025-06-23","arxiv_id":"2506.18668","repositories_listed":1,"syntology":null},{"url":"/paper/statistical-multicriteria-evaluation-of-llm","slug":"statistical-multicriteria-evaluation-of-llm","title":"Statistical Multicriteria Evaluation of LLM-Generated Text","date":"2025-06-22","arxiv_id":"2506.18082","repositories_listed":1,"syntology":null},{"url":"/paper/consumerbench-benchmarking-generative-ai","slug":"consumerbench-benchmarking-generative-ai","title":"ConsumerBench: Benchmarking Generative AI Applications on End-User Devices","date":"2025-06-21","arxiv_id":"2506.17538","repositories_listed":1,"syntology":null},{"url":"/paper/tabarena-a-living-benchmark-for-machine","slug":"tabarena-a-living-benchmark-for-machine","title":"TabArena: A Living Benchmark for Machine Learning on Tabular Data","date":"2025-06-20","arxiv_id":"2506.16791","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tabarena-a-living-benchmark-for-machine#ran","syntology_url":"https://syntology.ai/paper/2506.16791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.16791"}},"official":null}},{"url":"/paper/instructttseval-benchmarking-complex-natural","slug":"instructttseval-benchmarking-complex-natural","title":"InstructTTSEval: Benchmarking Complex Natural-Language Instruction Following in Text-to-Speech Systems","date":"2025-06-19","arxiv_id":"2506.16381","repositories_listed":1,"syntology":null},{"url":"/paper/bmfm-rna-an-open-framework-for-building-and","slug":"bmfm-rna-an-open-framework-for-building-and","title":"BMFM-RNA: An Open Framework for Building and Evaluating Transcriptomic Foundation Models","date":"2025-06-17","arxiv_id":"2506.14861","repositories_listed":1,"syntology":null},{"url":"/paper/gui-robust-a-comprehensive-dataset-for","slug":"gui-robust-a-comprehensive-dataset-for","title":"GUI-Robust: A Comprehensive Dataset for Testing GUI Agent Robustness in Real-World Anomalies","date":"2025-06-17","arxiv_id":"2506.14477","repositories_listed":1,"syntology":null},{"url":"/paper/impliret-benchmarking-the-implicit-fact","slug":"impliret-benchmarking-the-implicit-fact","title":"ImpliRet: Benchmarking the Implicit Fact Retrieval Challenge","date":"2025-06-17","arxiv_id":"2506.14407","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/impliret-benchmarking-the-implicit-fact#ran","syntology_url":"https://syntology.ai/paper/2506.14407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.14407"}},"official":{"repos":["zeinabtaghavi/impliret"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/c-tlsan-content-enhanced-time-aware-long-and","slug":"c-tlsan-content-enhanced-time-aware-long-and","title":"C-TLSAN: Content-Enhanced Time-Aware Long- and Short-Term Attention Network for Personalized Recommendation","date":"2025-06-16","arxiv_id":"2506.13021","repositories_listed":1,"syntology":null},{"url":"/paper/the-price-of-freedom-exploring-expressivity","slug":"the-price-of-freedom-exploring-expressivity","title":"The Price of Freedom: Exploring Expressivity and Runtime Tradeoffs in Equivariant Tensor Products","date":"2025-06-16","arxiv_id":"2506.13523","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-price-of-freedom-exploring-expressivity#ran","syntology_url":"https://syntology.ai/paper/2506.13523","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13523"}},"official":{"repos":["atomicarchitects/priceoffreedom"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/complexbench-edit-benchmarking-complex","slug":"complexbench-edit-benchmarking-complex","title":"ComplexBench-Edit: Benchmarking Complex Instruction-Driven Image Editing via Compositional Dependencies","date":"2025-06-15","arxiv_id":"2506.12830","repositories_listed":1,"syntology":null},{"url":"/paper/mldebugging-towards-benchmarking-code","slug":"mldebugging-towards-benchmarking-code","title":"MLDebugging: Towards Benchmarking Code Debugging Across Multi-Library Scenarios","date":"2025-06-15","arxiv_id":"2506.13824","repositories_listed":1,"syntology":null},{"url":"/paper/anira-an-architecture-for-neural-network","slug":"anira-an-architecture-for-neural-network","title":"ANIRA: An Architecture for Neural Network Inference in Real-Time Audio Applications","date":"2025-06-14","arxiv_id":"2506.12665","repositories_listed":1,"syntology":null},{"url":"/paper/delving-into-instance-dependent-label-noise","slug":"delving-into-instance-dependent-label-noise","title":"Delving into Instance-Dependent Label Noise in Graph Data: A Comprehensive Study and Benchmark","date":"2025-06-14","arxiv_id":"2506.12468","repositories_listed":1,"syntology":null},{"url":"/paper/openunlearning-accelerating-llm-unlearning","slug":"openunlearning-accelerating-llm-unlearning","title":"OpenUnlearning: Accelerating LLM Unlearning via Unified Benchmarking of Methods and Metrics","date":"2025-06-14","arxiv_id":"2506.12618","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/openunlearning-accelerating-llm-unlearning#ran","syntology_url":"https://syntology.ai/paper/2506.12618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12618"}},"official":{"repos":["locuslab/open-unlearning"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mind-the-xai-gap-a-human-centered-llm","slug":"mind-the-xai-gap-a-human-centered-llm","title":"Mind the XAI Gap: A Human-Centered LLM Framework for Democratizing Explainable AI","date":"2025-06-13","arxiv_id":"2506.12240","repositories_listed":1,"syntology":null},{"url":"/paper/sec-bench-automated-benchmarking-of-llm","slug":"sec-bench-automated-benchmarking-of-llm","title":"SEC-bench: Automated Benchmarking of LLM Agents on Real-World Software Security Tasks","date":"2025-06-13","arxiv_id":"2506.11791","repositories_listed":1,"syntology":null},{"url":"/paper/sdialog-a-python-toolkit-for-synthetic","slug":"sdialog-a-python-toolkit-for-synthetic","title":"SDialog: A Python Toolkit for Synthetic Dialogue Generation and Analysis","date":"2025-06-12","arxiv_id":"2506.10622","repositories_listed":1,"syntology":null},{"url":"/paper/2506-10117","slug":"2506-10117","title":"A Manually Annotated Image-Caption Dataset for Detecting Children in the Wild","date":"2025-06-11","arxiv_id":"2506.10117","repositories_listed":1,"syntology":null},{"url":"/paper/attention-please-revisiting-attentive-probing","slug":"attention-please-revisiting-attentive-probing","title":"Attention, Please! Revisiting Attentive Probing for Masked Image Modeling","date":"2025-06-11","arxiv_id":"2506.10178","repositories_listed":1,"syntology":null},{"url":"/paper/glgenn-a-novel-parameter-light-equivariant","slug":"glgenn-a-novel-parameter-light-equivariant","title":"GLGENN: A Novel Parameter-Light Equivariant Neural Networks Architecture Based on Clifford Geometric Algebras","date":"2025-06-11","arxiv_id":"2506.09625","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/glgenn-a-novel-parameter-light-equivariant#ran","syntology_url":"https://syntology.ai/paper/2506.09625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09625"}},"official":{"repos":["katyafilimoshina/glgenn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hopadiff-holistic-partial-aware-fourier","slug":"hopadiff-holistic-partial-aware-fourier","title":"HopaDIFF: Holistic-Partial Aware Fourier Conditioned Diffusion for Referring Human Action Segmentation in Multi-Person Scenarios","date":"2025-06-11","arxiv_id":"2506.09650","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":6,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 2 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hopadiff-holistic-partial-aware-fourier#ran","syntology_url":"https://syntology.ai/paper/2506.09650","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09650"}},"official":{"repos":["kpeng9510/hopadiff"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/intphys-2-benchmarking-intuitive-physics","slug":"intphys-2-benchmarking-intuitive-physics","title":"IntPhys 2: Benchmarking Intuitive Physics Understanding In Complex Synthetic Environments","date":"2025-06-11","arxiv_id":"2506.09849","repositories_listed":1,"syntology":null},{"url":"/paper/2506-10031","slug":"2506-10031","title":"scSSL-Bench: Benchmarking Self-Supervised Learning for Single-Cell Data","date":"2025-06-10","arxiv_id":"2506.10031","repositories_listed":1,"syntology":null},{"url":"/paper/counselbench-a-large-scale-expert-evaluation","slug":"counselbench-a-large-scale-expert-evaluation","title":"CounselBench: A Large-Scale Expert Evaluation and Adversarial Benchmark of Large Language Models in Mental Health Counseling","date":"2025-06-10","arxiv_id":"2506.08584","repositories_listed":1,"syntology":null},{"url":"/paper/2506-08249","slug":"2506-08249","title":"RADAR: Benchmarking Language Models on Imperfect Tabular Data","date":"2025-06-09","arxiv_id":"2506.08249","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2506-08249#ran","syntology_url":"https://syntology.ai/paper/2506.08249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08249"}},"official":{"repos":["kenqgu/radar"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/can-ai-validate-science-benchmarking-llms-for","slug":"can-ai-validate-science-benchmarking-llms-for","title":"Can AI Validate Science? Benchmarking LLMs for Accurate Scientific Claim $\\rightarrow$ Evidence Reasoning","date":"2025-06-09","arxiv_id":"2506.08235","repositories_listed":1,"syntology":null},{"url":"/paper/cure-cultural-gaps-in-the-long-tail-of-text","slug":"cure-cultural-gaps-in-the-long-tail-of-text","title":"CuRe: Cultural Gaps in the Long Tail of Text-to-Image Systems","date":"2025-06-09","arxiv_id":"2506.08071","repositories_listed":1,"syntology":null},{"url":"/paper/husc3d-human-sculpture-dataset-for-3d-object","slug":"husc3d-human-sculpture-dataset-for-3d-object","title":"HuSc3D: Human Sculpture dataset for 3D object reconstruction","date":"2025-06-09","arxiv_id":"2506.07628","repositories_listed":1,"syntology":null},{"url":"/paper/the-catechol-benchmark-time-series-solvent","slug":"the-catechol-benchmark-time-series-solvent","title":"The Catechol Benchmark: Time-series Solvent Selection Data for Few-shot Machine Learning","date":"2025-06-09","arxiv_id":"2506.07619","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-catechol-benchmark-time-series-solvent#ran","syntology_url":"https://syntology.ai/paper/2506.07619","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07619"}},"official":{"repos":["jpfolch/catechol_solvent_selection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-far-are-we-from-optimal-reasoning","slug":"how-far-are-we-from-optimal-reasoning","title":"How Far Are We from Optimal Reasoning Efficiency?","date":"2025-06-08","arxiv_id":"2506.07104","repositories_listed":1,"syntology":null},{"url":"/paper/loopdb-a-loop-closure-dataset-for-large-scale","slug":"loopdb-a-loop-closure-dataset-for-large-scale","title":"LoopDB: A Loop Closure Dataset for Large Scale Simultaneous Localization and Mapping","date":"2025-06-07","arxiv_id":"2506.06771","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-misuse-mitigation-against-covert","slug":"benchmarking-misuse-mitigation-against-covert","title":"Benchmarking Misuse Mitigation Against Covert Adversaries","date":"2025-06-06","arxiv_id":"2506.06414","repositories_listed":1,"syntology":null},{"url":"/paper/financereasoning-benchmarking-financial","slug":"financereasoning-benchmarking-financial","title":"FinanceReasoning: Benchmarking Financial Numerical Reasoning More Credible, Comprehensive and Challenging","date":"2025-06-06","arxiv_id":"2506.05828","repositories_listed":1,"syntology":null},{"url":"/paper/bsbench-will-your-llm-find-the-largest-prime","slug":"bsbench-will-your-llm-find-the-largest-prime","title":"BSBench: will your LLM find the largest prime number?","date":"2025-06-05","arxiv_id":"2506.04535","repositories_listed":1,"syntology":null},{"url":"/paper/debatable-intelligence-benchmarking-llm","slug":"debatable-intelligence-benchmarking-llm","title":"Debatable Intelligence: Benchmarking LLM Judges via Debate Speech Evaluation","date":"2025-06-05","arxiv_id":"2506.05062","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/debatable-intelligence-benchmarking-llm#ran","syntology_url":"https://syntology.ai/paper/2506.05062","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.05062"}},"official":{"repos":["noy-sternlicht/debatable-intelligence"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/megahan97k-a-large-scale-dataset-for-mega","slug":"megahan97k-a-large-scale-dataset-for-mega","title":"MegaHan97K: A Large-Scale Dataset for Mega-Category Chinese Character Recognition with over 97K Categories","date":"2025-06-05","arxiv_id":"2506.04807","repositories_listed":1,"syntology":null},{"url":"/paper/mmtu-a-massive-multi-task-table-understanding","slug":"mmtu-a-massive-multi-task-table-understanding","title":"MMTU: A Massive Multi-Task Table Understanding and Reasoning Benchmark","date":"2025-06-05","arxiv_id":"2506.05587","repositories_listed":1,"syntology":null},{"url":"/paper/a-kernel-based-approach-for-accurate-steady","slug":"a-kernel-based-approach-for-accurate-steady","title":"A Kernel-Based Approach for Accurate Steady-State Detection in Performance Time Series","date":"2025-06-04","arxiv_id":"2506.04204","repositories_listed":1,"syntology":null},{"url":"/paper/assetopsbench-benchmarking-ai-agents-for-task","slug":"assetopsbench-benchmarking-ai-agents-for-task","title":"AssetOpsBench: Benchmarking AI Agents for Task Automation in Industrial Asset Operations and Maintenance","date":"2025-06-04","arxiv_id":"2506.03828","repositories_listed":1,"syntology":null},{"url":"/paper/hssbench-benchmarking-humanities-and-social","slug":"hssbench-benchmarking-humanities-and-social","title":"HSSBench: Benchmarking Humanities and Social Sciences Ability for Multimodal Large Language Models","date":"2025-06-04","arxiv_id":"2506.03922","repositories_listed":1,"syntology":null},{"url":"/paper/macosworld-a-multilingual-interactive","slug":"macosworld-a-multilingual-interactive","title":"macOSWorld: A Multilingual Interactive Benchmark for GUI Agents","date":"2025-06-04","arxiv_id":"2506.04135","repositories_listed":1,"syntology":null},{"url":"/paper/n-2-a-unified-python-package-and-test-bench","slug":"n-2-a-unified-python-package-and-test-bench","title":"N$^2$: A Unified Python Package and Test Bench for Nearest Neighbor-Based Matrix Completion","date":"2025-06-04","arxiv_id":"2506.04166","repositories_listed":1,"syntology":null},{"url":"/paper/bytemorph-benchmarking-instruction-guided","slug":"bytemorph-benchmarking-instruction-guided","title":"ByteMorph: Benchmarking Instruction-Guided Image Editing with Non-Rigid Motions","date":"2025-06-03","arxiv_id":"2506.03107","repositories_listed":1,"syntology":null},{"url":"/paper/failuresensoriq-a-multi-choice-qa-dataset-for","slug":"failuresensoriq-a-multi-choice-qa-dataset-for","title":"FailureSensorIQ: A Multi-Choice QA Dataset for Understanding Sensor Relationships and Failure Modes","date":"2025-06-03","arxiv_id":"2506.03278","repositories_listed":1,"syntology":null},{"url":"/paper/netpress-dynamically-generated-llm-benchmarks","slug":"netpress-dynamically-generated-llm-benchmarks","title":"NetPress: Dynamically Generated LLM Benchmarks for Network Applications","date":"2025-06-03","arxiv_id":"2506.03231","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-machine-unlearning-in-image","slug":"rethinking-machine-unlearning-in-image","title":"Rethinking Machine Unlearning in Image Generation Models","date":"2025-06-03","arxiv_id":"2506.02761","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rethinking-machine-unlearning-in-image#ran","syntology_url":"https://syntology.ai/paper/2506.02761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.02761"}},"official":{"repos":["ryliu68/igmu"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cvc-a-large-scale-chinese-value-rule-corpus","slug":"cvc-a-large-scale-chinese-value-rule-corpus","title":"CVC: A Large-Scale Chinese Value Rule Corpus for Value Alignment of Large Language Models","date":"2025-06-02","arxiv_id":"2506.01495","repositories_listed":1,"syntology":null},{"url":"/paper/gscodec-studio-a-modular-framework-for","slug":"gscodec-studio-a-modular-framework-for","title":"GSCodec Studio: A Modular Framework for Gaussian Splat Compression","date":"2025-06-02","arxiv_id":"2506.01822","repositories_listed":1,"syntology":null},{"url":"/paper/access-denied-inc-the-first-benchmark","slug":"access-denied-inc-the-first-benchmark","title":"ACCESS DENIED INC: The First Benchmark Environment for Sensitivity Awareness","date":"2025-06-01","arxiv_id":"2506.00964","repositories_listed":1,"syntology":null}],"record_sha256":"0e09e629258ded353c0bfab6769853f1551c8465ffc023e3dbb536ac8951a114","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}