{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/12","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":56,"rows_per_page":100,"rows":[1101,1200],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/11","next":"/task/benchmarking/papers/13","papers":[{"url":"/paper/advances-in-appfl-a-comprehensive-and","slug":"advances-in-appfl-a-comprehensive-and","title":"Advances in APPFL: A Comprehensive and Extensible Federated Learning Framework","date":"2024-09-17","arxiv_id":"2409.11585","repositories_listed":1,"syntology":null},{"url":"/paper/improve-machine-learning-carbon-footprint-1","slug":"improve-machine-learning-carbon-footprint-1","title":"Improve Machine Learning carbon footprint using Parquet dataset format and Mixed Precision training for regression models -- Part II","date":"2024-09-17","arxiv_id":"2409.11071","repositories_listed":1,"syntology":null},{"url":"/paper/saged-a-holistic-bias-benchmarking-pipeline","slug":"saged-a-holistic-bias-benchmarking-pipeline","title":"SAGED: A Holistic Bias-Benchmarking Pipeline for Language Models with Customisable Fairness Calibration","date":"2024-09-17","arxiv_id":"2409.11149","repositories_listed":1,"syntology":null},{"url":"/paper/thames-an-end-to-end-tool-for-hallucination","slug":"thames-an-end-to-end-tool-for-hallucination","title":"THaMES: An End-to-End Tool for Hallucination Mitigation and Evaluation in Large Language Models","date":"2024-09-17","arxiv_id":"2409.11353","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/thames-an-end-to-end-tool-for-hallucination#ran","syntology_url":"https://syntology.ai/paper/2409.11353","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.11353"}},"official":{"repos":["holistic-ai/THaMES"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-sounds-of-home-a-speech-removed","slug":"the-sounds-of-home-a-speech-removed","title":"The Sounds of Home: A Speech-Removed Residential Audio Dataset for Sound Event Detection","date":"2024-09-17","arxiv_id":"2409.11262","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-model-uncertainty","slug":"benchmarking-large-language-model-uncertainty","title":"Benchmarking Large Language Model Uncertainty for Prompt Optimization","date":"2024-09-16","arxiv_id":"2409.10044","repositories_listed":1,"syntology":null},{"url":"/paper/metaformer-and-cnn-hybrid-model-for-polyp","slug":"metaformer-and-cnn-hybrid-model-for-polyp","title":"MetaFormer and CNN Hybrid Model for Polyp Image Segmentation","date":"2024-09-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/odaq-open-dataset-of-audio-quality-benchmark","slug":"odaq-open-dataset-of-audio-quality-benchmark","title":"ODAQ: Open Dataset of Audio Quality - Benchmark on GitHub","date":"2024-09-13","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/improve-machine-learning-carbon-footprint","slug":"improve-machine-learning-carbon-footprint","title":"Improve Machine Learning carbon footprint using Nvidia GPU and Mixed Precision training for classification models -- Part I","date":"2024-09-12","arxiv_id":"2409.07853","repositories_listed":1,"syntology":null},{"url":"/paper/linear-energy-storage-and-flexibility-model","slug":"linear-energy-storage-and-flexibility-model","title":"Linear energy storage and flexibility model with ramp rate, ramping, deadline and capacity constraints","date":"2024-09-12","arxiv_id":"2409.08084","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-novelty-detection-methods","slug":"unsupervised-novelty-detection-methods","title":"Unsupervised Novelty Detection Methods Benchmarking with Wavelet Decomposition","date":"2024-09-11","arxiv_id":"2409.07135","repositories_listed":1,"syntology":null},{"url":"/paper/mahalanobis-k-nn-a-statistical-lens-for","slug":"mahalanobis-k-nn-a-statistical-lens-for","title":"Mahalanobis k-NN: A Statistical Lens for Robust Point-Cloud Registrations","date":"2024-09-10","arxiv_id":"2409.06267","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-evaluating-pm2-5-forecasts","slug":"a-framework-for-evaluating-pm2-5-forecasts","title":"A Framework for Evaluating PM2.5 Forecasts from the Perspective of Individual Decision Making","date":"2024-09-09","arxiv_id":"2409.05866","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-chinese-knowledge-rectification","slug":"benchmarking-chinese-knowledge-rectification","title":"CKnowEdit: A New Chinese Knowledge Editing Dataset for Linguistics, Facts, and Logic Error Correction in LLMs","date":"2024-09-09","arxiv_id":"2409.05806","repositories_listed":1,"syntology":null},{"url":"/paper/insights-from-benchmarking-frontier-language","slug":"insights-from-benchmarking-frontier-language","title":"Insights from Benchmarking Frontier Language Models on Web App Code Generation","date":"2024-09-08","arxiv_id":"2409.05177","repositories_listed":1,"syntology":null},{"url":"/paper/plantseg-a-large-scale-in-the-wild-dataset","slug":"plantseg-a-large-scale-in-the-wild-dataset","title":"PlantSeg: A Large-Scale In-the-wild Dataset for Plant Disease Segmentation","date":"2024-09-06","arxiv_id":"2409.04038","repositories_listed":1,"syntology":null},{"url":"/paper/question-answering-dense-video-events","slug":"question-answering-dense-video-events","title":"Question-Answering Dense Video Events","date":"2024-09-06","arxiv_id":"2409.04388","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-spurious-bias-in-few-shot-image","slug":"benchmarking-spurious-bias-in-few-shot-image","title":"Benchmarking Spurious Bias in Few-Shot Image Classifiers","date":"2024-09-04","arxiv_id":"2409.02882","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-spurious-bias-in-few-shot-image#ran","syntology_url":"https://syntology.ai/paper/2409.02882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02882"}},"official":{"repos":["gtzheng/fewstab"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rtlrewriter-methodologies-for-large-models","slug":"rtlrewriter-methodologies-for-large-models","title":"RTLRewriter: Methodologies for Large Models aided RTL Code Optimization","date":"2024-09-04","arxiv_id":"2409.11414","repositories_listed":1,"syntology":null},{"url":"/paper/genagent-build-collaborative-ai-systems-with","slug":"genagent-build-collaborative-ai-systems-with","title":"ComfyBench: Benchmarking LLM-based Agents in ComfyUI for Autonomously Designing Collaborative AI Systems","date":"2024-09-02","arxiv_id":"2409.01392","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/genagent-build-collaborative-ai-systems-with#ran","syntology_url":"https://syntology.ai/paper/2409.01392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01392"}},"official":{"repos":["xxyQwQ/ComfyBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-student-actions-in-classroom-scenes","slug":"towards-student-actions-in-classroom-scenes","title":"Towards Student Actions in Classroom Scenes: New Dataset and Baseline","date":"2024-09-02","arxiv_id":"2409.00926","repositories_listed":1,"syntology":null},{"url":"/paper/syntheval-hybrid-behavioral-testing-of-nlp","slug":"syntheval-hybrid-behavioral-testing-of-nlp","title":"SYNTHEVAL: Hybrid Behavioral Testing of NLP Models with Synthetic CheckLists","date":"2024-08-30","arxiv_id":"2408.17437","repositories_listed":1,"syntology":null},{"url":"/paper/how-far-can-cantonese-nlp-go-benchmarking","slug":"how-far-can-cantonese-nlp-go-benchmarking","title":"How Well Do LLMs Handle Cantonese? Benchmarking Cantonese Capabilities of Large Language Models","date":"2024-08-29","arxiv_id":"2408.16756","repositories_listed":1,"syntology":null},{"url":"/paper/illuminating-the-diversity-fitness-trade-off","slug":"illuminating-the-diversity-fitness-trade-off","title":"Illuminating the Diversity-Fitness Trade-Off in Black-Box Optimization","date":"2024-08-29","arxiv_id":"2408.16393","repositories_listed":1,"syntology":null},{"url":"/paper/stereo-towards-adversarially-robust-concept","slug":"stereo-towards-adversarially-robust-concept","title":"STEREO: Towards Adversarially Robust Concept Erasing from Text-to-Image Generation Models","date":"2024-08-29","arxiv_id":"2408.16807","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/stereo-towards-adversarially-robust-concept#ran","syntology_url":"https://syntology.ai/paper/2408.16807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.16807"}},"official":{"repos":["koushiksrivats/robust-concept-erasing"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/interactive-agents-simulating-counselor","slug":"interactive-agents-simulating-counselor","title":"Interactive Agents: Simulating Counselor-Client Psychological Counseling via Role-Playing LLM-to-LLM Interactions","date":"2024-08-28","arxiv_id":"2408.15787","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interactive-agents-simulating-counselor#ran","syntology_url":"https://syntology.ai/paper/2408.15787","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15787"}},"official":{"repos":["qiuhuachuan/interactive-agents"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/logicgame-benchmarking-rule-based-reasoning","slug":"logicgame-benchmarking-rule-based-reasoning","title":"LogicGame: Benchmarking Rule-Based Reasoning Abilities of Large Language Models","date":"2024-08-28","arxiv_id":"2408.15778","repositories_listed":1,"syntology":null},{"url":"/paper/fasttextspotter-a-high-efficiency-transformer","slug":"fasttextspotter-a-high-efficiency-transformer","title":"FastTextSpotter: A High-Efficiency Transformer for Multilingual Scene Text Spotting","date":"2024-08-27","arxiv_id":"2408.14998","repositories_listed":1,"syntology":null},{"url":"/paper/comparative-analysis-violence-recognition","slug":"comparative-analysis-violence-recognition","title":"Comparative Analysis: Violence Recognition from Videos using Transfer Learning","date":"2024-08-26","arxiv_id":"2408.14659","repositories_listed":1,"syntology":null},{"url":"/paper/variational-autoencoder-for-anomaly-detection","slug":"variational-autoencoder-for-anomaly-detection","title":"Variational Autoencoder for Anomaly Detection: A Comparative Study","date":"2024-08-24","arxiv_id":"2408.13561","repositories_listed":1,"syntology":null},{"url":"/paper/scribbles-for-all-benchmarking-scribble","slug":"scribbles-for-all-benchmarking-scribble","title":"Scribbles for All: Benchmarking Scribble Supervised Segmentation Across Datasets","date":"2024-08-22","arxiv_id":"2408.12489","repositories_listed":1,"syntology":null},{"url":"/paper/wcebleedgen-a-wireless-capsule-endoscopy","slug":"wcebleedgen-a-wireless-capsule-endoscopy","title":"WCEbleedGen: A wireless capsule endoscopy dataset and its benchmarking for automatic bleeding classification, detection, and segmentation","date":"2024-08-22","arxiv_id":"2408.12466","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":17,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wcebleedgen-a-wireless-capsule-endoscopy#ran","syntology_url":"https://syntology.ai/paper/2408.12466","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12466"}},"official":{"repos":["misahub2023/benchmarking-codes-of-the-wcebleedgen-dataset"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/simbench-a-rule-based-multi-turn-interaction","slug":"simbench-a-rule-based-multi-turn-interaction","title":"SimBench: A Rule-Based Multi-Turn Interaction Benchmark for Evaluating an LLM's Ability to Generate Digital Twins","date":"2024-08-21","arxiv_id":"2408.11987","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-models-for-math","slug":"benchmarking-large-language-models-for-math","title":"Benchmarking Large Language Models for Math Reasoning Tasks","date":"2024-08-20","arxiv_id":"2408.10839","repositories_listed":1,"syntology":null},{"url":"/paper/perturbench-benchmarking-machine-learning","slug":"perturbench-benchmarking-machine-learning","title":"PerturBench: Benchmarking Machine Learning Models for Cellular Perturbation Analysis","date":"2024-08-20","arxiv_id":"2408.10609","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/perturbench-benchmarking-machine-learning#ran","syntology_url":"https://syntology.ai/paper/2408.10609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.10609"}},"official":{"repos":["altoslabs/perturbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/blade-benchmarking-language-model-agents-for","slug":"blade-benchmarking-language-model-agents-for","title":"BLADE: Benchmarking Language Model Agents for Data-Driven Science","date":"2024-08-19","arxiv_id":"2408.09667","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/blade-benchmarking-language-model-agents-for#ran","syntology_url":"https://syntology.ai/paper/2408.09667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09667"}},"official":{"repos":["behavioral-data/blade"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-quantum-machine-learning-kernel","slug":"benchmarking-quantum-machine-learning-kernel","title":"Benchmarking quantum machine learning kernel training for classification tasks","date":"2024-08-17","arxiv_id":"2408.10274","repositories_listed":1,"syntology":null},{"url":"/paper/ser-evals-in-domain-and-out-of-domain","slug":"ser-evals-in-domain-and-out-of-domain","title":"SER Evals: In-domain and Out-of-domain Benchmarking for Speech Emotion Recognition","date":"2024-08-14","arxiv_id":"2408.07851","repositories_listed":1,"syntology":null},{"url":"/paper/sustaindc-benchmarking-for-sustainable-data","slug":"sustaindc-benchmarking-for-sustainable-data","title":"SustainDC: Benchmarking for Sustainable Data Center Control","date":"2024-08-14","arxiv_id":"2408.07841","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sustaindc-benchmarking-for-sustainable-data#ran","syntology_url":"https://syntology.ai/paper/2408.07841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07841"}},"official":{"repos":["hewlettpackard/dc-rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tabularbench-benchmarking-adversarial","slug":"tabularbench-benchmarking-adversarial","title":"TabularBench: Benchmarking Adversarial Robustness for Tabular Deep Learning in Real-world Use-cases","date":"2024-08-14","arxiv_id":"2408.07579","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tabularbench-benchmarking-adversarial#ran","syntology_url":"https://syntology.ai/paper/2408.07579","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07579"}},"official":{"repos":["serval-uni-lu/tabularbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/xcompress-llm-assisted-python-based-text","slug":"xcompress-llm-assisted-python-based-text","title":"XCompress: LLM assisted Python-based text compression toolkit","date":"2024-08-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/the-impact-of-internal-variability-on","slug":"the-impact-of-internal-variability-on","title":"The impact of internal variability on benchmarking deep learning climate emulators","date":"2024-08-09","arxiv_id":"2408.05288","repositories_listed":1,"syntology":null},{"url":"/paper/uav-enhanced-combination-to-application","slug":"uav-enhanced-combination-to-application","title":"UAV-Enhanced Combination to Application: Comprehensive Analysis and Benchmarking of a Human Detection Dataset for Disaster Scenarios","date":"2024-08-09","arxiv_id":"2408.04922","repositories_listed":1,"syntology":null},{"url":"/paper/speech-massive-a-multilingual-speech-dataset","slug":"speech-massive-a-multilingual-speech-dataset","title":"Speech-MASSIVE: A Multilingual Speech Dataset for SLU and Beyond","date":"2024-08-07","arxiv_id":"2408.03900","repositories_listed":1,"syntology":null},{"url":"/paper/walledeval-a-comprehensive-safety-evaluation","slug":"walledeval-a-comprehensive-safety-evaluation","title":"WalledEval: A Comprehensive Safety Evaluation Toolkit for Large Language Models","date":"2024-08-07","arxiv_id":"2408.03837","repositories_listed":1,"syntology":null},{"url":"/paper/2408-03322","slug":"2408-03322","title":"Segment Anything in Medical Images and Videos: Benchmark and Deployment","date":"2024-08-06","arxiv_id":"2408.03322","repositories_listed":1,"syntology":null},{"url":"/paper/openomni-a-collaborative-open-source-tool-for","slug":"openomni-a-collaborative-open-source-tool-for","title":"OpenOmni: A Collaborative Open Source Tool for Building Future-Ready Multimodal Conversational Agents","date":"2024-08-06","arxiv_id":"2408.03047","repositories_listed":1,"syntology":null},{"url":"/paper/2408-02533","slug":"2408-02533","title":"LMEMs for post-hoc analysis of HPO Benchmarking","date":"2024-08-05","arxiv_id":"2408.02533","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01700","slug":"2408-01700","title":"Integrating Large Language Models and Knowledge Graphs for Extraction and Validation of Textual Test Data","date":"2024-08-03","arxiv_id":"2408.01700","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01716","slug":"2408-01716","title":"Visual-Inertial SLAM for Unstructured Outdoor Environments: Benchmarking the Benefits and Computational Costs of Loop Closing","date":"2024-08-03","arxiv_id":"2408.01716","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01262","slug":"2408-01262","title":"RAGEval: Scenario Specific RAG Evaluation Dataset Generation Framework","date":"2024-08-02","arxiv_id":"2408.01262","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01541","slug":"2408-01541","title":"Guardians of Image Quality: Benchmarking Defenses Against Adversarial Attacks on Image Quality Metrics","date":"2024-08-02","arxiv_id":"2408.01541","repositories_listed":1,"syntology":null},{"url":"/paper/dissecting-dissonance-benchmarking-large","slug":"dissecting-dissonance-benchmarking-large","title":"Dissecting Dissonance: Benchmarking Large Multimodal Models Against Self-Contradictory Instructions","date":"2024-08-02","arxiv_id":"2408.01091","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dissecting-dissonance-benchmarking-large#ran","syntology_url":"https://syntology.ai/paper/2408.01091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01091"}},"official":{"repos":["shiyegao/Self-Contradictory-Instructions-SCI"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/high-quality-ros-compatible-video-encoding","slug":"high-quality-ros-compatible-video-encoding","title":"High-Quality, ROS Compatible Video Encoding and Decoding for High-Definition Datasets","date":"2024-08-01","arxiv_id":"2408.00538","repositories_listed":1,"syntology":null},{"url":"/paper/appworld-a-controllable-world-of-apps-and","slug":"appworld-a-controllable-world-of-apps-and","title":"AppWorld: A Controllable World of Apps and People for Benchmarking Interactive Coding Agents","date":"2024-07-26","arxiv_id":"2407.18901","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/appworld-a-controllable-world-of-apps-and#ran","syntology_url":"https://syntology.ai/paper/2407.18901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18901"}},"official":{"repos":["stonybrooknlp/appworld"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-dependence-measures-to-prevent","slug":"benchmarking-dependence-measures-to-prevent","title":"Benchmarking Dependence Measures to Prevent Shortcut Learning in Medical Imaging","date":"2024-07-26","arxiv_id":"2407.18792","repositories_listed":1,"syntology":null},{"url":"/paper/is-larger-always-better-evaluating-and","slug":"is-larger-always-better-evaluating-and","title":"ClinicRealm: Re-evaluating Large Language Models with Conventional Machine Learning for Non-Generative Clinical Prediction Tasks","date":"2024-07-26","arxiv_id":"2407.18525","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-larger-always-better-evaluating-and#ran","syntology_url":"https://syntology.ai/paper/2407.18525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18525"}},"official":{"repos":["yhzhu99/ehr-llm-benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/officebench-benchmarking-language-agents","slug":"officebench-benchmarking-language-agents","title":"OfficeBench: Benchmarking Language Agents across Multiple Applications for Office Automation","date":"2024-07-26","arxiv_id":"2407.19056","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/officebench-benchmarking-language-agents#ran","syntology_url":"https://syntology.ai/paper/2407.19056","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.19056"}},"official":{"repos":["zlwang-cs/OfficeBench"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/voxsim-a-perceptual-voice-similarity-dataset","slug":"voxsim-a-perceptual-voice-similarity-dataset","title":"VoxSim: A perceptual voice similarity dataset","date":"2024-07-26","arxiv_id":"2407.18505","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/voxsim-a-perceptual-voice-similarity-dataset#ran","syntology_url":"https://syntology.ai/paper/2407.18505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18505"}},"official":{"repos":["kaistmm/voxsim_trainer"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/asep-benchmarking-deep-learning-methods-for","slug":"asep-benchmarking-deep-learning-methods-for","title":"AsEP: Benchmarking Deep Learning Methods for Antibody-specific Epitope Prediction","date":"2024-07-25","arxiv_id":"2407.18184","repositories_listed":1,"syntology":null},{"url":"/paper/mds-ed-multimodal-decision-support-in-the","slug":"mds-ed-multimodal-decision-support-in-the","title":"Enhancing clinical decision support with physiological waveforms -- a multimodal benchmark in emergency care","date":"2024-07-25","arxiv_id":"2407.17856","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mds-ed-multimodal-decision-support-in-the#ran","syntology_url":"https://syntology.ai/paper/2407.17856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17856"}},"official":{"repos":["ai4healthuol/mds-ed"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/humanvid-demystifying-training-data-for","slug":"humanvid-demystifying-training-data-for","title":"HumanVid: Demystifying Training Data for Camera-controllable Human Image Animation","date":"2024-07-24","arxiv_id":"2407.17438","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humanvid-demystifying-training-data-for#ran","syntology_url":"https://syntology.ai/paper/2407.17438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17438"}},"official":{"repos":["zhenzhiwang/humanvid"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aggregated-attributions-for-explanatory","slug":"aggregated-attributions-for-explanatory","title":"Aggregated Attributions for Explanatory Analysis of 3D Segmentation Models","date":"2024-07-23","arxiv_id":"2407.16653","repositories_listed":1,"syntology":null},{"url":"/paper/bones-a-benchmark-for-neural-estimation-of","slug":"bones-a-benchmark-for-neural-estimation-of","title":"BONES: a Benchmark fOr Neural Estimation of Shapley values","date":"2024-07-23","arxiv_id":"2407.16482","repositories_listed":1,"syntology":null},{"url":"/paper/coala-a-practical-and-vision-centric","slug":"coala-a-practical-and-vision-centric","title":"COALA: A Practical and Vision-Centric Federated Learning Platform","date":"2024-07-23","arxiv_id":"2407.16560","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/coala-a-practical-and-vision-centric#ran","syntology_url":"https://syntology.ai/paper/2407.16560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16560"}},"official":{"repos":["sonyresearch/coala"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/momaland-a-set-of-benchmarks-for-multi","slug":"momaland-a-set-of-benchmarks-for-multi","title":"MOMAland: A Set of Benchmarks for Multi-Objective Multi-Agent Reinforcement Learning","date":"2024-07-23","arxiv_id":"2407.16312","repositories_listed":1,"syntology":null},{"url":"/paper/haloquest-a-visual-hallucination-dataset-for","slug":"haloquest-a-visual-hallucination-dataset-for","title":"HaloQuest: A Visual Hallucination Dataset for Advancing Multimodal Reasoning","date":"2024-07-22","arxiv_id":"2407.15680","repositories_listed":1,"syntology":null},{"url":"/paper/lca-on-the-line-benchmarking-out-of","slug":"lca-on-the-line-benchmarking-out-of","title":"LCA-on-the-Line: Benchmarking Out-of-Distribution Generalization with Class Taxonomies","date":"2024-07-22","arxiv_id":"2407.16067","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lca-on-the-line-benchmarking-out-of#ran","syntology_url":"https://syntology.ai/paper/2407.16067","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16067"}},"official":{"repos":["elvishelvis/lca-on-the-line"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/open-cd-a-comprehensive-toolbox-for-change","slug":"open-cd-a-comprehensive-toolbox-for-change","title":"Open-CD: A Comprehensive Toolbox for Change Detection","date":"2024-07-22","arxiv_id":"2407.15317","repositories_listed":1,"syntology":null},{"url":"/paper/pogema-a-benchmark-platform-for-cooperative","slug":"pogema-a-benchmark-platform-for-cooperative","title":"POGEMA: A Benchmark Platform for Cooperative Multi-Agent Pathfinding","date":"2024-07-20","arxiv_id":"2407.14931","repositories_listed":1,"syntology":null},{"url":"/paper/thinking-racial-bias-in-fair-forgery","slug":"thinking-racial-bias-in-fair-forgery","title":"Thinking Racial Bias in Fair Forgery Detection: Models, Datasets and Evaluations","date":"2024-07-19","arxiv_id":"2407.14367","repositories_listed":1,"syntology":null},{"url":"/paper/any-image-restoration-with-efficient","slug":"any-image-restoration-with-efficient","title":"Restore Anything Model via Efficient Degradation Adaptation","date":"2024-07-18","arxiv_id":"2407.13372","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-biomedical-knowledge-discovery-for","slug":"enhancing-biomedical-knowledge-discovery-for","title":"Enhancing Biomedical Knowledge Discovery for Diseases: An Open-Source Framework Applied on Rett Syndrome and Alzheimer's Disease","date":"2024-07-18","arxiv_id":"2407.13492","repositories_listed":1,"syntology":null},{"url":"/paper/abstraction-alignment-comparing-model-and","slug":"abstraction-alignment-comparing-model-and","title":"Abstraction Alignment: Comparing Model-Learned and Human-Encoded Conceptual Relationships","date":"2024-07-17","arxiv_id":"2407.12543","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":16,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/abstraction-alignment-comparing-model-and#ran","syntology_url":"https://syntology.ai/paper/2407.12543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12543"}},"official":{"repos":["mitvis/abstraction-alignment"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-robust-self-supervised-learning","slug":"benchmarking-robust-self-supervised-learning","title":"Benchmarking Robust Self-Supervised Learning Across Diverse Downstream Tasks","date":"2024-07-17","arxiv_id":"2407.12588","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-robust-self-supervised-learning#ran","syntology_url":"https://syntology.ai/paper/2407.12588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12588"}},"official":{"repos":["layer6ai-labs/ssl-robustness"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lmms-eval-reality-check-on-the-evaluation-of","slug":"lmms-eval-reality-check-on-the-evaluation-of","title":"LMMs-Eval: Reality Check on the Evaluation of Large Multimodal Models","date":"2024-07-17","arxiv_id":"2407.12772","repositories_listed":1,"syntology":null},{"url":"/paper/reliable-and-efficient-concept-erasure-of","slug":"reliable-and-efficient-concept-erasure-of","title":"Reliable and Efficient Concept Erasure of Text-to-Image Diffusion Models","date":"2024-07-17","arxiv_id":"2407.12383","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reliable-and-efficient-concept-erasure-of#ran","syntology_url":"https://syntology.ai/paper/2407.12383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12383"}},"official":{"repos":["charlesgong12/rece"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-receptive-field-in-dynamic-graph","slug":"temporal-receptive-field-in-dynamic-graph","title":"Temporal receptive field in dynamic graph learning: A comprehensive analysis","date":"2024-07-17","arxiv_id":"2407.12370","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-the-attribution-quality-of","slug":"benchmarking-the-attribution-quality-of","title":"Benchmarking the Attribution Quality of Vision Models","date":"2024-07-16","arxiv_id":"2407.11910","repositories_listed":1,"syntology":{"n":14,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/benchmarking-the-attribution-quality-of#ran","syntology_url":"https://syntology.ai/paper/2407.11910","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11910"}},"official":{"repos":["visinf/idsds"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/feature-interpretability-in-bcis-exploring","slug":"feature-interpretability-in-bcis-exploring","title":"Feature interpretability in BCIs: exploring the role of network lateralization","date":"2024-07-16","arxiv_id":"2407.11617","repositories_listed":1,"syntology":null},{"url":"/paper/gv-bench-benchmarking-local-feature-matching","slug":"gv-bench-benchmarking-local-feature-matching","title":"GV-Bench: Benchmarking Local Feature Matching for Geometric Verification of Long-term Loop Closure Detection","date":"2024-07-16","arxiv_id":"2407.11736","repositories_listed":1,"syntology":null},{"url":"/paper/remm-rotation-equivariant-framework-for-end","slug":"remm-rotation-equivariant-framework-for-end","title":"REMM:Rotation-Equivariant Framework for End-to-End Multimodal Image Matching","date":"2024-07-16","arxiv_id":"2407.11637","repositories_listed":1,"syntology":null},{"url":"/paper/skada-bench-benchmarking-unsupervised-domain","slug":"skada-bench-benchmarking-unsupervised-domain","title":"SKADA-Bench: Benchmarking Unsupervised Domain Adaptation Methods with Realistic Validation On Diverse Modalities","date":"2024-07-16","arxiv_id":"2407.11676","repositories_listed":1,"syntology":null},{"url":"/paper/cibench-evaluating-your-llms-with-a-code","slug":"cibench-evaluating-your-llms-with-a-code","title":"CIBench: Evaluating Your LLMs with a Code Interpreter Plugin","date":"2024-07-15","arxiv_id":"2407.10499","repositories_listed":1,"syntology":null},{"url":"/paper/separable-operator-networks","slug":"separable-operator-networks","title":"Separable Operator Networks","date":"2024-07-15","arxiv_id":"2407.11253","repositories_listed":1,"syntology":null},{"url":"/paper/when-heterophily-meets-heterogeneity-new","slug":"when-heterophily-meets-heterogeneity-new","title":"When Heterophily Meets Heterogeneity: Challenges and a New Large-Scale Graph Benchmark","date":"2024-07-15","arxiv_id":"2407.10916","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-llms-for-optimization-modeling","slug":"benchmarking-llms-for-optimization-modeling","title":"OptiBench Meets ReSocratic: Measure and Improve LLMs for Optimization Modeling","date":"2024-07-13","arxiv_id":"2407.09887","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-llms-for-optimization-modeling#ran","syntology_url":"https://syntology.ai/paper/2407.09887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09887"}},"official":{"repos":["yangzhch6/ReSocratic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-language-model-creativity-a-case","slug":"benchmarking-language-model-creativity-a-case","title":"Benchmarking Language Model Creativity: A Case Study on Code Generation","date":"2024-07-12","arxiv_id":"2407.09007","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-language-model-creativity-a-case#ran","syntology_url":"https://syntology.ai/paper/2407.09007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09007"}},"official":{"repos":["JHU-CLSP/NeoCoder"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-attention-driven-reinforcement-learning","slug":"deep-attention-driven-reinforcement-learning","title":"Deep Attention Driven Reinforcement Learning (DAD-RL) for Autonomous Decision-Making in Dynamic Environment","date":"2024-07-12","arxiv_id":"2407.08932","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-is-not-enough-benchmarking","slug":"natural-language-is-not-enough-benchmarking","title":"Natural language is not enough: Benchmarking multi-modal generative AI for Verilog generation","date":"2024-07-11","arxiv_id":"2407.08473","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/natural-language-is-not-enough-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2407.08473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08473"}},"official":{"repos":["aichipdesign/chipgptv"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/predbench-benchmarking-spatio-temporal","slug":"predbench-benchmarking-spatio-temporal","title":"PredBench: Benchmarking Spatio-Temporal Prediction across Diverse Disciplines","date":"2024-07-11","arxiv_id":"2407.08418","repositories_listed":1,"syntology":null},{"url":"/paper/wayvescenes101-a-dataset-and-benchmark-for","slug":"wayvescenes101-a-dataset-and-benchmark-for","title":"WayveScenes101: A Dataset and Benchmark for Novel View Synthesis in Autonomous Driving","date":"2024-07-11","arxiv_id":"2407.08280","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/wayvescenes101-a-dataset-and-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2407.08280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08280"}},"official":{"repos":["wayveai/wayve_scenes"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-embedding-aggregation-methods-in","slug":"benchmarking-embedding-aggregation-methods-in","title":"Benchmarking Embedding Aggregation Methods in Computational Pathology: A Clinical Data Perspective","date":"2024-07-10","arxiv_id":"2407.07841","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-embedding-aggregation-methods-in#ran","syntology_url":"https://syntology.ai/paper/2407.07841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07841"}},"official":{"repos":["fuchs-lab-public/cpath_sabenchmark"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/instructlayout-instruction-driven-2d-and-3d","slug":"instructlayout-instruction-driven-2d-and-3d","title":"InstructLayout: Instruction-Driven 2D and 3D Layout Synthesis with Semantic Graph Prior","date":"2024-07-10","arxiv_id":"2407.07580","repositories_listed":1,"syntology":null},{"url":"/paper/training-on-the-test-task-confounds","slug":"training-on-the-test-task-confounds","title":"Training on the Test Task Confounds Evaluation and Emergence","date":"2024-07-10","arxiv_id":"2407.07890","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/training-on-the-test-task-confounds#ran","syntology_url":"https://syntology.ai/paper/2407.07890","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07890"}},"official":{"repos":["socialfoundations/training-on-the-test-task"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hermes-holographic-equivariant-neural-network","slug":"hermes-holographic-equivariant-neural-network","title":"HERMES: Holographic Equivariant neuRal network model for Mutational Effect and Stability prediction","date":"2024-07-09","arxiv_id":"2407.06703","repositories_listed":1,"syntology":null},{"url":"/paper/humanrefiner-benchmarking-abnormal-human","slug":"humanrefiner-benchmarking-abnormal-human","title":"HumanRefiner: Benchmarking Abnormal Human Generation and Refining with Coarse-to-fine Pose-Reversible Guidance","date":"2024-07-09","arxiv_id":"2407.06937","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-benchmarking-and-understanding-1","slug":"revisiting-benchmarking-and-understanding-1","title":"Revisiting, Benchmarking and Understanding Unsupervised Graph Domain Adaptation","date":"2024-07-09","arxiv_id":"2407.11052","repositories_listed":1,"syntology":{"n":47,"n_ran":38,"n_constructed":0,"n_ran_checked":32,"n_instrument":6,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":31,"n_pointer_only":21,"phrase":"38 ran (of which 0 constructed an object rather than computing a result; 32 with no instrument failure: 1 honoured, 0 violated, 31 with no contract checked; 6 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/revisiting-benchmarking-and-understanding-1#ran","syntology_url":"https://syntology.ai/paper/2407.11052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11052"}},"official":{"repos":["pygda-team/pygda"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/codeupdatearena-benchmarking-knowledge","slug":"codeupdatearena-benchmarking-knowledge","title":"CodeUpdateArena: Benchmarking Knowledge Editing on API Updates","date":"2024-07-08","arxiv_id":"2407.06249","repositories_listed":1,"syntology":null},{"url":"/paper/opencil-benchmarking-out-of-distribution","slug":"opencil-benchmarking-out-of-distribution","title":"OpenCIL: Benchmarking Out-of-Distribution Detection in Class-Incremental Learning","date":"2024-07-08","arxiv_id":"2407.06045","repositories_listed":1,"syntology":null}],"record_sha256":"47d295aa804edbd4db0ef430d5e664dc11bc10ee67c21d9d6f693245e9201bf4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}