{"url":"/sota/robot-manipulation-on-simpler-env","task":{"name":"Robot Manipulation","url":"/task/robot-manipulation","note":null},"dataset":{"name":"SimplerEnv-Google Robot","url":"/dataset/simpler-env"},"category":"Robots","categories":["Robots"],"category_note":null,"description":null,"description_from":null,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["Visual Matching","Visual Matching-Pick Coke Can","Visual Matching-Move Near","Visual Matching-Open/Close Drawer","Variant Aggregation","Variant Aggregation-Pick Coke Can","Variant Aggregation-Move Near","Variant Aggregation-Open/Close Drawer"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"Visual Matching":null,"Visual Matching-Pick Coke Can":null,"Visual Matching-Move Near":null,"Visual Matching-Open/Close Drawer":null,"Variant Aggregation":null,"Variant Aggregation-Pick Coke Can":null,"Variant Aggregation-Move Near":null,"Variant Aggregation-Open/Close Drawer":null}},"counts":{"rows":9,"rows_with_code":6,"rows_with_paper_page":9,"rows_dated":9,"rows_using_additional_data":8},"rows":[{"rank_in_archive_order":1,"model":"SoFar","metrics":{"Variant Aggregation":"0.676","Variant Aggregation-Move Near":"0.740","Variant Aggregation-Open/Close Drawer":"0.297","Variant Aggregation-Pick Coke Can":"0.907","Visual Matching":"0.749","Visual Matching-Move Near":"0.917","Visual Matching-Open/Close Drawer":"0.403","Visual Matching-Pick Coke Can":"0.923"},"uses_additional_data":false,"paper_date":"2025-02-18","paper":"/paper/sofar-language-grounded-orientation-bridges","paper_url":"https://arxiv.org/abs/2502.13143v1","paper_title":"SoFar: Language-Grounded Orientation Bridges Spatial Reasoning and Object Manipulation","code":"https://github.com/qizekun/SoFar","n_code_links":2,"syntology":null},{"rank_in_archive_order":2,"model":"SpatialVLA","metrics":{"Variant Aggregation":"0.688","Variant Aggregation-Move Near":"0.717","Variant Aggregation-Open/Close Drawer":"0.362","Variant Aggregation-Pick Coke Can":"0.895","Visual Matching":"0.719","Visual Matching-Move Near":"0.696","Visual Matching-Open/Close Drawer":"0.593","Visual Matching-Pick Coke Can":"0.810"},"uses_additional_data":true,"paper_date":"2025-01-27","paper":"/paper/spatialvla-exploring-spatial-representations","paper_url":"https://arxiv.org/abs/2501.15830v5","paper_title":"SpatialVLA: Exploring Spatial Representations for Visual-Language-Action Model","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":3,"model":"Dita-300M","metrics":{"Variant Aggregation":"0.652","Variant Aggregation-Move Near":"0.730","Variant Aggregation-Open/Close Drawer":"0.370","Variant Aggregation-Pick Coke Can":"0.855","Visual Matching":"0.687","Visual Matching-Move Near":"0.760","Visual Matching-Open/Close Drawer":"0.463","Visual Matching-Pick Coke Can":"0.837"},"uses_additional_data":true,"paper_date":"2025-03-25","paper":"/paper/dita-scaling-diffusion-transformer-for","paper_url":"https://arxiv.org/abs/2503.19757v1","paper_title":"Dita: Scaling Diffusion Transformer for Generalist Vision-Language-Action Policy","code":"https://github.com/RoboDita/Dita","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":1,"n_samples":3,"n_pointer_only_licence":0}},{"rank_in_archive_order":4,"model":"RT-2-X","metrics":{"Variant Aggregation":"0.661","Variant Aggregation-Move Near":"0.792","Variant Aggregation-Open/Close Drawer":"0.353","Variant Aggregation-Pick Coke Can":"0.823","Visual Matching":"0.606","Visual Matching-Move Near":"0.779","Visual Matching-Open/Close Drawer":"0.250","Visual Matching-Pick Coke Can":"0.787"},"uses_additional_data":true,"paper_date":"2023-07-28","paper":"/paper/rt-2-vision-language-action-models-transfer","paper_url":"https://arxiv.org/abs/2307.15818v1","paper_title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","code":"https://github.com/kyegomez/RT-2","n_code_links":1,"syntology":null},{"rank_in_archive_order":5,"model":"RoboVLM","metrics":{"Variant Aggregation":"0.463","Variant Aggregation-Move Near":"0.560","Variant Aggregation-Open/Close Drawer":"0.085","Variant Aggregation-Pick Coke Can":"0.683","Visual Matching":"0.563","Visual Matching-Move Near":"0.663","Visual Matching-Open/Close Drawer":"0.268","Visual Matching-Pick Coke Can":"0.727"},"uses_additional_data":true,"paper_date":"2024-12-18","paper":"/paper/towards-generalist-robot-policies-what","paper_url":"https://arxiv.org/abs/2412.14058v3","paper_title":"Towards Generalist Robot Policies: What Matters in Building Vision-Language-Action Models","code":"https://github.com/Robot-VLAs/RoboVLMs","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":7,"n_samples":8,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"RT-1-X","metrics":{"Variant Aggregation":"0.397","Variant Aggregation-Move Near":"0.323","Variant Aggregation-Open/Close Drawer":"0.294","Variant Aggregation-Pick Coke Can":"0.490","Visual Matching":"0.534","Visual Matching-Move Near":"0.317","Visual Matching-Open/Close Drawer":"0.597","Visual Matching-Pick Coke Can":"0.567"},"uses_additional_data":true,"paper_date":"2022-12-13","paper":"/paper/rt-1-robotics-transformer-for-real-world","paper_url":"https://arxiv.org/abs/2212.06817v2","paper_title":"RT-1: Robotics Transformer for Real-World Control at Scale","code":"https://github.com/google-research/robotics_transformer","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":4,"n_samples":4,"n_pointer_only_licence":0}},{"rank_in_archive_order":7,"model":"TraceVLA","metrics":{"Variant Aggregation":"0.450","Variant Aggregation-Move Near":"0.564","Variant Aggregation-Open/Close Drawer":"0.310","Variant Aggregation-Pick Coke Can":"0.600","Visual Matching":"0.460","Visual Matching-Move Near":"0.600","Visual Matching-Open/Close Drawer":"0.240","Visual Matching-Pick Coke Can":"0.560"},"uses_additional_data":true,"paper_date":"2024-12-13","paper":"/paper/tracevla-visual-trace-prompting-enhances","paper_url":"https://arxiv.org/abs/2412.10345v3","paper_title":"TraceVLA: Visual Trace Prompting Enhances Spatial-Temporal Awareness for Generalist Robotic Policies","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":8,"model":"OpenVLA","metrics":{"Variant Aggregation":"0.411","Variant Aggregation-Move Near":"0.477","Variant Aggregation-Open/Close Drawer":"0.177","Variant Aggregation-Pick Coke Can":"0.545","Visual Matching":"0.277","Visual Matching-Move Near":"0.462","Visual Matching-Open/Close Drawer":"0.356","Visual Matching-Pick Coke Can":"0.163"},"uses_additional_data":true,"paper_date":"2024-06-13","paper":"/paper/openvla-an-open-source-vision-language-action","paper_url":"https://arxiv.org/abs/2406.09246v3","paper_title":"OpenVLA: An Open-Source Vision-Language-Action Model","code":"https://github.com/openvla/openvla","n_code_links":3,"syntology":{"n_ran":2,"n_unverified":8,"n_samples":10,"n_pointer_only_licence":0}},{"rank_in_archive_order":9,"model":"Octo-Base","metrics":{"Variant Aggregation":"0.012","Variant Aggregation-Move Near":"0.031","Variant Aggregation-Open/Close Drawer":"0.011","Variant Aggregation-Pick Coke Can":"0.006","Visual Matching":"0.168","Visual Matching-Move Near":"0.042","Visual Matching-Open/Close Drawer":"0.227","Visual Matching-Pick Coke Can":"0.170"},"uses_additional_data":true,"paper_date":"2024-05-20","paper":"/paper/octo-an-open-source-generalist-robot-policy","paper_url":"https://arxiv.org/abs/2405.12213v2","paper_title":"Octo: An Open-Source Generalist Robot Policy","code":null,"n_code_links":0,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":4,"rows_with_any_sample_ran":3,"distinct_papers_with_graph_line":4,"distinct_papers_with_any_sample_ran":3,"samples_over_distinct_papers":{"n_ran":5,"n_unverified":20,"n_samples":25,"n_pointer_only_licence":0,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":5,"n_unverified":20,"n_samples":25,"n_pointer_only_licence":0,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}