{"url":"/sota/task-1-grouping-on-ocw","task":{"name":"Only Connect Walls Dataset Task 1 (Grouping)","url":"/task/task-1-grouping","note":null},"dataset":{"name":"OCW","url":"/dataset/only-connect-wall-ocw-dataset"},"category":"Natural Language Processing","categories":["Methodology","Natural Language Processing"],"category_note":null,"description":"Split data into groups, taking into account knowledge in the form of constraints on points, groups of points, or clusters.","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":[" Wasserstein Distance (WD)","# Correct Groups","Fowlkes Mallows Score (FMS)","Adjusted Rand Index (ARI)","Adjusted Mutual Information (AMI)","# Solved Walls","Wasserstein Distance (WD)"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{" Wasserstein Distance (WD)":"lower","# Correct Groups":null,"Fowlkes Mallows Score (FMS)":"higher","Adjusted Rand Index (ARI)":null,"Adjusted Mutual Information (AMI)":null,"# Solved Walls":null,"Wasserstein Distance (WD)":"lower"}},"counts":{"rows":22,"rows_with_code":22,"rows_with_paper_page":22,"rows_dated":22,"rows_using_additional_data":22},"rows":[{"rank_in_archive_order":1,"model":"GPT-4 (5-shot)","metrics":{" Wasserstein Distance (WD)":"72.9","# Correct Groups":"269","# Solved Walls":"7","Adjusted Mutual Information (AMI)":"32.8 ","Adjusted Rand Index (ARI)":"29.1","Fowlkes Mallows Score (FMS)":"43.4"},"uses_additional_data":true,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":2,"n_unverified":3,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":2,"model":"GPT-4 (1-shot)","metrics":{" Wasserstein Distance (WD)":"73.4","# Correct Groups":"262","# Solved Walls":" 4","Adjusted Mutual Information (AMI)":" 33.5","Adjusted Rand Index (ARI)":"29.7","Fowlkes Mallows Score (FMS)":"43.7"},"uses_additional_data":true,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":2,"n_unverified":3,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":3,"model":"GPT-4 (100-shot)","metrics":{" Wasserstein Distance (WD)":"73.6","# Correct Groups":"249","# Solved Walls":"3","Adjusted Mutual Information (AMI)":" 32.3","Adjusted Rand Index (ARI)":" 28.5","Fowlkes Mallows Score (FMS)":" 42.8"},"uses_additional_data":true,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":2,"n_unverified":3,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":4,"model":"GPT-4 (3-shot)","metrics":{" Wasserstein Distance (WD)":"73.7","# Correct Groups":"272","# Solved Walls":"5","Adjusted Mutual Information (AMI)":"33.6","Adjusted Rand Index (ARI)":"29.9","Fowlkes Mallows Score (FMS)":"43.9"},"uses_additional_data":true,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":2,"n_unverified":3,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":5,"model":"GPT-4 (0-shot)","metrics":{" Wasserstein Distance (WD)":"75.8","# Correct Groups":"239","# Solved Walls":"6","Adjusted Mutual Information (AMI)":"30.7","Adjusted Rand Index (ARI)":"27.2","Fowlkes Mallows Score (FMS)":"41.5"},"uses_additional_data":true,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":2,"n_unverified":3,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":6,"model":"GPT-3.5-turbo (5-shot)","metrics":{" Wasserstein Distance (WD)":"80.6","# Correct Groups":"149","# Solved Walls":"2","Adjusted Mutual Information (AMI)":" 25.4","Adjusted Rand Index (ARI)":" 22.0","Fowlkes Mallows Score (FMS)":" 37.3"},"uses_additional_data":true,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":2,"n_unverified":3,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":7,"model":"GPT-3.5-turbo (3-shot)","metrics":{" Wasserstein Distance (WD)":"80.9","# Correct Groups":"140","# Solved Walls":"0","Adjusted Mutual Information (AMI)":"24.7","Adjusted Rand Index (ARI)":"21.3","Fowlkes Mallows Score (FMS)":"36.8"},"uses_additional_data":true,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":2,"n_unverified":3,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":8,"model":"GPT-3.5-turbo (10-shot)","metrics":{" Wasserstein Distance (WD)":"81.2","# Correct Groups":"137","# Solved Walls":"2","Adjusted Mutual Information (AMI)":" 24.0","Adjusted Rand Index (ARI)":"20.4","Fowlkes Mallows Score (FMS)":"36.1"},"uses_additional_data":true,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":2,"n_unverified":3,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":9,"model":"GPT-3.5-turbo (1-shot)","metrics":{" Wasserstein Distance (WD)":"82.3","# Correct Groups":"123","# Solved Walls":"0","Adjusted Mutual Information (AMI)":" 21.2","Adjusted Rand Index (ARI)":" 18.2","Fowlkes Mallows Score (FMS)":" 34.4"},"uses_additional_data":true,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":2,"n_unverified":3,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":10,"model":"GPT-3.5-turbo (0-shot)","metrics":{" Wasserstein Distance (WD)":"82.5","# Correct Groups":"114","# Solved Walls":"0","Adjusted Mutual Information (AMI)":"21.6","Adjusted Rand Index (ARI)":" 18.4","Fowlkes Mallows Score (FMS)":" 34.0"},"uses_additional_data":true,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":2,"n_unverified":3,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":11,"model":"E5 (BASE)","metrics":{" Wasserstein Distance (WD)":"83.8 ± .6","# Correct Groups":"89 ± 6","# Solved Walls":" 1 ± 0","Adjusted Mutual Information (AMI)":"19.5 ± .4","Adjusted Rand Index (ARI)":" 16.3 ± .4","Fowlkes Mallows Score (FMS)":"33.1 ± .3"},"uses_additional_data":true,"paper_date":"2022-12-07","paper":"/paper/text-embeddings-by-weakly-supervised","paper_url":"https://arxiv.org/abs/2212.03533v2","paper_title":"Text Embeddings by Weakly-Supervised Contrastive Pre-training","code":"https://github.com/microsoft/unilm","n_code_links":1,"syntology":null},{"rank_in_archive_order":12,"model":"FastText (Crawl)","metrics":{" Wasserstein Distance (WD)":"84.2 ± .5","# Correct Groups":"80 ± 4","# Solved Walls":"0 ± 0","Adjusted Mutual Information (AMI)":"18.4 ± .4","Adjusted Rand Index (ARI)":"15.2 ± .3","Fowlkes Mallows Score (FMS)":"32.1 ± .3"},"uses_additional_data":true,"paper_date":"2018-02-19","paper":"/paper/learning-word-vectors-for-157-languages","paper_url":"http://arxiv.org/abs/1802.06893v2","paper_title":"Learning Word Vectors for 157 Languages","code":"https://github.com/dzieciou/lemmatizer-pl","n_code_links":2,"syntology":null},{"rank_in_archive_order":13,"model":"E5 (LARGE)","metrics":{" Wasserstein Distance (WD)":"84.4 ± .7","# Correct Groups":"76 ± 5","# Solved Walls":"0 ± 0","Adjusted Mutual Information (AMI)":"18.5 ± .6","Adjusted Rand Index (ARI)":"15.4 ± .5","Fowlkes Mallows Score (FMS)":"32.3 ± .4"},"uses_additional_data":true,"paper_date":"2022-12-07","paper":"/paper/text-embeddings-by-weakly-supervised","paper_url":"https://arxiv.org/abs/2212.03533v2","paper_title":"Text Embeddings by Weakly-Supervised Contrastive Pre-training","code":"https://github.com/microsoft/unilm","n_code_links":1,"syntology":null},{"rank_in_archive_order":14,"model":"GloVe","metrics":{" Wasserstein Distance (WD)":"84.9 ± .4","# Correct Groups":"68 ± 4","# Solved Walls":"0 ± 0","Adjusted Mutual Information (AMI)":"17.6 ± .4","Adjusted Rand Index (ARI)":"14.4 ± .3","Fowlkes Mallows Score (FMS)":" 31.5 ± .3"},"uses_additional_data":true,"paper_date":"2014-10-01","paper":"/paper/glove-global-vectors-for-word-representation","paper_url":"https://aclanthology.org/D14-1162","paper_title":"GloVe: Global Vectors for Word Representation","code":"https://github.com/stanfordnlp/GloVe","n_code_links":4,"syntology":null},{"rank_in_archive_order":15,"model":"FastText (News)","metrics":{" Wasserstein Distance (WD)":"85.5 ± .5","# Correct Groups":"62 ± 3","# Solved Walls":"0 ± 0","Adjusted Mutual Information (AMI)":" 15.8 ± .3","Adjusted Rand Index (ARI)":"13.0 ± .2","Fowlkes Mallows Score (FMS)":"30.4 ± .2"},"uses_additional_data":true,"paper_date":"2018-02-19","paper":"/paper/learning-word-vectors-for-157-languages","paper_url":"http://arxiv.org/abs/1802.06893v2","paper_title":"Learning Word Vectors for 157 Languages","code":"https://github.com/dzieciou/lemmatizer-pl","n_code_links":2,"syntology":null},{"rank_in_archive_order":16,"model":"all-mpnet (BASE)","metrics":{" Wasserstein Distance (WD)":"86.3 ± .4","# Correct Groups":" 50 ± 4","# Solved Walls":"0 ± 0","Adjusted Mutual Information (AMI)":"14.3 ± .5","Adjusted Rand Index (ARI)":" 11.7 ± .4","Fowlkes Mallows Score (FMS)":"29.4 ± .3"},"uses_additional_data":true,"paper_date":"2020-04-20","paper":"/paper/mpnet-masked-and-permuted-pre-training-for","paper_url":"https://arxiv.org/abs/2004.09297v2","paper_title":"MPNet: Masked and Permuted Pre-training for Language Understanding","code":"https://github.com/huggingface/transformers","n_code_links":7,"syntology":{"n_ran":5,"n_unverified":3,"n_samples":8,"n_pointer_only_licence":6}},{"rank_in_archive_order":17,"model":"BERT (LARGE)","metrics":{" Wasserstein Distance (WD)":"88.3 ± .5","# Correct Groups":"33 ± 2","# Solved Walls":"0 ± 0","Adjusted Mutual Information (AMI)":"10.3 ± .3","Adjusted Rand Index (ARI)":"8.2 ± .3","Fowlkes Mallows Score (FMS)":" 26.5 ± .2"},"uses_additional_data":true,"paper_date":"2019-11-25","paper":"/paper/pre-training-of-deep-bidirectional-protein","paper_url":"https://arxiv.org/abs/1912.05625v4","paper_title":"Pre-Training of Deep Bidirectional Protein Sequence Representations with Structural Information","code":"https://github.com/mswzeus/PLUS","n_code_links":1,"syntology":null},{"rank_in_archive_order":18,"model":"BERT (BASE)","metrics":{" Wasserstein Distance (WD)":"89.5 ± .4","# Correct Groups":"22 ± 2","# Solved Walls":"0 ± 0","Adjusted Mutual Information (AMI)":"8.1 ± .4","Adjusted Rand Index (ARI)":"6.4 ± .3","Fowlkes Mallows Score (FMS)":"25.1 ± .2"},"uses_additional_data":true,"paper_date":"2019-11-25","paper":"/paper/pre-training-of-deep-bidirectional-protein","paper_url":"https://arxiv.org/abs/1912.05625v4","paper_title":"Pre-Training of Deep Bidirectional Protein Sequence Representations with Structural Information","code":"https://github.com/mswzeus/PLUS","n_code_links":1,"syntology":null},{"rank_in_archive_order":19,"model":"Human Performance","metrics":{"# Correct Groups":"1405","# Solved Walls":"285"},"uses_additional_data":true,"paper_date":"2023-06-19","paper":"/paper/large-language-models-are-fixated-by-red-1","paper_url":"https://arxiv.org/abs/2306.11167v4","paper_title":"Large Language Models are Fixated by Red Herrings: Exploring Creative Problem Solving and Einstellung Effect using the Only Connect Wall Dataset","code":"https://github.com/taatiteam/ocw","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":1,"n_samples":1,"n_pointer_only_licence":0}},{"rank_in_archive_order":20,"model":"ELMo (LARGE)","metrics":{"# Correct Groups":"55 ± 4","# Solved Walls":"0 ± 0","Adjusted Mutual Information (AMI)":"14.5 ± .4","Adjusted Rand Index (ARI)":"11.8 ± .4","Fowlkes Mallows Score (FMS)":"29.5 ± .3","Wasserstein Distance (WD)":"86.3 ± .6"},"uses_additional_data":true,"paper_date":"2018-02-15","paper":"/paper/deep-contextualized-word-representations","paper_url":"http://arxiv.org/abs/1802.05365v2","paper_title":"Deep contextualized word representations","code":"https://github.com/flairNLP/flair","n_code_links":46,"syntology":{"n_ran":23,"n_unverified":35,"n_samples":58,"n_pointer_only_licence":25}},{"rank_in_archive_order":21,"model":"DistilBERT (BASE)","metrics":{"# Correct Groups":"49 ± 4","# Solved Walls":"0 ± 0","Adjusted Mutual Information (AMI)":" 14.0 ± .3","Adjusted Rand Index (ARI)":"11.3 ± .3","Fowlkes Mallows Score (FMS)":"29.1 ± .2","Wasserstein Distance (WD)":"86.7 ± .6"},"uses_additional_data":true,"paper_date":"2019-10-02","paper":"/paper/distilbert-a-distilled-version-of-bert","paper_url":"https://arxiv.org/abs/1910.01108v4","paper_title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","code":"https://github.com/huggingface/transformers","n_code_links":37,"syntology":{"n_ran":19,"n_unverified":8,"n_samples":27,"n_pointer_only_licence":0}},{"rank_in_archive_order":22,"model":"RoBERTa (LARGE)","metrics":{"# Correct Groups":"29 ± 3","# Solved Walls":"0 ± 0","Adjusted Mutual Information (AMI)":"9.4 ± .4","Adjusted Rand Index (ARI)":" 8.4 ± .3","Fowlkes Mallows Score (FMS)":" 26.7 ± .2","Wasserstein Distance (WD)":"88.4 ± .4"},"uses_additional_data":true,"paper_date":"2019-07-26","paper":"/paper/roberta-a-robustly-optimized-bert-pretraining","paper_url":"https://arxiv.org/abs/1907.11692v1","paper_title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","code":"https://github.com/huggingface/transformers","n_code_links":67,"syntology":{"n_ran":22,"n_unverified":26,"n_samples":48,"n_pointer_only_licence":23}}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":15,"rows_with_any_sample_ran":14,"distinct_papers_with_graph_line":6,"distinct_papers_with_any_sample_ran":5,"samples_over_distinct_papers":{"n_ran":71,"n_unverified":76,"n_samples":147,"n_pointer_only_licence":55,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":89,"n_unverified":103,"n_samples":192,"n_pointer_only_licence":64,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}