{"url":"/dataset/openapi-code-completion","name":"OpenAPI completion refined","full_name":null,"description_markdown":"A human-refined dataset of OpenAPI definitions based on the APIs.guru OpenAPI [directory](https://github.com/APIs-guru/openapi-directory).\r\n\r\nThe dataset was collected from the APIs.guru OpenAPI definitions [directory](https://github.com/APIs-guru/openapi-directory).\r\nThe directory contains more than 4,000 definitions in yaml format. Analysis of the repository revealed that about 75%\r\nof the definitions in the directory are produced by a handful of major companies like Amazon, Google, and Microsoft.\r\nTo avoid the dataset bias towards a specific producer, the maximum number of definitions from a single producer was limited\r\nto 20. Multiple versions of the same API were also excluded from the dataset as they are likely to contain very similar \r\ndefinitions.","description_withheld":null,"homepage":"https://huggingface.co/datasets/BohdanPetryshyn/openapi-completion-refined","introduced_date":"2024-05-24","introduced_date_note":null,"introduced_by":{"paper":"/paper/optimizing-large-language-models-for-openapi","title":"Optimizing Large Language Models for OpenAPI Code Completion","first_author":"Bohdan Petryshyn","url":null},"license":{"name":"MIT","url":"https://opensource.org/license/mit"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Code Generation","url":"/task/code-generation","datasets_with_task":"/datasets/task/code-generation"},{"name":"Code Completion","url":"/task/code-completion","datasets_with_task":"/datasets/task/code-completion"},{"name":"OpenAPI code completion","url":"/task/openapi-code-completion","datasets_with_task":"/datasets/task/openapi-code-completion"}],"languages":[],"variants":["OpenAPI completion refined"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/openapi-code-completion-on-openapi-code","task":"OpenAPI code completion","dataset_variant":"OpenAPI completion refined","rows":4,"metrics":["Correctness, avg., %","Correctness, max., %","Validness, avg., %","Validness, max., %"],"first_row_in_archive_order":{"model":"Code Llama 7B, fine-tuned with document splitting","paper":"/paper/optimizing-large-language-models-for-openapi","metrics":{"Correctness, avg., %":"34","Correctness, max., %":"42","Validness, avg., %":"69.1","Validness, max., %":"76"},"code_links":[{"title":"BohdanPetryshyn/code-llama-fim-fine-tuning","url":"https://github.com/BohdanPetryshyn/code-llama-fim-fine-tuning"},{"title":"BohdanPetryshyn/openapi-completion-benchmark","url":"https://github.com/BohdanPetryshyn/openapi-completion-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/optimizing-large-language-models-for-openapi","title":"Optimizing Large Language Models for OpenAPI Code Completion","date":"2024-05-24","rows_on_this_dataset":4,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}