{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/annotate","entry":"annotate","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":7,"n_papers_ran":2,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":9,"n_samples_ran":2,"n_samples_fingerprinted":0,"n_places":9,"n_places_pointer_only":4,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":2,"ran_fixture":0,"ran":0,"unverified":7},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2403.14312","paper":"/paper/chainlm-empowering-large-language-models-with","title":"ChainLM: Empowering Large Language Models with Improved Chain-of-Thought Prompting","date":"2024-03-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucaibox/chainlm","path":"generate/filter_claude.py","file_url":"https://github.com/rucaibox/chainlm/blob/HEAD/generate/filter_claude.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2cf2868f64341a27","mcp_get_code":{"code_sha256":"2cf2868f64341a27"}},{"arxiv_id":"2310.04678","paper":"/paper/doris-mae-scientific-document-retrieval-using","title":"DORIS-MAE: Scientific Document Retrieval using Multi-level Aspect-based Queries","date":"2023-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"real-doris-mae/doris-mae-dataset","path":"gpt_annotation/GPT_annotation.py","file_url":"https://github.com/real-doris-mae/doris-mae-dataset/blob/HEAD/gpt_annotation/GPT_annotation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"edecf512aaf44e23","mcp_get_code":{"code_sha256":"edecf512aaf44e23"}},{"arxiv_id":"2310.01377","paper":"/paper/ultrafeedback-boosting-language-models-with","title":"UltraFeedback: Boosting Language Models with Scaled AI Feedback","date":"2023-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/ultrafeedback","path":"src/data_annotation/annotate_critique.py","file_url":"https://github.com/thunlp/ultrafeedback/blob/HEAD/src/data_annotation/annotate_critique.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2b3cb1aee424f3c8","mcp_get_code":{"code_sha256":"2b3cb1aee424f3c8"}},{"arxiv_id":"2305.13112","paper":"/paper/rethinking-the-evaluation-for-conversational","title":"Rethinking the Evaluation for Conversational Recommendation in the Era of Large Language Models","date":"2023-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RUCAIBox/iEvaLM-CRS","path":"src/model/CHATGPT.py","file_url":"https://github.com/RUCAIBox/iEvaLM-CRS/blob/HEAD/src/model/CHATGPT.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1052ac20f3dfe2dc","mcp_get_code":{"code_sha256":"1052ac20f3dfe2dc"}},{"arxiv_id":"2305.13112","paper":"/paper/rethinking-the-evaluation-for-conversational","title":"Rethinking the Evaluation for Conversational Recommendation in the Era of Large Language Models","date":"2023-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucaibox/ievalm-crs","path":"script/cache_item.py","file_url":"https://github.com/rucaibox/ievalm-crs/blob/HEAD/script/cache_item.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4b4ed6d15480f5a5","mcp_get_code":{"code_sha256":"4b4ed6d15480f5a5"}},{"arxiv_id":"2305.13112","paper":"/paper/rethinking-the-evaluation-for-conversational","title":"Rethinking the Evaluation for Conversational Recommendation in the Era of Large Language Models","date":"2023-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucaibox/ievalm-crs","path":"src/model/CHATGPT.py","file_url":"https://github.com/rucaibox/ievalm-crs/blob/HEAD/src/model/CHATGPT.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"285d00ffcd75ae67","mcp_get_code":{"code_sha256":"285d00ffcd75ae67"}},{"arxiv_id":"2303.18223","paper":"/paper/a-survey-of-large-language-models","title":"A Survey of Large Language Models","date":"2023-03-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucaibox/llmsurvey","path":"Experiments/HumanAlignment/HaluEval/claude_halu.py","file_url":"https://github.com/rucaibox/llmsurvey/blob/HEAD/Experiments/HumanAlignment/HaluEval/claude_halu.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6794c98718b27e2e","mcp_get_code":{"code_sha256":"6794c98718b27e2e"}},{"arxiv_id":"2106.06981","paper":"/paper/thinking-like-transformers-1","title":"Thinking Like Transformers","date":"2021-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deepmind/tracr","path":"tracr/rasp/rasp.py","file_url":"https://github.com/deepmind/tracr/blob/HEAD/tracr/rasp/rasp.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"dacd53883797b544","mcp_get_code":{"code_sha256":"dacd53883797b544"}},{"arxiv_id":"1812.00899","paper":"/paper/toward-scalable-neural-dialogue-state","title":"Toward Scalable Neural Dialogue State Tracking Model","date":"2018-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"elnaaz/GCE-Model","path":"dataset.py","file_url":"https://github.com/elnaaz/GCE-Model/blob/HEAD/dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"a8db10c9004d9b8f","mcp_get_code":{"code_sha256":"a8db10c9004d9b8f"}}]}