{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/get-instruction","entry":"get_instruction","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":10,"n_papers_ran":4,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":11,"n_samples_ran":4,"n_samples_fingerprinted":1,"n_places":11,"n_places_pointer_only":5,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":4,"unverified":7},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.03047","paper":"/paper/arxiv-2609-03047","title":"SHELF: A Synthetic Harness for Multi-Task Bibliographic Benchmarking","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"mjbommar/shelf-benchmark","path":"src/shelf/evaluate/instructions.py","file_url":"https://github.com/mjbommar/shelf-benchmark/blob/HEAD/src/shelf/evaluate/instructions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"8e7ef1fe35d19eb3","mcp_get_code":{"code_sha256":"8e7ef1fe35d19eb3"}},{"arxiv_id":"2504.02953","paper":"/paper/cultural-learning-based-culture-adaptation-of","title":"Cultural Learning-Based Culture Adaptation of Language Models","date":"2025-04-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ukplab/arxiv2025-clca","path":"CLCA/llm_roleplay/common/generate_scenarios.py","file_url":"https://github.com/ukplab/arxiv2025-clca/blob/HEAD/CLCA/llm_roleplay/common/generate_scenarios.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5354bbb24cabe7bb","mcp_get_code":{"code_sha256":"5354bbb24cabe7bb"}},{"arxiv_id":"2406.11370","paper":"/paper/fairer-preferences-elicit-improved-human","title":"Fairer Preferences Elicit Improved Human-Aligned Large Language Model Judgments","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cambridgeltl/zepo","path":"zepo.py","file_url":"https://github.com/cambridgeltl/zepo/blob/HEAD/zepo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"902f7af3019415e0","mcp_get_code":{"code_sha256":"902f7af3019415e0"}},{"arxiv_id":"2405.14394","paper":"/paper/instruction-tuning-with-loss-over","title":"Instruction Tuning With Loss Over Instructions","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZhengxiangShi/InstructionModelling","path":"src/generate.py","file_url":"https://github.com/ZhengxiangShi/InstructionModelling/blob/HEAD/src/generate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"88c99652268092f3","mcp_get_code":{"code_sha256":"88c99652268092f3"}},{"arxiv_id":"2312.15661","paper":"/paper/unlocking-the-potential-of-large-language","title":"Unlocking the Potential of Large Language Models for Explainable Recommendations","date":"2023-12-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"godfire66666/llm_rec_explanation","path":"src/gen_compare_result/gen_compare_test_all.py","file_url":"https://github.com/godfire66666/llm_rec_explanation/blob/HEAD/src/gen_compare_result/gen_compare_test_all.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"af70d8fb79d3c260","mcp_get_code":{"code_sha256":"af70d8fb79d3c260"}},{"arxiv_id":"2308.14391","paper":"/paper/fire-food-image-to-recipe-generation","title":"FIRE: Food Image to REcipe generation","date":"2023-08-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"prateekchhikara/fire","path":"ingredients/build_vocab.py","file_url":"https://github.com/prateekchhikara/fire/blob/HEAD/ingredients/build_vocab.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5df75078e59cecb3","mcp_get_code":{"code_sha256":"5df75078e59cecb3"}},{"arxiv_id":"2305.13112","paper":"/paper/rethinking-the-evaluation-for-conversational","title":"Rethinking the Evaluation for Conversational Recommendation in the Era of Large Language Models","date":"2023-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucaibox/ievalm-crs","path":"script/ask.py","file_url":"https://github.com/rucaibox/ievalm-crs/blob/HEAD/script/ask.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c0900257616e5ccb","mcp_get_code":{"code_sha256":"c0900257616e5ccb"}},{"arxiv_id":"2305.13112","paper":"/paper/rethinking-the-evaluation-for-conversational","title":"Rethinking the Evaluation for Conversational Recommendation in the Era of Large Language Models","date":"2023-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucaibox/ievalm-crs","path":"script/chat.py","file_url":"https://github.com/rucaibox/ievalm-crs/blob/HEAD/script/chat.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8cfed84e6fae7269","mcp_get_code":{"code_sha256":"8cfed84e6fae7269"}},{"arxiv_id":"2111.09525","paper":"/paper/summac-re-visiting-nli-based-models-for","title":"SummaC: Re-Visiting NLI-based Models for Inconsistency Detection in Summarization","date":"2021-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforce/auditnlg","path":"src/auditnlg/factualness/utils.py","file_url":"https://github.com/salesforce/auditnlg/blob/HEAD/src/auditnlg/factualness/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"e8bd306b2dea16ca","mcp_get_code":{"code_sha256":"e8bd306b2dea16ca"}},{"arxiv_id":"2006.07185","paper":"/paper/decstr-learning-goal-directed-abstract","title":"Grounding Language to Autonomously-Acquired Skills via Goal Generation","date":"2020-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"akakzia/decstr","path":"rl_modules/language_models.py","file_url":"https://github.com/akakzia/decstr/blob/HEAD/rl_modules/language_models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"52528acbf7b146d7","mcp_get_code":{"code_sha256":"52528acbf7b146d7"}},{"arxiv_id":"2025.findings-acl.605","paper":null,"title":"arXiv:2025.findings-acl.605","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ytyz1307zzh/RefAug","path":"src/model/inference.py","file_url":"https://github.com/ytyz1307zzh/RefAug/blob/HEAD/src/model/inference.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7a35e514cdc2b061","mcp_get_code":{"code_sha256":"7a35e514cdc2b061"}}]}