{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/format-best","entry":"format_best","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":8,"n_papers_ran":0,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":1,"n_samples_ran":0,"n_samples_fingerprinted":0,"n_places":8,"n_places_pointer_only":3,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":0,"unverified":1},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2509.16598","paper":"/paper/arxiv-2509-16598","title":"PruneCD: Contrasting Pruned Self Model to Improve Decoding Factuality","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"hoeng4/PruneCD","path":"1_factual_layer_search/tfqa_mc.py","file_url":"https://github.com/hoeng4/PruneCD/blob/HEAD/1_factual_layer_search/tfqa_mc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"158b8847881a64b3","mcp_get_code":{"code_sha256":"158b8847881a64b3"}},{"arxiv_id":"2506.02973","paper":"/paper/expanding-before-inferring-enhancing","title":"Expanding before Inferring: Enhancing Factuality in Large Language Models through Premature Layers Interpolation","date":"2025-06-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CuSO4-Chen/PLI","path":"src/evaluation/truthfulqa_eval.py","file_url":"https://github.com/CuSO4-Chen/PLI/blob/HEAD/src/evaluation/truthfulqa_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"158b8847881a64b3","mcp_get_code":{"code_sha256":"158b8847881a64b3"}},{"arxiv_id":"2504.04635","paper":null,"title":"arXiv:2504.04635","date":null,"month_inferred_from_arxiv_id":"2025-04","title_source":null,"repo":"patqdasilva/steering-off-course","path":"DoLa/tfqa_mc_eval.py","file_url":"https://github.com/patqdasilva/steering-off-course/blob/HEAD/DoLa/tfqa_mc_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"158b8847881a64b3","mcp_get_code":{"code_sha256":"158b8847881a64b3"}},{"arxiv_id":"2410.21352","paper":"/paper/llmcbench-benchmarking-large-language-model","title":"LLMCBench: Benchmarking Large Language Model Compression for Efficient Deployment","date":"2024-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AboveParadise/LLMCBench","path":"evaluate_tQA.py","file_url":"https://github.com/AboveParadise/LLMCBench/blob/HEAD/evaluate_tQA.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"158b8847881a64b3","mcp_get_code":{"code_sha256":"158b8847881a64b3"}},{"arxiv_id":"2408.12325","paper":"/paper/improving-factuality-in-large-language-models","title":"Improving Factuality in Large Language Models via Decoding-Time Hallucinatory and Truthful Comparators","date":"2024-08-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ydk122024/cdt","path":"src/benchmark_evaluation/truthfulqa_eval.py","file_url":"https://github.com/ydk122024/cdt/blob/HEAD/src/benchmark_evaluation/truthfulqa_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"158b8847881a64b3","mcp_get_code":{"code_sha256":"158b8847881a64b3"}},{"arxiv_id":"2401.05930","paper":"/paper/sh2-self-highlighted-hesitation-helps-you","title":"SH2: Self-Highlighted Hesitation Helps You Decode More Truthfully","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"0-kaikai-0/sh2","path":"tfqa_mc_eval.py","file_url":"https://github.com/0-kaikai-0/sh2/blob/HEAD/tfqa_mc_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"158b8847881a64b3","mcp_get_code":{"code_sha256":"158b8847881a64b3"}},{"arxiv_id":"2312.15710","paper":"/paper/alleviating-hallucinations-of-large-language","title":"Alleviating Hallucinations of Large Language Models through Induced Hallucinations","date":"2023-12-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hillzhang1999/icd","path":"src/benchmark_evaluation/truthfulqa_eval.py","file_url":"https://github.com/hillzhang1999/icd/blob/HEAD/src/benchmark_evaluation/truthfulqa_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"158b8847881a64b3","mcp_get_code":{"code_sha256":"158b8847881a64b3"}},{"arxiv_id":"2312.04333","paper":"/paper/beyond-surface-probing-llama-across-scales","title":"Is Bigger and Deeper Always Better? Probing LLaMA Across Scales and Layers","date":"2023-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nuochenpku/llama_analysis","path":"factural_eval.py","file_url":"https://github.com/nuochenpku/llama_analysis/blob/HEAD/factural_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"158b8847881a64b3","mcp_get_code":{"code_sha256":"158b8847881a64b3"}}]}