{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/test-answer","entry":"test_answer","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":6,"n_papers_ran":2,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":7,"n_samples_ran":4,"n_samples_fingerprinted":2,"n_places":8,"n_places_pointer_only":3,"by_status":{"ran_honours":0,"ran_violates":2,"ran_draft_wrong":0,"ran_fixture":0,"ran":2,"unverified":3},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2411.05282","paper":"/paper/microscopiq-accelerating-foundational-models","title":"MicroScopiQ: Accelerating Foundational Models through Outlier-Aware Microscaling Quantization","date":"2024-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"georgia-tech-synergy-lab/microscopiq-llm-quantization","path":"kv_quant/evaluation_gsm8k.py","file_url":"https://github.com/georgia-tech-synergy-lab/microscopiq-llm-quantization/blob/HEAD/kv_quant/evaluation_gsm8k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b6065f749757a201","mcp_get_code":{"code_sha256":"b6065f749757a201"}},{"arxiv_id":"2410.09343","paper":"/paper/elicit-llm-augmentation-via-external-in","title":"ELICIT: LLM Augmentation via External In-Context Capability","date":"2024-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lins-lab/elicit","path":"elicit.py","file_url":"https://github.com/lins-lab/elicit/blob/HEAD/elicit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5ebbe03eec41de4b","mcp_get_code":{"code_sha256":"5ebbe03eec41de4b"}},{"arxiv_id":"2410.04199","paper":"/paper/longgenbench-long-context-generation","title":"LongGenBench: Long-context Generation Benchmark","date":"2024-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dominic789654/longgenbench","path":"longgenbench_GSM8K_openai.py","file_url":"https://github.com/dominic789654/longgenbench/blob/HEAD/longgenbench_GSM8K_openai.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"09fbc17b3ceebd85","mcp_get_code":{"code_sha256":"09fbc17b3ceebd85"}},{"arxiv_id":"2410.04199","paper":"/paper/longgenbench-long-context-generation","title":"LongGenBench: Long-context Generation Benchmark","date":"2024-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dominic789654/longgenbench","path":"longgenbench_MMLU_openai.py","file_url":"https://github.com/dominic789654/longgenbench/blob/HEAD/longgenbench_MMLU_openai.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1a0df6d93a9464f3","mcp_get_code":{"code_sha256":"1a0df6d93a9464f3"}},{"arxiv_id":"2407.13647","paper":"/paper/weak-to-strong-reasoning","title":"Weak-to-Strong Reasoning","date":"2024-07-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gair-nlp/weak-to-strong-reasoning","path":"src/eval/evaluation.py","file_url":"https://github.com/gair-nlp/weak-to-strong-reasoning/blob/HEAD/src/eval/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ae9238672a6e13e1","mcp_get_code":{"code_sha256":"ae9238672a6e13e1"}},{"arxiv_id":"2403.05527","paper":"/paper/gear-an-efficient-kv-cache-compression","title":"GEAR: An Efficient KV Cache Compression Recipe for Near-Lossless Generative Inference of LLM","date":"2024-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haokang-timmy/gear","path":"GenerationBench/GenerationTest/evaluation_gsm8k.py","file_url":"https://github.com/haokang-timmy/gear/blob/HEAD/GenerationBench/GenerationTest/evaluation_gsm8k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b6065f749757a201","mcp_get_code":{"code_sha256":"b6065f749757a201"}},{"arxiv_id":"2401.12242","paper":"/paper/badchain-backdoor-chain-of-thought-prompting","title":"BadChain: Backdoor Chain-of-Thought Prompting for Large Language Models","date":"2024-01-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"django-jiang/badchain","path":"ans_eval/ASDiv.py","file_url":"https://github.com/django-jiang/badchain/blob/HEAD/ans_eval/ASDiv.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"91bd5c1829919390","mcp_get_code":{"code_sha256":"91bd5c1829919390"}},{"arxiv_id":"2401.12242","paper":"/paper/badchain-backdoor-chain-of-thought-prompting","title":"BadChain: Backdoor Chain-of-Thought Prompting for Large Language Models","date":"2024-01-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"django-jiang/badchain","path":"ans_eval/csqa.py","file_url":"https://github.com/django-jiang/badchain/blob/HEAD/ans_eval/csqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b91f9f3a9644f423","mcp_get_code":{"code_sha256":"b91f9f3a9644f423"}}]}