{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/call","entry":"call","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":6,"n_papers_ran":2,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":9,"n_samples_ran":4,"n_samples_fingerprinted":0,"n_places":9,"n_places_pointer_only":2,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":4,"unverified":5},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.26119","paper":"/paper/arxiv-2608-26119","title":"DeflectBench: A Benchmark for Evaluating Rhetorical Fallacy Generation in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"ArtKanke/DeflectBench","path":"api/deepseek_client.py","file_url":"https://github.com/ArtKanke/DeflectBench/blob/HEAD/api/deepseek_client.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7ddce77b9e7eaf8d","mcp_get_code":{"code_sha256":"7ddce77b9e7eaf8d"}},{"arxiv_id":"2608.26119","paper":"/paper/arxiv-2608-26119","title":"DeflectBench: A Benchmark for Evaluating Rhetorical Fallacy Generation in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"ArtKanke/DeflectBench","path":"api/openai_client.py","file_url":"https://github.com/ArtKanke/DeflectBench/blob/HEAD/api/openai_client.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ea617968c1d64b1d","mcp_get_code":{"code_sha256":"ea617968c1d64b1d"}},{"arxiv_id":"2608.26119","paper":"/paper/arxiv-2608-26119","title":"DeflectBench: A Benchmark for Evaluating Rhetorical Fallacy Generation in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"ArtKanke/DeflectBench","path":"api/xai_client.py","file_url":"https://github.com/ArtKanke/DeflectBench/blob/HEAD/api/xai_client.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"01e3cd66934c566e","mcp_get_code":{"code_sha256":"01e3cd66934c566e"}},{"arxiv_id":"2608.26119","paper":"/paper/arxiv-2608-26119","title":"DeflectBench: A Benchmark for Evaluating Rhetorical Fallacy Generation in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"ArtKanke/DeflectBench","path":"api/anthropic_client.py","file_url":"https://github.com/ArtKanke/DeflectBench/blob/HEAD/api/anthropic_client.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d4bf8ab065ce119f","mcp_get_code":{"code_sha256":"d4bf8ab065ce119f"}},{"arxiv_id":"2607.10569","paper":"/paper/arxiv-2607-10569","title":"When Does Restricting a Coding Agent to execute_code Help? A Regime × Agent-Design Ablation","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"hyang0129/onlycodes","path":"exec_server/mcp_bridge.py","file_url":"https://github.com/hyang0129/onlycodes/blob/HEAD/exec_server/mcp_bridge.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6b3a5f0a1365973c","mcp_get_code":{"code_sha256":"6b3a5f0a1365973c"}},{"arxiv_id":"2605.07093","paper":"/paper/arxiv-2605-07093","title":"The Translation Tax Is Not a Scalar: A Counterfactual Audit of English-Source Cue Inheritance in Chinese Multilingual Benchmarks","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"chi-mi-rvard/translation-tax-supplement","path":"run_include_lean.py","file_url":"https://github.com/chi-mi-rvard/translation-tax-supplement/blob/HEAD/run_include_lean.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"edd21fee9d637703","mcp_get_code":{"code_sha256":"edd21fee9d637703"}},{"arxiv_id":"2307.02762","paper":"/paper/prd-peer-rank-and-discussion-improve-large","title":"PRD: Peer Rank and Discussion Improve Large Language Model based Evaluations","date":"2023-07-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bcdnlp/prd","path":"peer_discussion/openai_api.py","file_url":"https://github.com/bcdnlp/prd/blob/HEAD/peer_discussion/openai_api.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"306d213aec7cb87f","mcp_get_code":{"code_sha256":"306d213aec7cb87f"}},{"arxiv_id":"2008.11790","paper":"/paper/mutagan-a-seq2seq-gan-framework-to-predict","title":"MutaGAN: A Seq2seq GAN Framework to Predict Mutations of Evolving Protein Populations","date":"2020-08-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"anuprulez/clade_prediction","path":"encoder_decoder_attention.py","file_url":"https://github.com/anuprulez/clade_prediction/blob/HEAD/encoder_decoder_attention.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a179307e89c155b9","mcp_get_code":{"code_sha256":"a179307e89c155b9"}},{"arxiv_id":"1912.01217","paper":"/paper/safelife-10-exploring-side-effects-in-complex","title":"SafeLife 1.0: Exploring Side Effects in Complex Environments","date":"2019-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"PartnershipOnAI/safelife","path":"safelife/env_wrappers.py","file_url":"https://github.com/PartnershipOnAI/safelife/blob/HEAD/safelife/env_wrappers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ea59c489547b4c9b","mcp_get_code":{"code_sha256":"ea59c489547b4c9b"}}]}