{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/calculate-scores","entry":"calculate_scores","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":14,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":19,"n_samples_ran":8,"n_samples_fingerprinted":2,"n_places":19,"n_places_pointer_only":5,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":2,"ran_fixture":0,"ran":6,"unverified":11},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2607.23519","paper":"/paper/arxiv-2607-23519","title":"Auditing Alignment Controllability in LLMs via Political Axes","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"mbrcic/llm-political-steerability","path":"ingestion/01_collect/collect.py","file_url":"https://github.com/mbrcic/llm-political-steerability/blob/HEAD/ingestion/01_collect/collect.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e138e6d169cd6c8f","mcp_get_code":{"code_sha256":"e138e6d169cd6c8f"}},{"arxiv_id":"2602.11570","paper":"/paper/arxiv-2602-11570","title":"PRIME: A Process-Outcome Alignment Benchmark for Verifiable Reasoning in Mathematics and Engineering","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"wonderful9462/PRIME","path":"evaluation/calculate_accuracy.py","file_url":"https://github.com/wonderful9462/PRIME/blob/HEAD/evaluation/calculate_accuracy.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"32f5344f9c7eac24","mcp_get_code":{"code_sha256":"32f5344f9c7eac24"}},{"arxiv_id":"2510.08284","paper":"/paper/arxiv-2510-08284","title":"Neuron-Level Analysis of Cultural Understanding in Large Language Models","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"ynklab/CULNIG","path":"CULNIG/calc_neuron_score.py","file_url":"https://github.com/ynklab/CULNIG/blob/HEAD/CULNIG/calc_neuron_score.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"29f11de6c42b7989","mcp_get_code":{"code_sha256":"29f11de6c42b7989"}},{"arxiv_id":"2408.17443","paper":"/paper/bridging-episodes-and-semantics-a-novel","title":"HERMES: temporal-coHERent long-forM understanding with Episodes and Semantics","date":"2024-08-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joslefaure/HERMES","path":"lavis/tasks/moviecore_eval.py","file_url":"https://github.com/joslefaure/HERMES/blob/HEAD/lavis/tasks/moviecore_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6e0b77da286c6cb3","mcp_get_code":{"code_sha256":"6e0b77da286c6cb3"}},{"arxiv_id":"2406.11148","paper":"/paper/few-shot-recognition-via-stage-wise-augmented","title":"Few-Shot Recognition via Stage-Wise Retrieval-Augmented Finetuning","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tian1327/swat","path":"testing.py","file_url":"https://github.com/tian1327/swat/blob/HEAD/testing.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"482ccec6fd43050d","mcp_get_code":{"code_sha256":"482ccec6fd43050d"}},{"arxiv_id":"2403.14198","paper":"/paper/unleashing-unlabeled-data-a-paradigm-for","title":"Unleashing Unlabeled Data: A Paradigm for Cross-View Geo-Localization","date":"2024-03-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liguopeng0923/UCVGL","path":"train_crossview_sat.py","file_url":"https://github.com/liguopeng0923/UCVGL/blob/HEAD/train_crossview_sat.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ce5319f0ee24eecd","mcp_get_code":{"code_sha256":"ce5319f0ee24eecd"}},{"arxiv_id":"2312.15614","paper":"/paper/a-comprehensive-evaluation-of-parameter","title":"A Comprehensive Evaluation of Parameter-Efficient Fine-Tuning on Software Engineering Tasks","date":"2023-12-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zwtnju/peft","path":"clone/evaluator/evaluator.py","file_url":"https://github.com/zwtnju/peft/blob/HEAD/clone/evaluator/evaluator.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ccbc4540f0d52171","mcp_get_code":{"code_sha256":"ccbc4540f0d52171"}},{"arxiv_id":"2312.15614","paper":"/paper/a-comprehensive-evaluation-of-parameter","title":"A Comprehensive Evaluation of Parameter-Efficient Fine-Tuning on Software Engineering Tasks","date":"2023-12-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zwtnju/peft","path":"defect/evaluator/evaluator.py","file_url":"https://github.com/zwtnju/peft/blob/HEAD/defect/evaluator/evaluator.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"41d07eb6c8fa7fe3","mcp_get_code":{"code_sha256":"41d07eb6c8fa7fe3"}},{"arxiv_id":"2311.09677","paper":"/paper/r-tuning-teaching-large-language-models-to","title":"R-Tuning: Instructing Large Language Models to Say `I Don't Know'","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shizhediao/r-tuning","path":"evaluation/MMLU/results/calculate_ap.py","file_url":"https://github.com/shizhediao/r-tuning/blob/HEAD/evaluation/MMLU/results/calculate_ap.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4215906aecd42093","mcp_get_code":{"code_sha256":"4215906aecd42093"}},{"arxiv_id":"2206.08474","paper":"/paper/xlcost-a-benchmark-dataset-for-cross-lingual","title":"XLCoST: A Benchmark Dataset for Cross-lingual Code Intelligence","date":"2022-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"reddy-lab-code-research/xlcost","path":"code/codesearch/code2codesearch/evaluator/evaluator.py","file_url":"https://github.com/reddy-lab-code-research/xlcost/blob/HEAD/code/codesearch/code2codesearch/evaluator/evaluator.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c50ca9641c87e95d","mcp_get_code":{"code_sha256":"c50ca9641c87e95d"}},{"arxiv_id":"2206.08474","paper":"/paper/xlcost-a-benchmark-dataset-for-cross-lingual","title":"XLCoST: A Benchmark Dataset for Cross-lingual Code Intelligence","date":"2022-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"reddy-lab-code-research/xlcost","path":"code/codesearch/nl2codesearch/evaluator/evaluator.py","file_url":"https://github.com/reddy-lab-code-research/xlcost/blob/HEAD/code/codesearch/nl2codesearch/evaluator/evaluator.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7f2ade7249aa5a35","mcp_get_code":{"code_sha256":"7f2ade7249aa5a35"}},{"arxiv_id":"2112.02268","paper":"/paper/bridging-pre-trained-models-and-downstream","title":"Bridging Pre-trained Models and Downstream Tasks for Source Code Understanding","date":"2021-12-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wangdeze18/DACL","path":"algorithm_classification/evaluator/eva_MAP.py","file_url":"https://github.com/wangdeze18/DACL/blob/HEAD/algorithm_classification/evaluator/eva_MAP.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"790aecc4340767b7","mcp_get_code":{"code_sha256":"790aecc4340767b7"}},{"arxiv_id":"2112.02268","paper":"/paper/bridging-pre-trained-models-and-downstream","title":"Bridging Pre-trained Models and Downstream Tasks for Source Code Understanding","date":"2021-12-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wangdeze18/DACL","path":"algorithm_classification/evaluator/evaluator.py","file_url":"https://github.com/wangdeze18/DACL/blob/HEAD/algorithm_classification/evaluator/evaluator.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"83477337636fe464","mcp_get_code":{"code_sha256":"83477337636fe464"}},{"arxiv_id":"2112.02268","paper":"/paper/bridging-pre-trained-models-and-downstream","title":"Bridging Pre-trained Models and Downstream Tasks for Source Code Understanding","date":"2021-12-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wangdeze18/DACL","path":"algorithm_classification/evaluator/evaluator_score.py","file_url":"https://github.com/wangdeze18/DACL/blob/HEAD/algorithm_classification/evaluator/evaluator_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"77b01d53fe1ab5d8","mcp_get_code":{"code_sha256":"77b01d53fe1ab5d8"}},{"arxiv_id":"2011.01837","paper":"/paper/the-gap-on-gap-tackling-the-problem-of","title":"The Gap on GAP: Tackling the Problem of Differing Data Distributions in Bias-Measuring Datasets","date":"2020-11-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vid-koci/weightingGAP","path":"gap_scorer.py","file_url":"https://github.com/vid-koci/weightingGAP/blob/HEAD/gap_scorer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"db3dbbae0e8afc09","mcp_get_code":{"code_sha256":"db3dbbae0e8afc09"}},{"arxiv_id":"2010.02584","paper":"/paper/universal-natural-language-processing-with","title":"Universal Natural Language Processing with Limited Annotations: Try Few-shot Textual Entailment as a Start","date":"2020-10-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforce/UniversalFewShotNLP","path":"Coreference/gap_scorer.py","file_url":"https://github.com/salesforce/UniversalFewShotNLP/blob/HEAD/Coreference/gap_scorer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"84120ed8a77f6efa","mcp_get_code":{"code_sha256":"84120ed8a77f6efa"}},{"arxiv_id":"2010.02584","paper":"/paper/universal-natural-language-processing-with","title":"Universal Natural Language Processing with Limited Annotations: Try Few-shot Textual Entailment as a Start","date":"2020-10-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforce/UniversalFewShotNLP","path":"Non-Entail-Approach/gap_scorer_modified.py","file_url":"https://github.com/salesforce/UniversalFewShotNLP/blob/HEAD/Non-Entail-Approach/gap_scorer_modified.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"b2f6b157b471666d","mcp_get_code":{"code_sha256":"b2f6b157b471666d"}},{"arxiv_id":"2004.13590","paper":"/paper/maven-a-massive-general-domain-event","title":"MAVEN: A Massive General Domain Event Detection Dataset","date":"2020-04-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"THU-KEG/MAVEN-dataset","path":"baselines/DMBERT/run_ee.py","file_url":"https://github.com/THU-KEG/MAVEN-dataset/blob/HEAD/baselines/DMBERT/run_ee.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"06d672f377907b09","mcp_get_code":{"code_sha256":"06d672f377907b09"}},{"arxiv_id":"1806.06498","paper":"/paper/conditional-affordance-learning-for-driving","title":"Conditional Affordance Learning for Driving in Urban Environments","date":"2018-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xl-sr/CAL","path":"training/metrics.py","file_url":"https://github.com/xl-sr/CAL/blob/HEAD/training/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e077d404c9974a5b","mcp_get_code":{"code_sha256":"e077d404c9974a5b"}}]}