{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/collator","entry":"collator","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":16,"n_papers_ran":15,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":6,"n_samples_ran":5,"n_samples_fingerprinted":0,"n_places":16,"n_places_pointer_only":7,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":4,"ran_fixture":0,"ran":1,"unverified":1},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2411.11681","paper":"/paper/pspo-an-effective-process-supervised-policy","title":"PSPO*: An Effective Process-supervised Policy Optimization for Reasoning Alignment","date":"2024-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"direct-bit/pspo","path":"PSPO-code/ppo_trl.py","file_url":"https://github.com/direct-bit/pspo/blob/HEAD/PSPO-code/ppo_trl.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"db39a7bcb4f38d1f","mcp_get_code":{"code_sha256":"db39a7bcb4f38d1f"}},{"arxiv_id":"2410.06101","paper":"/paper/coevolving-with-the-other-you-fine-tuning-llm","title":"Coevolving with the Other You: Fine-Tuning LLM with Sequential Cooperative Multi-Agent Reinforcement Learning","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Harry67Hu/CORY","path":"imdb_train/cory.py","file_url":"https://github.com/Harry67Hu/CORY/blob/HEAD/imdb_train/cory.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bda36ad0c44dbd74","mcp_get_code":{"code_sha256":"bda36ad0c44dbd74"}},{"arxiv_id":"2407.05563","paper":"/paper/llmbox-a-comprehensive-library-for-large","title":"LLMBox: A Comprehensive Library for Large Language Models","date":"2024-07-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RUCAIBox/LLMBox","path":"training/ppo.py","file_url":"https://github.com/RUCAIBox/LLMBox/blob/HEAD/training/ppo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bda36ad0c44dbd74","mcp_get_code":{"code_sha256":"bda36ad0c44dbd74"}},{"arxiv_id":"2407.03856","paper":"/paper/q-adapter-training-your-llm-adapter-as-a","title":"Q-Adapter: Customizing Pre-trained LLMs to New Preferences with Forgetting Mitigation","date":"2024-07-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LAMDA-RL/Q-Adapter","path":"ppo_train.py","file_url":"https://github.com/LAMDA-RL/Q-Adapter/blob/HEAD/ppo_train.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fddcce152bc7d3bd","mcp_get_code":{"code_sha256":"fddcce152bc7d3bd"}},{"arxiv_id":"2406.10215","paper":"/paper/devbench-a-multimodal-developmental-benchmark","title":"DevBench: A multimodal developmental benchmark for language learning","date":"2024-06-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alvinwmtan/dev-bench","path":"data_handling.py","file_url":"https://github.com/alvinwmtan/dev-bench/blob/HEAD/data_handling.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5d819e34c493533d","mcp_get_code":{"code_sha256":"5d819e34c493533d"}},{"arxiv_id":"2406.07971","paper":"/paper/it-takes-two-on-the-seamlessness-between","title":"It Takes Two: On the Seamlessness between Reward and Policy Model in RLHF","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"taiminglu/seamless","path":"code/train/rl.py","file_url":"https://github.com/taiminglu/seamless/blob/HEAD/code/train/rl.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fddcce152bc7d3bd","mcp_get_code":{"code_sha256":"fddcce152bc7d3bd"}},{"arxiv_id":"2406.05654","paper":"/paper/domainrag-a-chinese-benchmark-for-evaluating","title":"DomainRAG: A Chinese Benchmark for Evaluating Domain-specific Retrieval-Augmented Generation","date":"2024-06-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ShootingWong/DomainRAG","path":"BCM/evaluation/tasks/Datasets.py","file_url":"https://github.com/ShootingWong/DomainRAG/blob/HEAD/BCM/evaluation/tasks/Datasets.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bda36ad0c44dbd74","mcp_get_code":{"code_sha256":"bda36ad0c44dbd74"}},{"arxiv_id":"2405.06639","paper":"/paper/value-augmented-sampling-for-language-model","title":"Value Augmented Sampling for Language Model Alignment and Personalization","date":"2024-05-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"idanshen/Value-Augmented-Sampling","path":"tinyllama_hh.py","file_url":"https://github.com/idanshen/Value-Augmented-Sampling/blob/HEAD/tinyllama_hh.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fddcce152bc7d3bd","mcp_get_code":{"code_sha256":"fddcce152bc7d3bd"}},{"arxiv_id":"2402.13623","paper":"/paper/flame-self-supervised-low-resource-taxonomy","title":"FLAME: Self-Supervised Low-Resource Taxonomy Expansion using Large Language Models","date":"2024-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sahilmishra0012/flame","path":"kshot_ppo.py","file_url":"https://github.com/sahilmishra0012/flame/blob/HEAD/kshot_ppo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"94816bbb3c11acec","mcp_get_code":{"code_sha256":"94816bbb3c11acec"}},{"arxiv_id":"2312.07551","paper":"/paper/language-model-alignment-with-elastic-reset-1","title":"Language Model Alignment with Elastic Reset","date":"2023-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mnoukhov/elastic-reset","path":"stackllama/er_training.py","file_url":"https://github.com/mnoukhov/elastic-reset/blob/HEAD/stackllama/er_training.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bda36ad0c44dbd74","mcp_get_code":{"code_sha256":"bda36ad0c44dbd74"}},{"arxiv_id":"2310.11564","paper":"/paper/personalized-soups-personalized-large","title":"Personalized Soups: Personalized Large Language Model Alignment via Post-hoc Parameter Merging","date":"2023-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joeljang/rlphf","path":"training/pmorl.py","file_url":"https://github.com/joeljang/rlphf/blob/HEAD/training/pmorl.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bda36ad0c44dbd74","mcp_get_code":{"code_sha256":"bda36ad0c44dbd74"}},{"arxiv_id":"2309.16155","paper":"/paper/the-trickle-down-impact-of-reward-in","title":"The Trickle-down Impact of Reward (In-)consistency on RLHF","date":"2023-09-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shadowkiller33/contrast-instruction","path":"code/int8_rl_training.py","file_url":"https://github.com/shadowkiller33/contrast-instruction/blob/HEAD/code/int8_rl_training.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bda36ad0c44dbd74","mcp_get_code":{"code_sha256":"bda36ad0c44dbd74"}},{"arxiv_id":"2304.08177","paper":"/paper/efficient-and-effective-text-encoding-for","title":"Efficient and Effective Text Encoding for Chinese LLaMA and Alpaca","date":"2023-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jackaduma/Alpaca-LoRA-RLHF-PyTorch","path":"tuning_lm_with_rl.py","file_url":"https://github.com/jackaduma/Alpaca-LoRA-RLHF-PyTorch/blob/HEAD/tuning_lm_with_rl.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bda36ad0c44dbd74","mcp_get_code":{"code_sha256":"bda36ad0c44dbd74"}},{"arxiv_id":"2301.02344","paper":"/paper/trojanpuzzle-covertly-poisoning-code","title":"TrojanPuzzle: Covertly Poisoning Code-Suggestion Models","date":"2023-01-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/codegenerationpoisoning","path":"SalesforceCodeGen/training/fine_tune.py","file_url":"https://github.com/microsoft/codegenerationpoisoning/blob/HEAD/SalesforceCodeGen/training/fine_tune.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0bbeee52d85f7d22","mcp_get_code":{"code_sha256":"0bbeee52d85f7d22"}},{"arxiv_id":"2204.07705","paper":"/paper/benchmarking-generalization-via-in-context","title":"Super-NaturalInstructions: Generalization via Declarative Instructions on 1600+ NLP Tasks","date":"2022-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"naver-ai/rethinking-proxy-reward","path":"run_ppo.py","file_url":"https://github.com/naver-ai/rethinking-proxy-reward/blob/HEAD/run_ppo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bda36ad0c44dbd74","mcp_get_code":{"code_sha256":"bda36ad0c44dbd74"}},{"arxiv_id":"1909.08593","paper":"/paper/fine-tuning-language-models-from-human","title":"Fine-Tuning Language Models from Human Preferences","date":"2019-09-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"bda36ad0c44dbd74","mcp_get_code":{"code_sha256":"bda36ad0c44dbd74"}}]}