{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/format-prompt","entry":"format_prompt","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":45,"n_papers_ran":28,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":53,"n_samples_ran":34,"n_samples_fingerprinted":6,"n_places":56,"n_places_pointer_only":23,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":10,"ran_fixture":1,"ran":23,"unverified":19},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.02618","paper":"/paper/arxiv-2608-02618","title":"Beyond the Hivemind: Escaping LLM Homogeneity via Meta-Persona Anchoring and Sequential Temperature Scaling","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"aMa2210/beyond-the-hivemind","path":"src/pipeline/step2_generate_responses.py","file_url":"https://github.com/aMa2210/beyond-the-hivemind/blob/HEAD/src/pipeline/step2_generate_responses.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"16b8fc2643349221","mcp_get_code":{"code_sha256":"16b8fc2643349221"}},{"arxiv_id":"2605.24286","paper":"/paper/arxiv-2605-24286","title":"Faithfulness as Information Flow: Evaluating and Training Faithful Chain-of-Thought Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"safety-research/faithful-cot","path":"eval/evaluate_hacking_ratio.py","file_url":"https://github.com/safety-research/faithful-cot/blob/HEAD/eval/evaluate_hacking_ratio.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"980d1a07fda8a0db","mcp_get_code":{"code_sha256":"980d1a07fda8a0db"}},{"arxiv_id":"2605.18352","paper":"/paper/arxiv-2605-18352","title":"Presupposition and Reasoning in Conditionals: A Theory-Based Study of Humans and LLMs","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"proviso-bench/Presupposition-and-Reasoning-in-Conditionals","path":"inference/closedsource/inference_closedsource.py","file_url":"https://github.com/proviso-bench/Presupposition-and-Reasoning-in-Conditionals/blob/HEAD/inference/closedsource/inference_closedsource.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"248a640efdc7464c","mcp_get_code":{"code_sha256":"248a640efdc7464c"}},{"arxiv_id":"2604.14513","paper":"/paper/arxiv-2604-14513","title":"PeerPrism: Peer Evaluation Expertise vs Review-writing AI","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Reviewerly-Inc/PeerPrism","path":"02_review_transformation/transformations/hybrid_reviews.py","file_url":"https://github.com/Reviewerly-Inc/PeerPrism/blob/HEAD/02_review_transformation/transformations/hybrid_reviews.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"68efb2a62e8a8147","mcp_get_code":{"code_sha256":"68efb2a62e8a8147"}},{"arxiv_id":"2603.08286","paper":"/paper/arxiv-2603-08286","title":"LAMUS: A Large-Scale Corpus for Legal Argument Mining from U.S. Caselaw using LLMs","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"LavanyaPobbathi/LAMUS","path":"code/experiment/C_finetuning_ablation_v2.py","file_url":"https://github.com/LavanyaPobbathi/LAMUS/blob/HEAD/code/experiment/C_finetuning_ablation_v2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9ccf27d30fdb741b","mcp_get_code":{"code_sha256":"9ccf27d30fdb741b"}},{"arxiv_id":"2602.15143","paper":"/paper/arxiv-2602-15143","title":"Protecting Language Models Against Unauthorized Distillation through Trace Rewriting","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"xhOwenMa/trace-rewriting","path":"optimize/propose_candidates.py","file_url":"https://github.com/xhOwenMa/trace-rewriting/blob/HEAD/optimize/propose_candidates.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fb2d68d136e3fdb1","mcp_get_code":{"code_sha256":"fb2d68d136e3fdb1"}},{"arxiv_id":"2602.03876","paper":"/paper/arxiv-2602-03876","title":"GOPO: Policy Optimization using Ranked Rewards","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"friendshipkim/gopo","path":"generate/generate_completions.py","file_url":"https://github.com/friendshipkim/gopo/blob/HEAD/generate/generate_completions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7720ea43a933acbd","mcp_get_code":{"code_sha256":"7720ea43a933acbd"}},{"arxiv_id":"2602.02510","paper":"/paper/arxiv-2602-02510","title":"Beyond Translation: Cross-Cultural Meme Transcreation with Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"AIM-SCU/MemeXGen","path":"judges/_common/prompt.py","file_url":"https://github.com/AIM-SCU/MemeXGen/blob/HEAD/judges/_common/prompt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3b648343d9b97d8e","mcp_get_code":{"code_sha256":"3b648343d9b97d8e"}},{"arxiv_id":"2601.22124","paper":"/paper/arxiv-2601-22124","title":"Toward Federated Large Language Models in Medicine: A Parameter-Efficient Framework for Privacy-Preserving, Multi-Institutional Adaptation","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Yale-BIDS-Chen-Lab/FL_LLM_Med","path":"src/command/medical_re_csv_parse.py","file_url":"https://github.com/Yale-BIDS-Chen-Lab/FL_LLM_Med/blob/HEAD/src/command/medical_re_csv_parse.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1c1226b88bdf7048","mcp_get_code":{"code_sha256":"1c1226b88bdf7048"}},{"arxiv_id":"2510.00685","paper":"/paper/arxiv-2510-00685","title":"Stochastic Self-Organization in Multi-Agent Systems","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"tnurbek/selforg","path":"evaluations/evaluate_xverify.py","file_url":"https://github.com/tnurbek/selforg/blob/HEAD/evaluations/evaluate_xverify.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ca9fc2b740a5a87e","mcp_get_code":{"code_sha256":"ca9fc2b740a5a87e"}},{"arxiv_id":"2505.21795","paper":"/paper/sansa-unleashing-the-hidden-semantics-in-sam2","title":"SANSA: Unleashing the Hidden Semantics in SAM2 for Few-Shot Segmentation","date":"2025-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ClaudiaCuttano/SANSA","path":"util/demo_sansa.py","file_url":"https://github.com/ClaudiaCuttano/SANSA/blob/HEAD/util/demo_sansa.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9f2171cc1758c725","mcp_get_code":{"code_sha256":"9f2171cc1758c725"}},{"arxiv_id":"2504.08231","paper":"/paper/out-of-style-rag-s-fragility-to-linguistic","title":"Out of Style: RAG's Fragility to Linguistic Variation","date":"2025-04-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"springcty/rag-fragility-to-linguistic-variation","path":"LLM_generation/utils/vllm_inference.py","file_url":"https://github.com/springcty/rag-fragility-to-linguistic-variation/blob/HEAD/LLM_generation/utils/vllm_inference.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"afab3aee4c1d97a0","mcp_get_code":{"code_sha256":"afab3aee4c1d97a0"}},{"arxiv_id":"2504.05632","paper":"/paper/reasoning-towards-fairness-mitigating-bias-in","title":"Reasoning Towards Fairness: Mitigating Bias in Language Models through Reasoning-Guided Fine-Tuning","date":"2025-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Sanchit-404/Reasoing-Towards-Fairness","path":"scripts/finetune_on_traces.py","file_url":"https://github.com/Sanchit-404/Reasoing-Towards-Fairness/blob/HEAD/scripts/finetune_on_traces.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6b9e5582f91b19d7","mcp_get_code":{"code_sha256":"6b9e5582f91b19d7"}},{"arxiv_id":"2503.23513","paper":"/paper/rare-retrieval-augmented-reasoning-modeling","title":"RARE: Retrieval-Augmented Reasoning Modeling","date":"2025-03-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"open-dataflow/rare","path":"inference/vllm_infer_mm.py","file_url":"https://github.com/open-dataflow/rare/blob/HEAD/inference/vllm_infer_mm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"43b4dce1bb401779","mcp_get_code":{"code_sha256":"43b4dce1bb401779"}},{"arxiv_id":"2503.23513","paper":"/paper/rare-retrieval-augmented-reasoning-modeling","title":"RARE: Retrieval-Augmented Reasoning Modeling","date":"2025-03-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"open-dataflow/rare","path":"inference/vllm_infer_text.py","file_url":"https://github.com/open-dataflow/rare/blob/HEAD/inference/vllm_infer_text.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0fb897a1530a0ca3","mcp_get_code":{"code_sha256":"0fb897a1530a0ca3"}},{"arxiv_id":"2503.05188","paper":null,"title":"arXiv:2503.05188","date":null,"month_inferred_from_arxiv_id":"2025-03","title_source":null,"repo":"BugMakerzzz/CRISP","path":"crisp_reason.py","file_url":"https://github.com/BugMakerzzz/CRISP/blob/HEAD/crisp_reason.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4b6f78016356fabe","mcp_get_code":{"code_sha256":"4b6f78016356fabe"}},{"arxiv_id":"2503.00223","paper":"/paper/deepretrieval-powerful-query-generation-for","title":"DeepRetrieval: Hacking Real Search Engines and Retrievers with Large Language Models via Reinforcement Learning","date":"2025-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pat-jj/deepretrieval","path":"query_rewrite.py","file_url":"https://github.com/pat-jj/deepretrieval/blob/HEAD/query_rewrite.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f4549a2164452ce1","mcp_get_code":{"code_sha256":"f4549a2164452ce1"}},{"arxiv_id":"2503.00223","paper":"/paper/deepretrieval-powerful-query-generation-for","title":"DeepRetrieval: Hacking Real Search Engines and Retrievers with Large Language Models via Reinforcement Learning","date":"2025-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pat-jj/deepretrieval","path":"code/src/query_rewrite.py","file_url":"https://github.com/pat-jj/deepretrieval/blob/HEAD/code/src/query_rewrite.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dbf81b46206bc092","mcp_get_code":{"code_sha256":"dbf81b46206bc092"}},{"arxiv_id":"2502.17407","paper":"/paper/linguistic-generalizability-of-test-time","title":"Linguistic Generalizability of Test-Time Scaling in Mathematical Reasoning","date":"2025-02-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gauss5930/mclm","path":"src/prm.py","file_url":"https://github.com/gauss5930/mclm/blob/HEAD/src/prm.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b960d1f9e15f6bd3","mcp_get_code":{"code_sha256":"b960d1f9e15f6bd3"}},{"arxiv_id":"2502.14507","paper":"/paper/can-llms-simulate-l2-english-dialogue-an","title":"Can LLMs Simulate L2-English Dialogue? An Information-Theoretic Analysis of L1-Dependent Biases","date":"2025-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RenaGao/LLMPirorknowledge","path":"lib/annotation.py","file_url":"https://github.com/RenaGao/LLMPirorknowledge/blob/HEAD/lib/annotation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6340c353203620a9","mcp_get_code":{"code_sha256":"6340c353203620a9"}},{"arxiv_id":"2501.09620","paper":"/paper/beyond-reward-hacking-causal-rewards-for","title":"Beyond Reward Hacking: Causal Rewards for Large Language Model Alignment","date":"2025-01-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tatsu-lab/alpaca_farm","path":"src/alpaca_farm/data_preprocessor.py","file_url":"https://github.com/tatsu-lab/alpaca_farm/blob/HEAD/src/alpaca_farm/data_preprocessor.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a71a4bdc37586fe2","mcp_get_code":{"code_sha256":"a71a4bdc37586fe2"}},{"arxiv_id":"2412.06660","paper":"/paper/mumu-llama-multi-modal-music-understanding","title":"MuMu-LLaMA: Multi-modal Music Understanding and Generation via Large Language Models","date":null,"month_inferred_from_arxiv_id":"2024-12","title_source":"archive","repo":"shansongliu/MuMu-LLaMA","path":"MuMu-LLaMA/llama/utils.py","file_url":"https://github.com/shansongliu/MuMu-LLaMA/blob/HEAD/MuMu-LLaMA/llama/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fa5a132156cdaf6d","mcp_get_code":{"code_sha256":"fa5a132156cdaf6d"}},{"arxiv_id":"2411.16318","paper":"/paper/one-diffusion-to-generate-them-all","title":"One Diffusion to Generate Them All","date":"2024-11-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lehduong/onediffusion","path":"gradio_demo.py","file_url":"https://github.com/lehduong/onediffusion/blob/HEAD/gradio_demo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"362b432acdaa105e","mcp_get_code":{"code_sha256":"362b432acdaa105e"}},{"arxiv_id":"2410.14516","paper":"/paper/do-llms-know-internally-when-they-follow","title":"Do LLMs \"know\" internally when they follow instructions?","date":"2024-10-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"apple/ml-internal-llms-instruction-following","path":"run/save_LLMs_activations.py","file_url":"https://github.com/apple/ml-internal-llms-instruction-following/blob/HEAD/run/save_LLMs_activations.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"ced84377f1394101","mcp_get_code":{"code_sha256":"ced84377f1394101"}},{"arxiv_id":"2410.08067","paper":"/paper/reward-augmented-data-enhances-direct","title":"Reward-Augmented Data Enhances Direct Preference Alignment of LLMs","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shenao-zhang/reward-augmented-preference","path":"scripts/preprocess.py","file_url":"https://github.com/shenao-zhang/reward-augmented-preference/blob/HEAD/scripts/preprocess.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1290bdbfd3cfa6e5","mcp_get_code":{"code_sha256":"1290bdbfd3cfa6e5"}},{"arxiv_id":"2409.15380","paper":"/paper/kalahi-a-handcrafted-grassroots-cultural-llm","title":"Kalahi: A handcrafted, grassroots cultural LLM evaluation suite for Filipino","date":"2024-09-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aisingapore/kalahi","path":"kalahi/utilities.py","file_url":"https://github.com/aisingapore/kalahi/blob/HEAD/kalahi/utilities.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"b89a6d986d96e4fa","mcp_get_code":{"code_sha256":"b89a6d986d96e4fa"}},{"arxiv_id":"2409.15254","paper":"/paper/archon-an-architecture-search-framework-for","title":"Archon: An Architecture Search Framework for Inference-Time Techniques","date":"2024-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"scalingintelligence/archon","path":"src/archon/completions/utils.py","file_url":"https://github.com/scalingintelligence/archon/blob/HEAD/src/archon/completions/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a3621beb3eb4d82a","mcp_get_code":{"code_sha256":"a3621beb3eb4d82a"}},{"arxiv_id":"2409.15154","paper":"/paper/rmcbench-benchmarking-large-language-models","title":"RMCBench: Benchmarking Large Language Models' Resistance to Malicious Code","date":"2024-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qing-yuan233/RMCBench","path":"script/run_open_llm.py","file_url":"https://github.com/qing-yuan233/RMCBench/blob/HEAD/script/run_open_llm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"e0e266948769ee25","mcp_get_code":{"code_sha256":"e0e266948769ee25"}},{"arxiv_id":"2406.14155","paper":"/paper/aligning-large-language-models-with-diverse","title":"Aligning Large Language Models with Diverse Political Viewpoints","date":"2024-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dominiksinsaarland/aligning-LLMs-with-political-views","path":"src/inference.py","file_url":"https://github.com/dominiksinsaarland/aligning-LLMs-with-political-views/blob/HEAD/src/inference.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"223868879693b620","mcp_get_code":{"code_sha256":"223868879693b620"}},{"arxiv_id":"2406.13629","paper":"/paper/instructrag-instructing-retrieval-augmented","title":"InstructRAG: Instructing Retrieval-Augmented Generation via Self-Synthesized Rationales","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"weizhepei/instructrag","path":"src/data_utils.py","file_url":"https://github.com/weizhepei/instructrag/blob/HEAD/src/data_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"106a37e7db749702","mcp_get_code":{"code_sha256":"106a37e7db749702"}},{"arxiv_id":"2406.04823","paper":"/paper/berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltgoslo/bert-in-context","path":"language-modeling/hellaswag.py","file_url":"https://github.com/ltgoslo/bert-in-context/blob/HEAD/language-modeling/hellaswag.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"28110ec97c757855","mcp_get_code":{"code_sha256":"28110ec97c757855"}},{"arxiv_id":"2406.04823","paper":"/paper/berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltgoslo/bert-in-context","path":"language-modeling/story_cloze.py","file_url":"https://github.com/ltgoslo/bert-in-context/blob/HEAD/language-modeling/story_cloze.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"11065df7f8515818","mcp_get_code":{"code_sha256":"11065df7f8515818"}},{"arxiv_id":"2406.04823","paper":"/paper/berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltgoslo/bert-in-context","path":"language-modeling/winograd.py","file_url":"https://github.com/ltgoslo/bert-in-context/blob/HEAD/language-modeling/winograd.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"522de10ce348cb0c","mcp_get_code":{"code_sha256":"522de10ce348cb0c"}},{"arxiv_id":"2406.04823","paper":"/paper/berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltgoslo/bert-in-context","path":"glue/glue.py","file_url":"https://github.com/ltgoslo/bert-in-context/blob/HEAD/glue/glue.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8131636afc6ecc7d","mcp_get_code":{"code_sha256":"8131636afc6ecc7d"}},{"arxiv_id":"2406.04823","paper":"/paper/berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltgoslo/bert-in-context","path":"glue/multirc.py","file_url":"https://github.com/ltgoslo/bert-in-context/blob/HEAD/glue/multirc.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1edc73f4d0b284d9","mcp_get_code":{"code_sha256":"1edc73f4d0b284d9"}},{"arxiv_id":"2406.04823","paper":"/paper/berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltgoslo/bert-in-context","path":"needle-in-a-haystack/haystack.py","file_url":"https://github.com/ltgoslo/bert-in-context/blob/HEAD/needle-in-a-haystack/haystack.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"67264be51677507e","mcp_get_code":{"code_sha256":"67264be51677507e"}},{"arxiv_id":"2406.04823","paper":"/paper/berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltgoslo/bert-in-context","path":"question-answering/arc.py","file_url":"https://github.com/ltgoslo/bert-in-context/blob/HEAD/question-answering/arc.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f423eb95fad23611","mcp_get_code":{"code_sha256":"f423eb95fad23611"}},{"arxiv_id":"2406.04823","paper":"/paper/berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltgoslo/bert-in-context","path":"question-answering/natural_questions.py","file_url":"https://github.com/ltgoslo/bert-in-context/blob/HEAD/question-answering/natural_questions.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4f852be96a94479f","mcp_get_code":{"code_sha256":"4f852be96a94479f"}},{"arxiv_id":"2406.04823","paper":"/paper/berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltgoslo/bert-in-context","path":"question-answering/openbookqa.py","file_url":"https://github.com/ltgoslo/bert-in-context/blob/HEAD/question-answering/openbookqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"41aac6a681c0af69","mcp_get_code":{"code_sha256":"41aac6a681c0af69"}},{"arxiv_id":"2405.17374","paper":"/paper/navigating-the-safety-landscape-measuring","title":"Navigating the Safety Landscape: Measuring Risks in Finetuning Large Language Models","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ShengYun-Peng/llm-landscape","path":"src/llm/util.py","file_url":"https://github.com/ShengYun-Peng/llm-landscape/blob/HEAD/src/llm/util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c50b059fdffe4c7c","mcp_get_code":{"code_sha256":"c50b059fdffe4c7c"}},{"arxiv_id":"2404.15846","paper":"/paper/from-complex-to-simple-enhancing-multi","title":"From Complex to Simple: Enhancing Multi-Constraint Complex Instruction Following Ability of Large Language Models","date":"2024-04-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meowpass/followcomplexinstruction","path":"get_data/do_inference.py","file_url":"https://github.com/meowpass/followcomplexinstruction/blob/HEAD/get_data/do_inference.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ee0fb943a653c769","mcp_get_code":{"code_sha256":"ee0fb943a653c769"}},{"arxiv_id":"2403.19589","paper":"/paper/tod3cap-towards-3d-dense-captioning-in","title":"TOD3Cap: Towards 3D Dense Captioning in Outdoor Scenes","date":"2024-03-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jxbbb/tod3cap","path":"tod3cap_camera/llama/utils.py","file_url":"https://github.com/jxbbb/tod3cap/blob/HEAD/tod3cap_camera/llama/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7006a7479b6f348f","mcp_get_code":{"code_sha256":"7006a7479b6f348f"}},{"arxiv_id":"2403.06833","paper":"/paper/can-llms-separate-instructions-from-data-and","title":"Can LLMs Separate Instructions From Data? And What Do We Even Mean By That?","date":"2024-03-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"egozverev/Shold-It-Be-Executed-Or-Processed","path":"model_eval/get_model_outputs.py","file_url":"https://github.com/egozverev/Shold-It-Be-Executed-Or-Processed/blob/HEAD/model_eval/get_model_outputs.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ed71b768b354fced","mcp_get_code":{"code_sha256":"ed71b768b354fced"}},{"arxiv_id":"2402.05133","paper":"/paper/personalized-language-modeling-from","title":"Personalized Language Modeling from Personalized Human Feedback","date":"2024-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"humainlab/personalized_rlhf","path":"evaluate/alpaca_farm/data_preprocessor.py","file_url":"https://github.com/humainlab/personalized_rlhf/blob/HEAD/evaluate/alpaca_farm/data_preprocessor.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a71a4bdc37586fe2","mcp_get_code":{"code_sha256":"a71a4bdc37586fe2"}},{"arxiv_id":"2401.15884","paper":"/paper/corrective-retrieval-augmented-generation","title":"Corrective Retrieval Augmented Generation","date":"2024-01-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"huskyinsalt/crag","path":"scripts/CRAG_Inference.py","file_url":"https://github.com/huskyinsalt/crag/blob/HEAD/scripts/CRAG_Inference.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3881f4fd643eb25c","mcp_get_code":{"code_sha256":"3881f4fd643eb25c"}},{"arxiv_id":"2401.06949","paper":"/paper/organa-a-robotic-assistant-for-automated","title":"ORGANA: A Robotic Assistant for Automated Chemistry Experimentation and Characterization","date":"2024-01-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ac-rad/organa","path":"exp_task_manager.py","file_url":"https://github.com/ac-rad/organa/blob/HEAD/exp_task_manager.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"53def563fe69199b","mcp_get_code":{"code_sha256":"53def563fe69199b"}},{"arxiv_id":"2401.06949","paper":"/paper/organa-a-robotic-assistant-for-automated","title":"ORGANA: A Robotic Assistant for Automated Chemistry Experimentation and Characterization","date":"2024-01-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ac-rad/organa","path":"nlp_service.py","file_url":"https://github.com/ac-rad/organa/blob/HEAD/nlp_service.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"de90ed129c9b8178","mcp_get_code":{"code_sha256":"de90ed129c9b8178"}},{"arxiv_id":"2312.04350","paper":"/paper/cladder-a-benchmark-to-assess-causal-1","title":"CLadder: Assessing Causal Reasoning in Language Models","date":"2023-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"causalNLP/cladder","path":"causalbenchmark/eval/generate_data_llama.py","file_url":"https://github.com/causalNLP/cladder/blob/HEAD/causalbenchmark/eval/generate_data_llama.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a5a90249b782a1b6","mcp_get_code":{"code_sha256":"a5a90249b782a1b6"}},{"arxiv_id":"2311.16079","paper":"/paper/meditron-70b-scaling-medical-pretraining-for","title":"MEDITRON-70B: Scaling Medical Pretraining for Large Language Models","date":"2023-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"epfllm/meditron","path":"evaluation/inference.py","file_url":"https://github.com/epfllm/meditron/blob/HEAD/evaluation/inference.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1ababc7e1e681f41","mcp_get_code":{"code_sha256":"1ababc7e1e681f41"}},{"arxiv_id":"2311.04901","paper":"/paper/genome-generative-neuro-symbolic-visual","title":"GENOME: GenerativE Neuro-symbOlic visual reasoning by growing and reusing ModulEs","date":"2023-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"umass-foundation-model/genome","path":"engine/prompt.py","file_url":"https://github.com/umass-foundation-model/genome/blob/HEAD/engine/prompt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8a2d342cd5ddce4c","mcp_get_code":{"code_sha256":"8a2d342cd5ddce4c"}},{"arxiv_id":"2310.11564","paper":"/paper/personalized-soups-personalized-large","title":"Personalized Soups: Personalized Large Language Model Alignment via Post-hoc Parameter Merging","date":"2023-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joeljang/rlphf","path":"gpt4_evaluate/alpaca_farm/data_preprocessor.py","file_url":"https://github.com/joeljang/rlphf/blob/HEAD/gpt4_evaluate/alpaca_farm/data_preprocessor.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a71a4bdc37586fe2","mcp_get_code":{"code_sha256":"a71a4bdc37586fe2"}},{"arxiv_id":"2310.03185","paper":"/paper/misusing-tools-in-large-language-models-with","title":"Misusing Tools in Large Language Models With Visual Adversarial Examples","date":"2023-10-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZihanWangKi/VLMToolMisuse","path":"llama_adapter/utils.py","file_url":"https://github.com/ZihanWangKi/VLMToolMisuse/blob/HEAD/llama_adapter/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7006a7479b6f348f","mcp_get_code":{"code_sha256":"7006a7479b6f348f"}},{"arxiv_id":"2309.09055","paper":"/paper/exploring-the-impact-of-low-rank-adaptation","title":"Exploring the impact of low-rank adaptation on the performance, efficiency, and regularization of RLHF","date":"2023-09-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"simengsun/alpaca_farm_lora","path":"alpaca_farm/src/alpaca_farm/data_preprocessor.py","file_url":"https://github.com/simengsun/alpaca_farm_lora/blob/HEAD/alpaca_farm/src/alpaca_farm/data_preprocessor.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5ef0d2db8e674cd5","mcp_get_code":{"code_sha256":"5ef0d2db8e674cd5"}},{"arxiv_id":"2309.00615","paper":"/paper/point-bind-point-llm-aligning-point-cloud","title":"Point-Bind & Point-LLM: Aligning Point Cloud with Multi-modality for 3D Understanding, Generation, and Instruction Following","date":"2023-09-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ziyuguo99/point-bind_point-llm","path":"Point-LLM/llama/utils.py","file_url":"https://github.com/ziyuguo99/point-bind_point-llm/blob/HEAD/Point-LLM/llama/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b58b8fd28d0ff117","mcp_get_code":{"code_sha256":"b58b8fd28d0ff117"}},{"arxiv_id":"2304.14979","paper":"/paper/mlcopilot-unleashing-the-power-of-large","title":"MLCopilot: Unleashing the Power of Large Language Models in Solving Machine Learning Tasks","date":"2023-04-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/CoML","path":"coml/configagent/suggest.py","file_url":"https://github.com/microsoft/CoML/blob/HEAD/coml/configagent/suggest.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8a9ff381ef3920b0","mcp_get_code":{"code_sha256":"8a9ff381ef3920b0"}},{"arxiv_id":"2303.16749","paper":"/paper/improving-code-generation-by-training-with","title":"Improving Code Generation by Training with Natural Language Feedback","date":"2023-03-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nyu-mll/ILF-for-code-generation","path":"create_finetuning_data_from_refinements.py","file_url":"https://github.com/nyu-mll/ILF-for-code-generation/blob/HEAD/create_finetuning_data_from_refinements.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d661e4c718b84a32","mcp_get_code":{"code_sha256":"d661e4c718b84a32"}}]}