{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/create-prompt","entry":"create_prompt","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":50,"n_papers_ran":35,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":63,"n_samples_ran":42,"n_samples_fingerprinted":8,"n_places":67,"n_places_pointer_only":40,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":15,"ran_fixture":1,"ran":26,"unverified":21},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.00728","paper":"/paper/arxiv-2609-00728","title":"SOVER: Formal Certification of Optimization Reformulations via LLM-Assisted SMT Verification","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"baranwa2/SOVER","path":"code/Variablemapping_extraction.py","file_url":"https://github.com/baranwa2/SOVER/blob/HEAD/code/Variablemapping_extraction.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"f78f6711b1fdcd4f","mcp_get_code":{"code_sha256":"f78f6711b1fdcd4f"}},{"arxiv_id":"2607.19181","paper":"/paper/arxiv-2607-19181","title":"Reasoning Before Translation: Enhancing Legal Machine Translation with Structured Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"aixiuxiuxiu/Legal-MT-SFT-RL","path":"convert_jsonl_to_prompts.py","file_url":"https://github.com/aixiuxiuxiu/Legal-MT-SFT-RL/blob/HEAD/convert_jsonl_to_prompts.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"07986447a705f50c","mcp_get_code":{"code_sha256":"07986447a705f50c"}},{"arxiv_id":"2606.07936","paper":"/paper/arxiv-2606-07936","title":"Illusions of the Gold Standard: A Large-scale Analysis of Human Evaluation Protocols for Long-form Text Generation","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"larchlab/Illusions-of-the-Gold-Standard","path":"llm_label_v5.py","file_url":"https://github.com/larchlab/Illusions-of-the-Gold-Standard/blob/HEAD/llm_label_v5.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6f867f364d45edd5","mcp_get_code":{"code_sha256":"6f867f364d45edd5"}},{"arxiv_id":"2604.15958","paper":"/paper/arxiv-2604-15958","title":"A Case Study on the Impact of Anonymization Along the RAG Pipeline","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"andreea-bodea/GuardRAG","path":"src/Presidio/Presidio_OpenAI.py","file_url":"https://github.com/andreea-bodea/GuardRAG/blob/HEAD/src/Presidio/Presidio_OpenAI.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3814d2a46da525ad","mcp_get_code":{"code_sha256":"3814d2a46da525ad"}},{"arxiv_id":"2604.15958","paper":"/paper/arxiv-2604-15958","title":"A Case Study on the Impact of Anonymization Along the RAG Pipeline","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"andreea-bodea/GuardRAG","path":"src/Presidio/Presidio_OpenAI.py","file_url":"https://github.com/andreea-bodea/GuardRAG/blob/HEAD/src/Presidio/Presidio_OpenAI.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"648834d74ca15c46","mcp_get_code":{"code_sha256":"648834d74ca15c46"}},{"arxiv_id":"2602.11909","paper":"/paper/arxiv-2602-11909","title":"Echo: Towards Advanced Audio Comprehension via Audio-Interleaved Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"wdqqdw/Echo","path":"inference/inference_multiturn.py","file_url":"https://github.com/wdqqdw/Echo/blob/HEAD/inference/inference_multiturn.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5def991b7b2b440d","mcp_get_code":{"code_sha256":"5def991b7b2b440d"}},{"arxiv_id":"2601.03042","paper":"/paper/arxiv-2601-03042","title":"BaseCal: Unsupervised Confidence Calibration via Base Model Signals","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Tan-Hexiang/BaseCal","path":"src/prompts.py","file_url":"https://github.com/Tan-Hexiang/BaseCal/blob/HEAD/src/prompts.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"db9655767f69c29f","mcp_get_code":{"code_sha256":"db9655767f69c29f"}},{"arxiv_id":"2601.03042","paper":"/paper/arxiv-2601-03042","title":"BaseCal: Unsupervised Confidence Calibration via Base Model Signals","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Tan-Hexiang/BaseCal","path":"src/prompts_wo_prefix_answer.py","file_url":"https://github.com/Tan-Hexiang/BaseCal/blob/HEAD/src/prompts_wo_prefix_answer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d16a3b512854596e","mcp_get_code":{"code_sha256":"d16a3b512854596e"}},{"arxiv_id":"2509.24088","paper":"/paper/arxiv-2509-24088","title":"CORRECT: Condensed Error Recognition via Knowledge Transfer in Multi-agent Systems","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"UIUC-MLSys/CORRECT","path":"src/error_schema_generator_cloud.py","file_url":"https://github.com/UIUC-MLSys/CORRECT/blob/HEAD/src/error_schema_generator_cloud.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"4e0a44a338e3f88e","mcp_get_code":{"code_sha256":"4e0a44a338e3f88e"}},{"arxiv_id":"2508.06492","paper":"/paper/arxiv-2508-06492","title":"Effective Training Data Synthesis for Improving MLLM Chart Understanding","date":null,"month_inferred_from_arxiv_id":"2025-08","title_source":"syntology","repo":"yuweiyang-anu/ECD","path":"data_generation_pipeline/figure_size_post_processing.py","file_url":"https://github.com/yuweiyang-anu/ECD/blob/HEAD/data_generation_pipeline/figure_size_post_processing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b740188ede016838","mcp_get_code":{"code_sha256":"b740188ede016838"}},{"arxiv_id":"2507.08267","paper":"/paper/a-practical-two-stage-recipe-for-mathematical","title":"A Practical Two-Stage Recipe for Mathematical LLMs: Maximizing Accuracy with SFT and Efficiency with Reinforcement Learning","date":"2025-07-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"analokmaus/kaggle-aimo2-fast-math-r1","path":"experiments/train_fast_nemotron_14b.py","file_url":"https://github.com/analokmaus/kaggle-aimo2-fast-math-r1/blob/HEAD/experiments/train_fast_nemotron_14b.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"522a7f189e5ea3b4","mcp_get_code":{"code_sha256":"522a7f189e5ea3b4"}},{"arxiv_id":"2507.08267","paper":"/paper/a-practical-two-stage-recipe-for-mathematical","title":"A Practical Two-Stage Recipe for Mathematical LLMs: Maximizing Accuracy with SFT and Efficiency with Reinforcement Learning","date":"2025-07-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"analokmaus/kaggle-aimo2-fast-math-r1","path":"experiments/train_first_stage.py","file_url":"https://github.com/analokmaus/kaggle-aimo2-fast-math-r1/blob/HEAD/experiments/train_first_stage.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e14b5e417be024bf","mcp_get_code":{"code_sha256":"e14b5e417be024bf"}},{"arxiv_id":"2506.06295","paper":"/paper/dllm-cache-accelerating-diffusion-large","title":"dLLM-Cache: Accelerating Diffusion Large Language Models with Adaptive Caching","date":"2025-05-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"maomaocun/dLLM-cache","path":"LLama_test_flops_script.py","file_url":"https://github.com/maomaocun/dLLM-cache/blob/HEAD/LLama_test_flops_script.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0976b0b29f8d1d22","mcp_get_code":{"code_sha256":"0976b0b29f8d1d22"}},{"arxiv_id":"2505.13995","paper":"/paper/social-sycophancy-a-broader-understanding-of","title":"Social Sycophancy: A Broader Understanding of LLM Sycophancy","date":"2025-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"myracheng/elephant","path":"sycophancy_scorers.py","file_url":"https://github.com/myracheng/elephant/blob/HEAD/sycophancy_scorers.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"CC0-1.0","inline_ok":true,"code_sha256_prefix":"01384b0681810072","mcp_get_code":{"code_sha256":"01384b0681810072"}},{"arxiv_id":"2504.05632","paper":"/paper/reasoning-towards-fairness-mitigating-bias-in","title":"Reasoning Towards Fairness: Mitigating Bias in Language Models through Reasoning-Guided Fine-Tuning","date":"2025-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Sanchit-404/Reasoing-Towards-Fairness","path":"scripts/extract_reasoning_traces.py","file_url":"https://github.com/Sanchit-404/Reasoing-Towards-Fairness/blob/HEAD/scripts/extract_reasoning_traces.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f2ba52311094f608","mcp_get_code":{"code_sha256":"f2ba52311094f608"}},{"arxiv_id":"2503.05728","paper":"/paper/political-neutrality-in-ai-is-impossible-but","title":"Political Neutrality in AI Is Impossible- But Here Is How to Approximate It","date":"2025-02-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jfisher52/approximation_political_neutrality","path":"src/eval_templates.py","file_url":"https://github.com/jfisher52/approximation_political_neutrality/blob/HEAD/src/eval_templates.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"29e163b5d3fd0c27","mcp_get_code":{"code_sha256":"29e163b5d3fd0c27"}},{"arxiv_id":"2410.10783","paper":"/paper/livexiv-a-multi-modal-live-benchmark-based-on","title":"LiveXiv -- A Multi-Modal Live Benchmark Based on Arxiv Papers Content","date":"2024-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nimrodshabtay/livexiv","path":"vqa_generation/filter_utils.py","file_url":"https://github.com/nimrodshabtay/livexiv/blob/HEAD/vqa_generation/filter_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"730b80d67018d2aa","mcp_get_code":{"code_sha256":"730b80d67018d2aa"}},{"arxiv_id":"2410.08827","paper":"/paper/do-unlearning-methods-remove-information-from","title":"Do Unlearning Methods Remove Information from Language Model Weights?","date":"2024-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aghyad-deeb/unlearning_evaluation","path":"unlearn_corpus.py","file_url":"https://github.com/aghyad-deeb/unlearning_evaluation/blob/HEAD/unlearn_corpus.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"814f2de8f4bd281e","mcp_get_code":{"code_sha256":"814f2de8f4bd281e"}},{"arxiv_id":"2410.08827","paper":"/paper/do-unlearning-methods-remove-information-from","title":"Do Unlearning Methods Remove Information from Language Model Weights?","date":"2024-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aghyad-deeb/unlearning_evaluation","path":"finetune_corpus.py","file_url":"https://github.com/aghyad-deeb/unlearning_evaluation/blob/HEAD/finetune_corpus.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"246c7ef8f90b3e7e","mcp_get_code":{"code_sha256":"246c7ef8f90b3e7e"}},{"arxiv_id":"2410.07054","paper":"/paper/mitigating-the-language-mismatch-and","title":"Mitigating the Language Mismatch and Repetition Issues in LLM-based Machine Translation via Model Editing","date":"2024-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"weichuanw/llm-based-mt-via-model-editing","path":"Model_Editing_Adaptation/fv_task_adaptation/src/utils/prompt_utils.py","file_url":"https://github.com/weichuanw/llm-based-mt-via-model-editing/blob/HEAD/Model_Editing_Adaptation/fv_task_adaptation/src/utils/prompt_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5daf29ec78f21d13","mcp_get_code":{"code_sha256":"5daf29ec78f21d13"}},{"arxiv_id":"2410.05695","paper":"/paper/unlocking-the-boundaries-of-thought-a","title":"Unlocking the Capabilities of Thought: A Reasoning Boundary Framework to Quantify and Optimize Chain-of-Thought","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LightChen233/reasoning-boundary","path":"request_text.py","file_url":"https://github.com/LightChen233/reasoning-boundary/blob/HEAD/request_text.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"46be0dbb921f8d85","mcp_get_code":{"code_sha256":"46be0dbb921f8d85"}},{"arxiv_id":"2410.05695","paper":"/paper/unlocking-the-boundaries-of-thought-a","title":"Unlocking the Capabilities of Thought: A Reasoning Boundary Framework to Quantify and Optimize Chain-of-Thought","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LightChen233/reasoning-boundary","path":"request_multimodal.py","file_url":"https://github.com/LightChen233/reasoning-boundary/blob/HEAD/request_multimodal.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"419ea57087ee3539","mcp_get_code":{"code_sha256":"419ea57087ee3539"}},{"arxiv_id":"2410.02584","paper":"/paper/towards-implicit-bias-detection-and","title":"Towards Implicit Bias Detection and Mitigation in Multi-Agent LLM Interactions","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MichiganNLP/MultiAgent_ImplicitBias","path":"Code/mistral_ft.py","file_url":"https://github.com/MichiganNLP/MultiAgent_ImplicitBias/blob/HEAD/Code/mistral_ft.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a2ba955201953f73","mcp_get_code":{"code_sha256":"a2ba955201953f73"}},{"arxiv_id":"2407.09014","paper":"/paper/compact-compressing-retrieved-documents","title":"CompAct: Compressing Retrieved Documents Actively for Question Answering","date":"2024-07-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dmis-lab/CompAct","path":"utils.py","file_url":"https://github.com/dmis-lab/CompAct/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"de1f2a7546b67b6a","mcp_get_code":{"code_sha256":"de1f2a7546b67b6a"}},{"arxiv_id":"2406.19999","paper":"/paper/the-sifo-benchmark-investigating-the","title":"The SIFo Benchmark: Investigating the Sequential Instruction Following Ability of Large Language Models","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shin-ee-chen/SIFo","path":"llm_inference/get_responses_from_vllm_models.py","file_url":"https://github.com/shin-ee-chen/SIFo/blob/HEAD/llm_inference/get_responses_from_vllm_models.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f2a96da09d0ad98f","mcp_get_code":{"code_sha256":"f2a96da09d0ad98f"}},{"arxiv_id":"2406.19073","paper":"/paper/ambrosia-a-benchmark-for-parsing-ambiguous","title":"AMBROSIA: A Benchmark for Parsing Ambiguous Questions into Database Queries","date":"2024-06-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"saparina/ambrosia","path":"src/db_generation/generate_databases.py","file_url":"https://github.com/saparina/ambrosia/blob/HEAD/src/db_generation/generate_databases.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ba28dc7e25b4ea7b","mcp_get_code":{"code_sha256":"ba28dc7e25b4ea7b"}},{"arxiv_id":"2406.14508","paper":"/paper/evidence-of-a-log-scaling-law-for-political","title":"Evidence of a log scaling law for political persuasion with large language models","date":"2024-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kobihackenburg/scaling-llm-persuasion","path":"main_study/code/01_instructionTune.py","file_url":"https://github.com/kobihackenburg/scaling-llm-persuasion/blob/HEAD/main_study/code/01_instructionTune.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"40357b3d64db3d0d","mcp_get_code":{"code_sha256":"40357b3d64db3d0d"}},{"arxiv_id":"2406.12950","paper":"/paper/moleculargpt-open-large-language-model-llm","title":"MolecularGPT: Open Large Language Model (LLM) for Few-Shot Molecular Property Prediction","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nyushcs/moleculargpt","path":"ICL_test_diversity.py","file_url":"https://github.com/nyushcs/moleculargpt/blob/HEAD/ICL_test_diversity.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5cbb7bb23f2c061d","mcp_get_code":{"code_sha256":"5cbb7bb23f2c061d"}},{"arxiv_id":"2406.12950","paper":"/paper/moleculargpt-open-large-language-model-llm","title":"MolecularGPT: Open Large Language Model (LLM) for Few-Shot Molecular Property Prediction","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nyushcs/moleculargpt","path":"ICL_test_reverse_cls.py","file_url":"https://github.com/nyushcs/moleculargpt/blob/HEAD/ICL_test_reverse_cls.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"437f17c491f15275","mcp_get_code":{"code_sha256":"437f17c491f15275"}},{"arxiv_id":"2406.12950","paper":"/paper/moleculargpt-open-large-language-model-llm","title":"MolecularGPT: Open Large Language Model (LLM) for Few-Shot Molecular Property Prediction","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nyushcs/moleculargpt","path":"ICL_test_reverse_reg.py","file_url":"https://github.com/nyushcs/moleculargpt/blob/HEAD/ICL_test_reverse_reg.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"40ec3f3c72f787b8","mcp_get_code":{"code_sha256":"40ec3f3c72f787b8"}},{"arxiv_id":"2406.12950","paper":"/paper/moleculargpt-open-large-language-model-llm","title":"MolecularGPT: Open Large Language Model (LLM) for Few-Shot Molecular Property Prediction","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nyushcs/moleculargpt","path":"ICL_test_sim_cls.py","file_url":"https://github.com/nyushcs/moleculargpt/blob/HEAD/ICL_test_sim_cls.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"504b9ab37dc7458f","mcp_get_code":{"code_sha256":"504b9ab37dc7458f"}},{"arxiv_id":"2406.12950","paper":"/paper/moleculargpt-open-large-language-model-llm","title":"MolecularGPT: Open Large Language Model (LLM) for Few-Shot Molecular Property Prediction","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nyushcs/moleculargpt","path":"ICL_test_sim_reg.py","file_url":"https://github.com/nyushcs/moleculargpt/blob/HEAD/ICL_test_sim_reg.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"32661c96034aa972","mcp_get_code":{"code_sha256":"32661c96034aa972"}},{"arxiv_id":"2406.09411","paper":"/paper/muirbench-a-comprehensive-benchmark-for","title":"MuirBench: A Comprehensive Benchmark for Robust Multi-image Understanding","date":"2024-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"muirbench/MuirBench","path":"eval/utils/preprocess.py","file_url":"https://github.com/muirbench/MuirBench/blob/HEAD/eval/utils/preprocess.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1cc5bed6d9e89200","mcp_get_code":{"code_sha256":"1cc5bed6d9e89200"}},{"arxiv_id":"2406.08587","paper":"/paper/cs-bench-a-comprehensive-benchmark-for-large","title":"CS-Bench: A Comprehensive Benchmark for Large Language Models towards Computer Science Mastery","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"csbench/csbench","path":"vllm-main/examples/csbench/gen_model_answer_en.py","file_url":"https://github.com/csbench/csbench/blob/HEAD/vllm-main/examples/csbench/gen_model_answer_en.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"05cde4d9afed5c53","mcp_get_code":{"code_sha256":"05cde4d9afed5c53"}},{"arxiv_id":"2406.07145","paper":"/paper/failures-are-fated-but-can-be-faded","title":"Failures Are Fated, But Can Be Faded: Characterizing and Mitigating Unwanted Behaviors in Large-Scale Vision and Language Models","date":"2024-06-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"somsagar07/FailureShiftRL","path":"Baselines/Generation/config.py","file_url":"https://github.com/somsagar07/FailureShiftRL/blob/HEAD/Baselines/Generation/config.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2af53582222a76e3","mcp_get_code":{"code_sha256":"2af53582222a76e3"}},{"arxiv_id":"2405.00722","paper":"/paper/llms-for-generating-and-evaluating","title":"LLMs for Generating and Evaluating Counterfactuals: A Comprehensive Study","date":"2024-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aix-group/llms-for-cfs","path":"src/gen_cf/generate_hatespeech.py","file_url":"https://github.com/aix-group/llms-for-cfs/blob/HEAD/src/gen_cf/generate_hatespeech.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4bb808bb894ec755","mcp_get_code":{"code_sha256":"4bb808bb894ec755"}},{"arxiv_id":"2405.00722","paper":"/paper/llms-for-generating-and-evaluating","title":"LLMs for Generating and Evaluating Counterfactuals: A Comprehensive Study","date":"2024-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aix-group/llms-for-cfs","path":"src/gen_cf/generate_imdb.py","file_url":"https://github.com/aix-group/llms-for-cfs/blob/HEAD/src/gen_cf/generate_imdb.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"91e0ef1413933bfa","mcp_get_code":{"code_sha256":"91e0ef1413933bfa"}},{"arxiv_id":"2405.00722","paper":"/paper/llms-for-generating-and-evaluating","title":"LLMs for Generating and Evaluating Counterfactuals: A Comprehensive Study","date":"2024-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aix-group/llms-for-cfs","path":"src/gen_cf/generate_snli.py","file_url":"https://github.com/aix-group/llms-for-cfs/blob/HEAD/src/gen_cf/generate_snli.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d94640c9148fbf50","mcp_get_code":{"code_sha256":"d94640c9148fbf50"}},{"arxiv_id":"2403.17508","paper":"/paper/correlation-of-frechet-audio-distance-with","title":"Correlation of Fréchet Audio Distance With Human Perception of Environmental Audio Is Embedding Dependant","date":null,"month_inferred_from_arxiv_id":"2024-03","title_source":"archive","repo":"dcase2024-task7-sound-scene-synthesis/fadtk","path":"example/prompts/gpt4_quality.py","file_url":"https://github.com/dcase2024-task7-sound-scene-synthesis/fadtk/blob/HEAD/example/prompts/gpt4_quality.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ca4b05e19c89061a","mcp_get_code":{"code_sha256":"ca4b05e19c89061a"}},{"arxiv_id":"2403.09040","paper":"/paper/ragged-towards-informed-design-of-retrieval","title":"RAGGED: Towards Informed Design of Retrieval Augmented Generation Systems","date":"2024-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"neulab/ragged","path":"reader/utils.py","file_url":"https://github.com/neulab/ragged/blob/HEAD/reader/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"289d7efd299b90d4","mcp_get_code":{"code_sha256":"289d7efd299b90d4"}},{"arxiv_id":"2403.02839","paper":"/paper/an-empirical-study-of-llm-as-a-judge-for-llm","title":"An Empirical Study of LLM-as-a-Judge for LLM Evaluation: Fine-tuned Judge Model is not a General Substitute for GPT-4","date":"2024-03-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"huihuichyan/unlimitedjudge","path":"src/build_prompt_judge.py","file_url":"https://github.com/huihuichyan/unlimitedjudge/blob/HEAD/src/build_prompt_judge.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5bb3a6f55ea7c8a7","mcp_get_code":{"code_sha256":"5bb3a6f55ea7c8a7"}},{"arxiv_id":"2402.14874","paper":"/paper/distillation-contrastive-decoding-improving","title":"Distillation Contrastive Decoding: Improving LLMs Reasoning with Contrastive Decoding and Distillation","date":"2024-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pphuc25/distil-cd","path":"src/dcd/prompts.py","file_url":"https://github.com/pphuc25/distil-cd/blob/HEAD/src/dcd/prompts.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8e4e00b93dabaf41","mcp_get_code":{"code_sha256":"8e4e00b93dabaf41"}},{"arxiv_id":"2402.12974","paper":"/paper/visual-style-prompting-with-swapping-self","title":"Visual Style Prompting with Swapping Self-Attention","date":"2024-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"naver-ai/Visual-Style-Prompting","path":"vsp_real_script.py","file_url":"https://github.com/naver-ai/Visual-Style-Prompting/blob/HEAD/vsp_real_script.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"56be14bdb80a0850","mcp_get_code":{"code_sha256":"56be14bdb80a0850"}},{"arxiv_id":"2402.10189","paper":"/paper/uncertainty-decomposition-and-quantification","title":"Uncertainty Quantification for In-Context Learning of Large Language Models","date":"2024-02-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lingchen0331/uq_icl","path":"sources/utils.py","file_url":"https://github.com/lingchen0331/uq_icl/blob/HEAD/sources/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"221aadea28e3360f","mcp_get_code":{"code_sha256":"221aadea28e3360f"}},{"arxiv_id":"2402.07616","paper":"/paper/anchor-based-large-language-models","title":"Anchor-based Large Language Models","date":"2024-02-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pangjh3/anllm","path":"inference.py","file_url":"https://github.com/pangjh3/anllm/blob/HEAD/inference.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4a62d89cb68ccb26","mcp_get_code":{"code_sha256":"4a62d89cb68ccb26"}},{"arxiv_id":"2402.07616","paper":"/paper/anchor-based-large-language-models","title":"Anchor-based Large Language Models","date":"2024-02-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pangjh3/anllm","path":"analysis/get_infer_attention.py","file_url":"https://github.com/pangjh3/anllm/blob/HEAD/analysis/get_infer_attention.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"65a7d55fbb2fbdf6","mcp_get_code":{"code_sha256":"65a7d55fbb2fbdf6"}},{"arxiv_id":"2402.07616","paper":"/paper/anchor-based-large-language-models","title":"Anchor-based Large Language Models","date":"2024-02-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pangjh3/anllm","path":"applications/translation/inference_allmnoacinput.py","file_url":"https://github.com/pangjh3/anllm/blob/HEAD/applications/translation/inference_allmnoacinput.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6ed5904743254ffa","mcp_get_code":{"code_sha256":"6ed5904743254ffa"}},{"arxiv_id":"2402.07270","paper":"/paper/open-ended-vqa-benchmarking-of-vision","title":"Open-ended VQA benchmarking of Vision-Language models by exploiting Classification datasets and their semantic hierarchy","date":"2024-02-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lmb-freiburg/ovqa","path":"ovqa/create_prompt.py","file_url":"https://github.com/lmb-freiburg/ovqa/blob/HEAD/ovqa/create_prompt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e1cf91cb1d73f4b7","mcp_get_code":{"code_sha256":"e1cf91cb1d73f4b7"}},{"arxiv_id":"2401.12873","paper":"/paper/improving-machine-translation-with-human","title":"Improving Machine Translation with Human Feedback: An Exploration of Quality Estimation as a Reward Model","date":"2024-01-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zwhe99/FeedbackMT","path":"src/inference_sft.py","file_url":"https://github.com/zwhe99/FeedbackMT/blob/HEAD/src/inference_sft.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4a62d89cb68ccb26","mcp_get_code":{"code_sha256":"4a62d89cb68ccb26"}},{"arxiv_id":"2401.08350","paper":"/paper/salute-the-classic-revisiting-challenges-of","title":"Salute the Classic: Revisiting Challenges of Machine Translation in the Age of Large Language Models","date":"2024-01-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pangjh3/llm4mt","path":"train/inference.py","file_url":"https://github.com/pangjh3/llm4mt/blob/HEAD/train/inference.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4a62d89cb68ccb26","mcp_get_code":{"code_sha256":"4a62d89cb68ccb26"}},{"arxiv_id":"2401.08350","paper":"/paper/salute-the-classic-revisiting-challenges-of","title":"Salute the Classic: Revisiting Challenges of Machine Translation in the Age of Large Language Models","date":"2024-01-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pangjh3/llm4mt","path":"train/attention_alignment_llama2.py","file_url":"https://github.com/pangjh3/llm4mt/blob/HEAD/train/attention_alignment_llama2.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"65a7d55fbb2fbdf6","mcp_get_code":{"code_sha256":"65a7d55fbb2fbdf6"}},{"arxiv_id":"2312.02852","paper":"/paper/expert-guided-bayesian-optimisation-for-human","title":"Expert-guided Bayesian Optimisation for Human-in-the-loop Experimental Design of Known Systems","date":"2023-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"trsav/hitl-bo","path":"bo/reccomender.py","file_url":"https://github.com/trsav/hitl-bo/blob/HEAD/bo/reccomender.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d16d3b2b4597fe7a","mcp_get_code":{"code_sha256":"d16d3b2b4597fe7a"}},{"arxiv_id":"2311.01616","paper":"/paper/adapting-frechet-audio-distance-for","title":"Adapting Frechet Audio Distance for Generative Music Evaluation","date":"2023-11-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"ca4b05e19c89061a","mcp_get_code":{"code_sha256":"ca4b05e19c89061a"}},{"arxiv_id":"2308.16458","paper":"/paper/biocoder-a-benchmark-for-bioinformatics-code","title":"BioCoder: A Benchmark for Bioinformatics Code Generation with Large Language Models","date":"2023-08-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gersteinlab/biocoder","path":"inference/final_batch_run.py","file_url":"https://github.com/gersteinlab/biocoder/blob/HEAD/inference/final_batch_run.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d4614f1cd57cdba4","mcp_get_code":{"code_sha256":"d4614f1cd57cdba4"}},{"arxiv_id":"2308.12674","paper":"/paper/improving-translation-faithfulness-of-large","title":"Improving Translation Faithfulness of Large Language Models via Augmenting Instructions","date":"2023-08-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pppa2019/swie_overmiss_llm4mt","path":"src/inference.py","file_url":"https://github.com/pppa2019/swie_overmiss_llm4mt/blob/HEAD/src/inference.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"258ba32f9e3eb1d9","mcp_get_code":{"code_sha256":"258ba32f9e3eb1d9"}},{"arxiv_id":"2308.12674","paper":"/paper/improving-translation-faithfulness-of-large","title":"Improving Translation Faithfulness of Large Language Models via Augmenting Instructions","date":"2023-08-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pppa2019/swie_overmiss_llm4mt","path":"utils/convert_alpaca_to_hf.py","file_url":"https://github.com/pppa2019/swie_overmiss_llm4mt/blob/HEAD/utils/convert_alpaca_to_hf.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"54458e263ddc9222","mcp_get_code":{"code_sha256":"54458e263ddc9222"}},{"arxiv_id":"2308.12097","paper":"/paper/instruction-position-matters-in-sequence","title":"Instruction Position Matters in Sequence Generation with Large Language Models","date":"2023-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"adaxry/post-instruction","path":"test/inference.py","file_url":"https://github.com/adaxry/post-instruction/blob/HEAD/test/inference.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1ccc9517aa991b74","mcp_get_code":{"code_sha256":"1ccc9517aa991b74"}},{"arxiv_id":"2307.07889","paper":"/paper/zero-shot-nlg-evaluation-through-pairware","title":"LLM Comparative Assessment: Zero-shot NLG Evaluation through Pairwise Comparisons using Large Language Models","date":"2023-07-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"adianliusie/comparative-assessment","path":"src/prompts/load_prompt.py","file_url":"https://github.com/adianliusie/comparative-assessment/blob/HEAD/src/prompts/load_prompt.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a0709570914a889f","mcp_get_code":{"code_sha256":"a0709570914a889f"}},{"arxiv_id":"2211.07517","paper":"/paper/are-hard-examples-also-harder-to-explain-a","title":"Are Hard Examples also Harder to Explain? A Study with Human and Model-Generated Explanations","date":"2022-11-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"swarnahub/explanationhardness","path":"generate_gpt3_explanations.py","file_url":"https://github.com/swarnahub/explanationhardness/blob/HEAD/generate_gpt3_explanations.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8b736157e0d6c782","mcp_get_code":{"code_sha256":"8b736157e0d6c782"}},{"arxiv_id":"2209.15517","paper":"/paper/medical-image-understanding-with-pretrained","title":"Medical Image Understanding with Pretrained Vision Language Models: A Comprehensive Study","date":"2022-09-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"membrai/miu-vl","path":"make_autopromptsv2.py","file_url":"https://github.com/membrai/miu-vl/blob/HEAD/make_autopromptsv2.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2b0f47de5fbaeb12","mcp_get_code":{"code_sha256":"2b0f47de5fbaeb12"}},{"arxiv_id":"2208.11057","paper":"/paper/prompting-as-probing-using-language-models","title":"Prompting as Probing: Using Language Models for Knowledge Base Construction","date":"2022-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hemile/iswc-challenge","path":"baseline.py","file_url":"https://github.com/hemile/iswc-challenge/blob/HEAD/baseline.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"72ee5ba094d0c0bb","mcp_get_code":{"code_sha256":"72ee5ba094d0c0bb"}},{"arxiv_id":"2208.11057","paper":"/paper/prompting-as-probing-using-language-models","title":"Prompting as Probing: Using Language Models for Knowledge Base Construction","date":"2022-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hemile/iswc-challenge","path":"gpt3_baseline.py","file_url":"https://github.com/hemile/iswc-challenge/blob/HEAD/gpt3_baseline.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6fd929182163484e","mcp_get_code":{"code_sha256":"6fd929182163484e"}},{"arxiv_id":"2025.findings-acl.51","paper":null,"title":"arXiv:2025.findings-acl.51","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ModeEric/ORBIT-Llama","path":"orbit/evaluation/astrobench_tests.py","file_url":"https://github.com/ModeEric/ORBIT-Llama/blob/HEAD/orbit/evaluation/astrobench_tests.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8d33d2648796fc52","mcp_get_code":{"code_sha256":"8d33d2648796fc52"}},{"arxiv_id":"2025.emnlp-industry.190","paper":null,"title":"arXiv:2025.emnlp-industry.190","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ismail31416/CAPSTONE","path":"capstone/reasoning/prompts.py","file_url":"https://github.com/ismail31416/CAPSTONE/blob/HEAD/capstone/reasoning/prompts.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b1ef4f8c3585b0c5","mcp_get_code":{"code_sha256":"b1ef4f8c3585b0c5"}},{"arxiv_id":"2024.findings-emnlp.365","paper":null,"title":"arXiv:2024.findings-emnlp.365","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"UKPLab/m2qa","path":"Experiments/LLM_evaluation/prompts.py","file_url":"https://github.com/UKPLab/m2qa/blob/HEAD/Experiments/LLM_evaluation/prompts.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1f8df1be24e8f46c","mcp_get_code":{"code_sha256":"1f8df1be24e8f46c"}},{"arxiv_id":"2024.findings-acl.255","paper":null,"title":"arXiv:2024.findings-acl.255","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ShubhamKumarNigam/PredEx","path":"Code/LLMs/Prompt_based_Inference/inference_LLAMA-2-7B_prediction.py","file_url":"https://github.com/ShubhamKumarNigam/PredEx/blob/HEAD/Code/LLMs/Prompt_based_Inference/inference_LLAMA-2-7B_prediction.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3b529eee45ab8e87","mcp_get_code":{"code_sha256":"3b529eee45ab8e87"}},{"arxiv_id":"2024.findings-acl.255","paper":null,"title":"arXiv:2024.findings-acl.255","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ShubhamKumarNigam/PredEx","path":"Code/LLMs/Prompt_based_Inference/inference_LLAMA-2-7B_prediction_explanation.py","file_url":"https://github.com/ShubhamKumarNigam/PredEx/blob/HEAD/Code/LLMs/Prompt_based_Inference/inference_LLAMA-2-7B_prediction_explanation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"81749bb8b058f914","mcp_get_code":{"code_sha256":"81749bb8b058f914"}}]}