{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/world-knowledge/papers/ran/1","list_of":"/task/world-knowledge","task":"World Knowledge","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":142,"counts":{"archive_papers_tagged":818,"with_a_code_link":358,"where_syntology_ran_a_sample":142,"not_listed_spam_title":0,"listed":818,"listed_where_code_ran":142,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":124,"every_run_a_failure_of_syntologys_instrument":18,"listed_with_a_run_with_no_instrument_failure":124,"listed_every_run_a_failure_of_syntologys_instrument":18,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/world-knowledge/papers/ran/1","prev":null,"next":"/task/world-knowledge/papers/ran/2","papers":[{"url":"/paper/dreamvla-a-vision-language-action-model-1","slug":"dreamvla-a-vision-language-action-model-1","title":"DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge","date":"2025-07-06","arxiv_id":"2507.04447","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dreamvla-a-vision-language-action-model-1#ran","syntology_url":"https://syntology.ai/paper/2507.04447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.04447"}},"official":{"repos":["Zhangwenyao1/DreamVLA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mirage-a-benchmark-for-multimodal-information","slug":"mirage-a-benchmark-for-multimodal-information","title":"MIRAGE: A Benchmark for Multimodal Information-Seeking and Reasoning in Agricultural Expert-Guided Conversations","date":"2025-06-25","arxiv_id":"2506.20100","repositories_listed":1,"syntology":{"n":18,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":18,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mirage-a-benchmark-for-multimodal-information#ran","syntology_url":"https://syntology.ai/paper/2506.20100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20100"}},"official":{"repos":["mirage-benchmark/mirage-benchmark"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/impliret-benchmarking-the-implicit-fact","slug":"impliret-benchmarking-the-implicit-fact","title":"ImpliRet: Benchmarking the Implicit Fact Retrieval Challenge","date":"2025-06-17","arxiv_id":"2506.14407","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/impliret-benchmarking-the-implicit-fact#ran","syntology_url":"https://syntology.ai/paper/2506.14407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.14407"}},"official":{"repos":["zeinabtaghavi/impliret"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autovla-a-vision-language-action-model-for","slug":"autovla-a-vision-language-action-model-for","title":"AutoVLA: A Vision-Language-Action Model for End-to-End Autonomous Driving with Adaptive Reasoning and Reinforcement Fine-Tuning","date":"2025-06-16","arxiv_id":"2506.13757","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autovla-a-vision-language-action-model-for#ran","syntology_url":"https://syntology.ai/paper/2506.13757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13757"}},"official":{"repos":["ucla-mobility/AutoVLA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contexttab-a-semantics-aware-tabular-in","slug":"contexttab-a-semantics-aware-tabular-in","title":"ConTextTab: A Semantics-Aware Tabular In-Context Learner","date":"2025-06-12","arxiv_id":"2506.10707","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/contexttab-a-semantics-aware-tabular-in#ran","syntology_url":"https://syntology.ai/paper/2506.10707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.10707"}},"official":{"repos":["sap-samples/contexttab"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/probing-the-geometry-of-truth-consistency-and","slug":"probing-the-geometry-of-truth-consistency-and","title":"Probing the Geometry of Truth: Consistency and Generalization of Truth Directions in LLMs Across Logical Transformations and Question Answering Tasks","date":"2025-06-01","arxiv_id":"2506.00823","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/probing-the-geometry-of-truth-consistency-and#ran","syntology_url":"https://syntology.ai/paper/2506.00823","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00823"}},"official":{"repos":["colored-dye/truthfulness_probe_generalization"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/o-2-searcher-a-searching-based-agent-model","slug":"o-2-searcher-a-searching-based-agent-model","title":"O$^2$-Searcher: A Searching-based Agent Model for Open-Domain Open-Ended Question Answering","date":"2025-05-22","arxiv_id":"2505.16582","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/o-2-searcher-a-searching-based-agent-model#ran","syntology_url":"https://syntology.ai/paper/2505.16582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16582"}},"official":{"repos":["acade-mate/o2-searcher"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-10940","slug":"2505-10940","title":"Who You Are Matters: Bridging Topics and Social Roles via LLM-Enhanced Logical Recommendation","date":"2025-05-16","arxiv_id":"2505.10940","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2505-10940#ran","syntology_url":"https://syntology.ai/paper/2505.10940","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10940"}},"official":null}},{"url":"/paper/doxing-via-the-lens-revealing-privacy-leakage","slug":"doxing-via-the-lens-revealing-privacy-leakage","title":"Doxing via the Lens: Revealing Location-related Privacy Leakage on Multi-modal Large Reasoning Models","date":"2025-04-27","arxiv_id":"2504.19373","repositories_listed":0,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/doxing-via-the-lens-revealing-privacy-leakage#ran","syntology_url":"https://syntology.ai/paper/2504.19373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.19373"}},"official":null}},{"url":"/paper/weathergen-a-unified-diverse-weather","slug":"weathergen-a-unified-diverse-weather","title":"WeatherGen: A Unified Diverse Weather Generator for LiDAR Point Clouds via Spider Mamba Diffusion","date":"2025-04-18","arxiv_id":"2504.13561","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/weathergen-a-unified-diverse-weather#ran","syntology_url":"https://syntology.ai/paper/2504.13561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13561"}},"official":{"repos":["wuyang98/weathergen"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-elicitation-of-latent-information","slug":"adaptive-elicitation-of-latent-information","title":"Adaptive Elicitation of Latent Information Using Natural Language","date":"2025-04-05","arxiv_id":"2504.04204","repositories_listed":0,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adaptive-elicitation-of-latent-information#ran","syntology_url":"https://syntology.ai/paper/2504.04204","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.04204"}},"official":null}},{"url":"/paper/llm-based-agent-simulation-for-maternal","slug":"llm-based-agent-simulation-for-maternal","title":"LLM-based Agent Simulation for Maternal Health Interventions: Uncertainty Estimation and Decision-focused Evaluation","date":"2025-03-25","arxiv_id":"2503.22719","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-based-agent-simulation-for-maternal#ran","syntology_url":"https://syntology.ai/paper/2503.22719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.22719"}},"official":{"repos":["sarahmart/llm-abs-armman-prediction"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wise-a-world-knowledge-informed-semantic","slug":"wise-a-world-knowledge-informed-semantic","title":"WISE: A World Knowledge-Informed Semantic Evaluation for Text-to-Image Generation","date":"2025-03-10","arxiv_id":"2503.07265","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wise-a-world-knowledge-informed-semantic#ran","syntology_url":"https://syntology.ai/paper/2503.07265","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07265"}},"official":{"repos":["pku-yuangroup/wise"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pointvla-injecting-the-3d-world-into-vision","slug":"pointvla-injecting-the-3d-world-into-vision","title":"PointVLA: Injecting the 3D World into Vision-Language-Action Models","date":"2025-03-10","arxiv_id":"2503.07511","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pointvla-injecting-the-3d-world-into-vision#ran","syntology_url":"https://syntology.ai/paper/2503.07511","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07511"}},"official":null}},{"url":"/paper/hermes-a-unified-self-driving-world-model-for","slug":"hermes-a-unified-self-driving-world-model-for","title":"HERMES: A Unified Self-Driving World Model for Simultaneous 3D Scene Understanding and Generation","date":"2025-01-24","arxiv_id":"2501.14729","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hermes-a-unified-self-driving-world-model-for#ran","syntology_url":"https://syntology.ai/paper/2501.14729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.14729"}},"official":{"repos":["lmd0311/hermes"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voxeval-benchmarking-the-knowledge","slug":"voxeval-benchmarking-the-knowledge","title":"VoxEval: Benchmarking the Knowledge Understanding Capabilities of End-to-End Spoken Language Models","date":"2025-01-09","arxiv_id":"2501.04962","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voxeval-benchmarking-the-knowledge#ran","syntology_url":"https://syntology.ai/paper/2501.04962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04962"}},"official":{"repos":["dreamtheater123/voxeval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/antileak-bench-preventing-data-contamination","slug":"antileak-bench-preventing-data-contamination","title":"AntiLeak-Bench: Preventing Data Contamination by Automatically Constructing Benchmarks with Updated Real-World Knowledge","date":"2024-12-18","arxiv_id":"2412.13670","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/antileak-bench-preventing-data-contamination#ran","syntology_url":"https://syntology.ai/paper/2412.13670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13670"}},"official":{"repos":["bobxwu/antileak-bench"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adapting-to-non-stationary-environments-multi","slug":"adapting-to-non-stationary-environments-multi","title":"Adapting to Non-Stationary Environments: Multi-Armed Bandit Enhanced Retrieval-Augmented Generation on Knowledge Graphs","date":"2024-12-10","arxiv_id":"2412.07618","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adapting-to-non-stationary-environments-multi#ran","syntology_url":"https://syntology.ai/paper/2412.07618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07618"}},"official":{"repos":["futureeeeee/dynamic-rag"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/i-don-t-know-explicit-modeling-of-uncertainty","slug":"i-don-t-know-explicit-modeling-of-uncertainty","title":"I Don't Know: Explicit Modeling of Uncertainty with an [IDK] Token","date":"2024-12-09","arxiv_id":"2412.06676","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/i-don-t-know-explicit-modeling-of-uncertainty#ran","syntology_url":"https://syntology.ai/paper/2412.06676","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06676"}},"official":{"repos":["roi-hpi/IDK-token-tuning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperseg-towards-universal-visual","slug":"hyperseg-towards-universal-visual","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","date":"2024-11-26","arxiv_id":"2411.17606","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/hyperseg-towards-universal-visual#ran","syntology_url":"https://syntology.ai/paper/2411.17606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17606"}},"official":{"repos":["congvvc/HyperSeg"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/chatsearch-a-dataset-and-a-generative","slug":"chatsearch-a-dataset-and-a-generative","title":"ChatSearch: a Dataset and a Generative Retrieval Model for General Conversational Image Retrieval","date":"2024-10-24","arxiv_id":"2410.18715","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chatsearch-a-dataset-and-a-generative#ran","syntology_url":"https://syntology.ai/paper/2410.18715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18715"}},"official":{"repos":["joez17/chatsearch"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/livexiv-a-multi-modal-live-benchmark-based-on","slug":"livexiv-a-multi-modal-live-benchmark-based-on","title":"LiveXiv -- A Multi-Modal Live Benchmark Based on Arxiv Papers Content","date":"2024-10-14","arxiv_id":"2410.10783","repositories_listed":1,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":16,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/livexiv-a-multi-modal-live-benchmark-based-on#ran","syntology_url":"https://syntology.ai/paper/2410.10783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10783"}},"official":{"repos":["nimrodshabtay/livexiv"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-embeddings-improve-test-time-adaptation","slug":"llm-embeddings-improve-test-time-adaptation","title":"LLM Embeddings Improve Test-time Adaptation to Tabular $Y|X$-Shifts","date":"2024-10-09","arxiv_id":"2410.07395","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llm-embeddings-improve-test-time-adaptation#ran","syntology_url":"https://syntology.ai/paper/2410.07395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07395"}},"official":{"repos":["namkoong-lab/llm-tabular-shifts"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/one-token-to-seg-them-all-language-instructed","slug":"one-token-to-seg-them-all-language-instructed","title":"One Token to Seg Them All: Language Instructed Reasoning Segmentation in Videos","date":"2024-09-29","arxiv_id":"2409.19603","repositories_listed":1,"syntology":{"n":16,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/one-token-to-seg-them-all-language-instructed#ran","syntology_url":"https://syntology.ai/paper/2409.19603","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19603"}},"official":{"repos":["showlab/videolisa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/curricullm-automatic-task-curricula-design","slug":"curricullm-automatic-task-curricula-design","title":"CurricuLLM: Automatic Task Curricula Design for Learning Complex Robot Skills using Large Language Models","date":"2024-09-27","arxiv_id":"2409.18382","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/curricullm-automatic-task-curricula-design#ran","syntology_url":"https://syntology.ai/paper/2409.18382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18382"}},"official":{"repos":["labicon/curricullm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hllm-enhancing-sequential-recommendations-via","slug":"hllm-enhancing-sequential-recommendations-via","title":"HLLM: Enhancing Sequential Recommendations via Hierarchical Large Language Models for Item and User Modeling","date":"2024-09-19","arxiv_id":"2409.12740","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hllm-enhancing-sequential-recommendations-via#ran","syntology_url":"https://syntology.ai/paper/2409.12740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12740"}},"official":{"repos":["bytedance/hllm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/synthetic-continued-pretraining","slug":"synthetic-continued-pretraining","title":"Synthetic continued pretraining","date":"2024-09-11","arxiv_id":"2409.07431","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/synthetic-continued-pretraining#ran","syntology_url":"https://syntology.ai/paper/2409.07431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.07431"}},"official":{"repos":["zitongyang/synthetic_continued_pretraining"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/can-ood-object-detectors-learn-from","slug":"can-ood-object-detectors-learn-from","title":"Can OOD Object Detectors Learn from Foundation Models?","date":"2024-09-08","arxiv_id":"2409.05162","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-ood-object-detectors-learn-from#ran","syntology_url":"https://syntology.ai/paper/2409.05162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.05162"}},"official":{"repos":["cvmi-lab/syncood"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/text2sql-is-not-enough-unifying-ai-and","slug":"text2sql-is-not-enough-unifying-ai-and","title":"Text2SQL is Not Enough: Unifying AI and Databases with TAG","date":"2024-08-27","arxiv_id":"2408.14717","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/text2sql-is-not-enough-unifying-ai-and#ran","syntology_url":"https://syntology.ai/paper/2408.14717","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.14717"}},"official":{"repos":["tag-research/tag-bench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/agentmove-predicting-human-mobility-anywhere","slug":"agentmove-predicting-human-mobility-anywhere","title":"AgentMove: Predicting Human Mobility Anywhere Using Large Language Model based Agentic Framework","date":"2024-08-26","arxiv_id":"2408.13986","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agentmove-predicting-human-mobility-anywhere#ran","syntology_url":"https://syntology.ai/paper/2408.13986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.13986"}},"official":{"repos":["tsinghua-fib-lab/agentmove"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/blade-benchmarking-language-model-agents-for","slug":"blade-benchmarking-language-model-agents-for","title":"BLADE: Benchmarking Language Model Agents for Data-Driven Science","date":"2024-08-19","arxiv_id":"2408.09667","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/blade-benchmarking-language-model-agents-for#ran","syntology_url":"https://syntology.ai/paper/2408.09667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09667"}},"official":{"repos":["behavioral-data/blade"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimus-1-hybrid-multimodal-memory-empowered","slug":"optimus-1-hybrid-multimodal-memory-empowered","title":"Optimus-1: Hybrid Multimodal Memory Empowered Agents Excel in Long-Horizon Tasks","date":"2024-08-07","arxiv_id":"2408.03615","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/optimus-1-hybrid-multimodal-memory-empowered#ran","syntology_url":"https://syntology.ai/paper/2408.03615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03615"}},"official":{"repos":["JiuTian-VL/Optimus-1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/seallms-3-open-foundation-and-chat","slug":"seallms-3-open-foundation-and-chat","title":"SeaLLMs 3: Open Foundation and Chat Multilingual Large Language Models for Southeast Asian Languages","date":"2024-07-29","arxiv_id":"2407.19672","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/seallms-3-open-foundation-and-chat#ran","syntology_url":"https://syntology.ai/paper/2407.19672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.19672"}},"official":{"repos":["DAMO-NLP-SG/SeaExam"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/from-internal-conflict-to-contextual","slug":"from-internal-conflict-to-contextual","title":"DYNAMICQA: Tracing Internal Knowledge Conflicts in Language Models","date":"2024-07-24","arxiv_id":"2407.17023","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/from-internal-conflict-to-contextual#ran","syntology_url":"https://syntology.ai/paper/2407.17023","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17023"}},"official":{"repos":["copenlu/dynamicqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-v-s-memorization-tracing","slug":"generalization-v-s-memorization-tracing","title":"Generalization v.s. Memorization: Tracing Language Models' Capabilities Back to Pretraining Data","date":"2024-07-20","arxiv_id":"2407.14985","repositories_listed":0,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/generalization-v-s-memorization-tracing#ran","syntology_url":"https://syntology.ai/paper/2407.14985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14985"}},"official":null}},{"url":"/paper/visa-reasoning-video-object-segmentation-via","slug":"visa-reasoning-video-object-segmentation-via","title":"VISA: Reasoning Video Object Segmentation via Large Language Models","date":"2024-07-16","arxiv_id":"2407.11325","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visa-reasoning-video-object-segmentation-via#ran","syntology_url":"https://syntology.ai/paper/2407.11325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11325"}},"official":{"repos":["cilinyan/VISA","cilinyan/revos-api"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flooding-spread-of-manipulated-knowledge-in","slug":"flooding-spread-of-manipulated-knowledge-in","title":"Flooding Spread of Manipulated Knowledge in LLM-Based Multi-Agent Communities","date":"2024-07-10","arxiv_id":"2407.07791","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/flooding-spread-of-manipulated-knowledge-in#ran","syntology_url":"https://syntology.ai/paper/2407.07791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07791"}},"official":{"repos":["Jometeorie/KnowledgeSpread"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-encode-collaborative-signals","slug":"language-models-encode-collaborative-signals","title":"Language Representations Can be What Recommenders Need: Findings and Potentials","date":"2024-07-07","arxiv_id":"2407.05441","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-encode-collaborative-signals#ran","syntology_url":"https://syntology.ai/paper/2407.05441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05441"}},"official":{"repos":["lehengthu/alpharec"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-synthetic-data-creation-with","slug":"scaling-synthetic-data-creation-with","title":"Scaling Synthetic Data Creation with 1,000,000,000 Personas","date":"2024-06-28","arxiv_id":"2406.20094","repositories_listed":4,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-synthetic-data-creation-with#ran","syntology_url":"https://syntology.ai/paper/2406.20094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20094"}},"official":{"repos":["tencent-ailab/persona-hub"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/llara-supercharging-robot-learning-data-for","slug":"llara-supercharging-robot-learning-data-for","title":"LLaRA: Supercharging Robot Learning Data for Vision-Language Policy","date":"2024-06-28","arxiv_id":"2406.20095","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llara-supercharging-robot-learning-data-for#ran","syntology_url":"https://syntology.ai/paper/2406.20095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20095"}},"official":{"repos":["lostxine/llara"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-multi-image-understanding-in","slug":"benchmarking-multi-image-understanding-in","title":"Benchmarking Multi-Image Understanding in Vision and Language Models: Perception, Knowledge, Reasoning, and Multi-Hop Reasoning","date":"2024-06-18","arxiv_id":"2406.12742","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarking-multi-image-understanding-in#ran","syntology_url":"https://syntology.ai/paper/2406.12742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12742"}},"official":{"repos":["dtennant/mirb_eval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rwku-benchmarking-real-world-knowledge","slug":"rwku-benchmarking-real-world-knowledge","title":"RWKU: Benchmarking Real-World Knowledge Unlearning for Large Language Models","date":"2024-06-16","arxiv_id":"2406.10890","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rwku-benchmarking-real-world-knowledge#ran","syntology_url":"https://syntology.ai/paper/2406.10890","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10890"}},"official":{"repos":["jinzhuoran/rwku"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-softmax-direct-preference-optimization-for","slug":"on-softmax-direct-preference-optimization-for","title":"On Softmax Direct Preference Optimization for Recommendation","date":"2024-06-13","arxiv_id":"2406.09215","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-softmax-direct-preference-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2406.09215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09215"}},"official":{"repos":["chenyuxin1999/s-dpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-synthetic-dataset-for-personal-attribute","slug":"a-synthetic-dataset-for-personal-attribute","title":"A Synthetic Dataset for Personal Attribute Inference","date":"2024-06-11","arxiv_id":"2406.07217","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-synthetic-dataset-for-personal-attribute#ran","syntology_url":"https://syntology.ai/paper/2406.07217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07217"}},"official":{"repos":["eth-sri/synthpai"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/corda-context-oriented-decomposition","slug":"corda-context-oriented-decomposition","title":"CorDA: Context-Oriented Decomposition Adaptation of Large Language Models for Task-Aware Parameter-Efficient Fine-tuning","date":"2024-06-07","arxiv_id":"2406.05223","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/corda-context-oriented-decomposition#ran","syntology_url":"https://syntology.ai/paper/2406.05223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05223"}},"official":{"repos":["iboing/corda"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kg-fit-knowledge-graph-fine-tuning-upon-open","slug":"kg-fit-knowledge-graph-fine-tuning-upon-open","title":"KG-FIT: Knowledge Graph Fine-Tuning Upon Open-World Knowledge","date":"2024-05-26","arxiv_id":"2405.16412","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/kg-fit-knowledge-graph-fine-tuning-upon-open#ran","syntology_url":"https://syntology.ai/paper/2405.16412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16412"}},"official":{"repos":["pat-jj/KG-FIT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-scale-knowledge-washing","slug":"large-scale-knowledge-washing","title":"Large Scale Knowledge Washing","date":"2024-05-26","arxiv_id":"2405.16720","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-scale-knowledge-washing#ran","syntology_url":"https://syntology.ai/paper/2405.16720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16720"}},"official":{"repos":["wangyu-ustc/largescalewashing"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unke-unstructured-knowledge-editing-in-large","slug":"unke-unstructured-knowledge-editing-in-large","title":"Everything is Editable: Extend Knowledge Editing to Unstructured Data in Large Language Models","date":"2024-05-24","arxiv_id":"2405.15349","repositories_listed":1,"syntology":{"n":12,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":12,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/unke-unstructured-knowledge-editing-in-large#ran","syntology_url":"https://syntology.ai/paper/2405.15349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15349"}},"official":{"repos":["TrustedLLM/UnKE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/meteor-mamba-based-traversal-of-rationale-for","slug":"meteor-mamba-based-traversal-of-rationale-for","title":"Meteor: Mamba-based Traversal of Rationale for Large Language and Vision Models","date":"2024-05-24","arxiv_id":"2405.15574","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/meteor-mamba-based-traversal-of-rationale-for#ran","syntology_url":"https://syntology.ai/paper/2405.15574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15574"}},"official":{"repos":["byungkwanlee/meteor"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-planning-with-world-knowledge-model","slug":"agent-planning-with-world-knowledge-model","title":"Agent Planning with World Knowledge Model","date":"2024-05-23","arxiv_id":"2405.14205","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/agent-planning-with-world-knowledge-model#ran","syntology_url":"https://syntology.ai/paper/2405.14205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14205"}},"official":{"repos":["zjunlp/wkm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/elements-of-world-knowledge-ewok-a-cognition","slug":"elements-of-world-knowledge-ewok-a-cognition","title":"Elements of World Knowledge (EWOK): A cognition-inspired framework for evaluating basic world knowledge in language models","date":"2024-05-15","arxiv_id":"2405.09605","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/elements-of-world-knowledge-ewok-a-cognition#ran","syntology_url":"https://syntology.ai/paper/2405.09605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.09605"}},"official":{"repos":["ewok-core/ewok-paper","ewok-core/ewok"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learnable-tokenizer-for-llm-based-generative","slug":"learnable-tokenizer-for-llm-based-generative","title":"Learnable Item Tokenization for Generative Recommendation","date":"2024-05-12","arxiv_id":"2405.07314","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learnable-tokenizer-for-llm-based-generative#ran","syntology_url":"https://syntology.ai/paper/2405.07314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07314"}},"official":{"repos":["honghuibao2000/letter"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pac-bayesian-generalization-bounds-for-2","slug":"pac-bayesian-generalization-bounds-for-2","title":"PAC-Bayesian Generalization Bounds for Knowledge Graph Representation Learning","date":"2024-05-10","arxiv_id":"2405.06418","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pac-bayesian-generalization-bounds-for-2#ran","syntology_url":"https://syntology.ai/paper/2405.06418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06418"}},"official":{"repos":["bdi-lab/reed"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-care-assessing-the-healthcare","slug":"cross-care-assessing-the-healthcare","title":"Cross-Care: Assessing the Healthcare Implications of Pre-training Data on Language Model Bias","date":"2024-05-09","arxiv_id":"2405.05506","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cross-care-assessing-the-healthcare#ran","syntology_url":"https://syntology.ai/paper/2405.05506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05506"}},"official":{"repos":["shan23chen/cross-care"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-adaptation-from-large-language","slug":"knowledge-adaptation-from-large-language","title":"LEARN: Knowledge Adaptation from Large Language Model to Recommendation for Practical Industrial Application","date":"2024-05-07","arxiv_id":"2405.03988","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/knowledge-adaptation-from-large-language#ran","syntology_url":"https://syntology.ai/paper/2405.03988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.03988"}},"official":{"repos":["adxcreative/LEARN"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-user-centric-benchmark-for-evaluating-large","slug":"a-user-centric-benchmark-for-evaluating-large","title":"A User-Centric Multi-Intent Benchmark for Evaluating Large Language Models","date":"2024-04-22","arxiv_id":"2404.13940","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-user-centric-benchmark-for-evaluating-large#ran","syntology_url":"https://syntology.ai/paper/2404.13940","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13940"}},"official":{"repos":["alice1998/urs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/boter-bootstrapping-knowledge-selection-and","slug":"boter-bootstrapping-knowledge-selection-and","title":"Self-Bootstrapped Visual-Language Model for Knowledge Selection and Question Answering","date":"2024-04-22","arxiv_id":"2404.13947","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/boter-bootstrapping-knowledge-selection-and#ran","syntology_url":"https://syntology.ai/paper/2404.13947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13947"}},"official":{"repos":["haodongze/self-ksel-qans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/more-room-for-language-investigating-the","slug":"more-room-for-language-investigating-the","title":"More Room for Language: Investigating the Effect of Retrieval on Language Models","date":"2024-04-16","arxiv_id":"2404.10939","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/more-room-for-language-investigating-the#ran","syntology_url":"https://syntology.ai/paper/2404.10939","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10939"}},"official":{"repos":["ltgoslo/more-room-for-language"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-potential-of-large-foundation","slug":"exploring-the-potential-of-large-foundation","title":"Exploring the Potential of Large Foundation Models for Open-Vocabulary HOI Detection","date":"2024-04-09","arxiv_id":"2404.06194","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/exploring-the-potential-of-large-foundation#ran","syntology_url":"https://syntology.ai/paper/2404.06194","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.06194"}},"official":{"repos":["ltttpku/cmd-se-release"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/elephants-never-forget-memorization-and","slug":"elephants-never-forget-memorization-and","title":"Elephants Never Forget: Memorization and Learning of Tabular Data in Large Language Models","date":"2024-04-09","arxiv_id":"2404.06209","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/elephants-never-forget-memorization-and#ran","syntology_url":"https://syntology.ai/paper/2404.06209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.06209"}},"official":{"repos":["interpretml/llm-tabular-memorization-checker"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bear-a-unified-framework-for-evaluating","slug":"bear-a-unified-framework-for-evaluating","title":"BEAR: A Unified Framework for Evaluating Relational Knowledge in Causal and Masked Language Models","date":"2024-04-05","arxiv_id":"2404.04113","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bear-a-unified-framework-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2404.04113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04113"}},"official":{"repos":["lm-pub-quiz/lm-pub-quiz"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/surface-reconstruction-from-gaussian","slug":"surface-reconstruction-from-gaussian","title":"GS2Mesh: Surface Reconstruction from Gaussian Splatting via Novel Stereo Views","date":"2024-04-02","arxiv_id":"2404.01810","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/surface-reconstruction-from-gaussian#ran","syntology_url":"https://syntology.ai/paper/2404.01810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01810"}},"official":{"repos":["yanivw12/gs2mesh"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/are-we-on-the-right-way-for-evaluating-large","slug":"are-we-on-the-right-way-for-evaluating-large","title":"Are We on the Right Way for Evaluating Large Vision-Language Models?","date":"2024-03-29","arxiv_id":"2403.20330","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/are-we-on-the-right-way-for-evaluating-large#ran","syntology_url":"https://syntology.ai/paper/2403.20330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20330"}},"official":{"repos":["MMStar-Benchmark/MMStar"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mechanisms-of-non-factual-hallucinations-in","slug":"mechanisms-of-non-factual-hallucinations-in","title":"Mechanistic Understanding and Mitigation of Language Model Non-Factual Hallucinations","date":"2024-03-27","arxiv_id":"2403.18167","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mechanisms-of-non-factual-hallucinations-in#ran","syntology_url":"https://syntology.ai/paper/2403.18167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18167"}},"official":{"repos":["jadeleiyu/lm_hallucination_mechanisms"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-embeddings-the-promise-of-visual-table","slug":"beyond-embeddings-the-promise-of-visual-table","title":"Beyond Embeddings: The Promise of Visual Table in Visual Reasoning","date":"2024-03-27","arxiv_id":"2403.18252","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 2 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-embeddings-the-promise-of-visual-table#ran","syntology_url":"https://syntology.ai/paper/2403.18252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18252"}},"official":{"repos":["lavi-lab/visual-table"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sequential-recommendation-with-latent","slug":"sequential-recommendation-with-latent","title":"Sequential Recommendation with Latent Relations based on Large Language Model","date":"2024-03-27","arxiv_id":"2403.18348","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sequential-recommendation-with-latent#ran","syntology_url":"https://syntology.ai/paper/2403.18348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18348"}},"official":{"repos":["ysh-1998/lrd"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-long-videos-in-one-multimodal","slug":"understanding-long-videos-in-one-multimodal","title":"Understanding Long Videos with Multimodal Language Models","date":"2024-03-25","arxiv_id":"2403.16998","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-long-videos-in-one-multimodal#ran","syntology_url":"https://syntology.ai/paper/2403.16998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16998"}},"official":{"repos":["kahnchana/mvu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/embodied-llm-agents-learn-to-cooperate-in","slug":"embodied-llm-agents-learn-to-cooperate-in","title":"Embodied LLM Agents Learn to Cooperate in Organized Teams","date":"2024-03-19","arxiv_id":"2403.12482","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/embodied-llm-agents-learn-to-cooperate-in#ran","syntology_url":"https://syntology.ai/paper/2403.12482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12482"}},"official":{"repos":["tobeatraceur/Organized-LLM-Agents"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-llms-tuning-methods-work-in-medical","slug":"can-llms-tuning-methods-work-in-medical","title":"Can LLMs' Tuning Methods Work in Medical Multimodal Domain?","date":"2024-03-11","arxiv_id":"2403.06407","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/can-llms-tuning-methods-work-in-medical#ran","syntology_url":"https://syntology.ai/paper/2403.06407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.06407"}},"official":{"repos":["timmy-chan/mile"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/towards-efficient-and-effective-unlearning-of","slug":"towards-efficient-and-effective-unlearning-of","title":"Towards Efficient and Effective Unlearning of Large Language Models for Recommendation","date":"2024-03-06","arxiv_id":"2403.03536","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/towards-efficient-and-effective-unlearning-of#ran","syntology_url":"https://syntology.ai/paper/2403.03536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03536"}},"official":{"repos":["justarter/e2urec"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/meacap-memory-augmented-zero-shot-image","slug":"meacap-memory-augmented-zero-shot-image","title":"MeaCap: Memory-Augmented Zero-shot Image Captioning","date":"2024-03-06","arxiv_id":"2403.03715","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":1,"n_ran_checked":4,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/meacap-memory-augmented-zero-shot-image#ran","syntology_url":"https://syntology.ai/paper/2403.03715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03715"}},"official":{"repos":["joeyz0z/meacap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-or-self-aligning-rethinking","slug":"learning-or-self-aligning-rethinking","title":"Learning or Self-aligning? Rethinking Instruction Fine-tuning","date":"2024-02-28","arxiv_id":"2402.18243","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":17,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-or-self-aligning-rethinking#ran","syntology_url":"https://syntology.ai/paper/2402.18243","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18243"}},"official":{"repos":["renmengjie7/self-aligning"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/finer-investigating-and-enhancing-fine","slug":"finer-investigating-and-enhancing-fine","title":"Finer: Investigating and Enhancing Fine-Grained Visual Concept Recognition in Large Vision Language Models","date":"2024-02-26","arxiv_id":"2402.16315","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finer-investigating-and-enhancing-fine#ran","syntology_url":"https://syntology.ai/paper/2402.16315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16315"}},"official":null}},{"url":"/paper/flame-self-supervised-low-resource-taxonomy","slug":"flame-self-supervised-low-resource-taxonomy","title":"FLAME: Self-Supervised Low-Resource Taxonomy Expansion using Large Language Models","date":"2024-02-21","arxiv_id":"2402.13623","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/flame-self-supervised-low-resource-taxonomy#ran","syntology_url":"https://syntology.ai/paper/2402.13623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13623"}},"official":{"repos":["sahilmishra0012/flame"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gerea-question-aware-prompt-captions-for","slug":"gerea-question-aware-prompt-captions-for","title":"GeReA: Question-Aware Prompt Captions for Knowledge-based Visual Question Answering","date":"2024-02-04","arxiv_id":"2402.02503","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/gerea-question-aware-prompt-captions-for#ran","syntology_url":"https://syntology.ai/paper/2402.02503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02503"}},"official":{"repos":["upper9527/gerea"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/good-at-captioning-bad-at-counting","slug":"good-at-captioning-bad-at-counting","title":"Good at captioning, bad at counting: Benchmarking GPT-4V on Earth observation data","date":"2024-01-31","arxiv_id":"2401.17600","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/good-at-captioning-bad-at-counting#ran","syntology_url":"https://syntology.ai/paper/2401.17600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17600"}},"official":{"repos":["Earth-Intelligence-Lab/vleo-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/can-ai-assistants-know-what-they-don-t-know","slug":"can-ai-assistants-know-what-they-don-t-know","title":"Can AI Assistants Know What They Don't Know?","date":"2024-01-24","arxiv_id":"2401.13275","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-ai-assistants-know-what-they-don-t-know#ran","syntology_url":"https://syntology.ai/paper/2401.13275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13275"}},"official":{"repos":["openmoss/say-i-dont-know"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-hallucinations-of-large-language","slug":"mitigating-hallucinations-of-large-language","title":"Knowledge Verification to Nip Hallucination in the Bud","date":"2024-01-19","arxiv_id":"2401.10768","repositories_listed":1,"syntology":{"n":22,"n_ran":17,"n_constructed":0,"n_ran_checked":16,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":5,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mitigating-hallucinations-of-large-language#ran","syntology_url":"https://syntology.ai/paper/2401.10768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10768"}},"official":{"repos":["fanqiwan/kca"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/pokemqa-programmable-knowledge-editing-for","slug":"pokemqa-programmable-knowledge-editing-for","title":"PokeMQA: Programmable knowledge editing for Multi-hop Question Answering","date":"2023-12-23","arxiv_id":"2312.15194","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pokemqa-programmable-knowledge-editing-for#ran","syntology_url":"https://syntology.ai/paper/2312.15194","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15194"}},"official":{"repos":["hengrui-gu/pokemqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/textit-v-guided-visual-search-as-a-core","slug":"textit-v-guided-visual-search-as-a-core","title":"V*: Guided Visual Search as a Core Mechanism in Multimodal LLMs","date":"2023-12-21","arxiv_id":"2312.14135","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textit-v-guided-visual-search-as-a-core#ran","syntology_url":"https://syntology.ai/paper/2312.14135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14135"}},"official":{"repos":["penghao-wu/vstar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-art-of-balancing-revolutionizing-mixture","slug":"the-art-of-balancing-revolutionizing-mixture","title":"LoRAMoE: Alleviate World Knowledge Forgetting in Large Language Models via MoE-Style Plugin","date":"2023-12-15","arxiv_id":"2312.09979","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/the-art-of-balancing-revolutionizing-mixture#ran","syntology_url":"https://syntology.ai/paper/2312.09979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09979"}},"official":{"repos":["ablustrund/loramoe"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-surface-probing-llama-across-scales","slug":"beyond-surface-probing-llama-across-scales","title":"Is Bigger and Deeper Always Better? Probing LLaMA Across Scales and Layers","date":"2023-12-07","arxiv_id":"2312.04333","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-surface-probing-llama-across-scales#ran","syntology_url":"https://syntology.ai/paper/2312.04333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04333"}},"official":{"repos":["nuochenpku/llama_analysis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llara-aligning-large-language-models-with","slug":"llara-aligning-large-language-models-with","title":"LLaRA: Large Language-Recommendation Assistant","date":"2023-12-05","arxiv_id":"2312.02445","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llara-aligning-large-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2312.02445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02445"}},"official":{"repos":["ljy0ustc/llara"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recexplainer-aligning-large-language-models","slug":"recexplainer-aligning-large-language-models","title":"RecExplainer: Aligning Large Language Models for Explaining Recommendation Models","date":"2023-11-18","arxiv_id":"2311.10947","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recexplainer-aligning-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2311.10947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10947"}},"official":{"repos":["microsoft/recai"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/carpe-diem-on-the-evaluation-of-world","slug":"carpe-diem-on-the-evaluation-of-world","title":"Carpe Diem: On the Evaluation of World Knowledge in Lifelong Language Models","date":"2023-11-14","arxiv_id":"2311.08106","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/carpe-diem-on-the-evaluation-of-world#ran","syntology_url":"https://syntology.ai/paper/2311.08106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08106"}},"official":{"repos":["kimyuji/evolvingqa_benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-open-ended-visual-recognition-with","slug":"towards-open-ended-visual-recognition-with","title":"Towards Open-Ended Visual Recognition with Large Language Model","date":"2023-11-14","arxiv_id":"2311.08400","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-open-ended-visual-recognition-with#ran","syntology_url":"https://syntology.ai/paper/2311.08400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08400"}},"official":{"repos":["bytedance/omniscient-model"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/do-llms-implicitly-exhibit-user","slug":"do-llms-implicitly-exhibit-user","title":"A Study of Implicit Ranking Unfairness in Large Language Models","date":"2023-11-13","arxiv_id":"2311.07054","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/do-llms-implicitly-exhibit-user#ran","syntology_url":"https://syntology.ai/paper/2311.07054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07054"}},"official":{"repos":["xuchen0427/implicit_rank_unfairness"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/language-guided-visual-question-answering","slug":"language-guided-visual-question-answering","title":"Language Guided Visual Question Answering: Elevate Your Multimodal Language Model Using Knowledge-Enriched Prompts","date":"2023-10-31","arxiv_id":"2310.20159","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-guided-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2310.20159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20159"}},"official":{"repos":["declare-lab/lg-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/capsfusion-rethinking-image-text-data-at","slug":"capsfusion-rethinking-image-text-data-at","title":"CapsFusion: Rethinking Image-Text Data at Scale","date":"2023-10-31","arxiv_id":"2310.20550","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/capsfusion-rethinking-image-text-data-at#ran","syntology_url":"https://syntology.ai/paper/2310.20550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20550"}},"official":{"repos":["baaivision/capsfusion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-factuality-a-comprehensive-evaluation","slug":"beyond-factuality-a-comprehensive-evaluation","title":"Beyond Factuality: A Comprehensive Evaluation of Large Language Models as Knowledge Generators","date":"2023-10-11","arxiv_id":"2310.07289","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":14,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/beyond-factuality-a-comprehensive-evaluation#ran","syntology_url":"https://syntology.ai/paper/2310.07289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07289"}},"official":{"repos":["chanliang/conner"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mistral-7b","slug":"mistral-7b","title":"Mistral 7B","date":"2023-10-10","arxiv_id":"2310.06825","repositories_listed":6,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mistral-7b#ran","syntology_url":"https://syntology.ai/paper/2310.06825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06825"}},"official":{"repos":["mistralai/mistral-src"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/large-language-models-only-pass-primary","slug":"large-language-models-only-pass-primary","title":"Large Language Models Only Pass Primary School Exams in Indonesia: A Comprehensive Test on IndoMMLU","date":"2023-10-07","arxiv_id":"2310.04928","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/large-language-models-only-pass-primary#ran","syntology_url":"https://syntology.ai/paper/2310.04928","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04928"}},"official":{"repos":["fajri91/indommlu"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/corex-pushing-the-boundaries-of-complex","slug":"corex-pushing-the-boundaries-of-complex","title":"Corex: Pushing the Boundaries of Complex Reasoning through Multi-Model Collaboration","date":"2023-09-30","arxiv_id":"2310.00280","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/corex-pushing-the-boundaries-of-complex#ran","syntology_url":"https://syntology.ai/paper/2310.00280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00280"}},"official":{"repos":["qiushisun/corex"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/retrieve-rewrite-answer-a-kg-to-text-enhanced","slug":"retrieve-rewrite-answer-a-kg-to-text-enhanced","title":"Retrieve-Rewrite-Answer: A KG-to-Text Enhanced LLMs Framework for Knowledge Graph Question Answering","date":"2023-09-20","arxiv_id":"2309.11206","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/retrieve-rewrite-answer-a-kg-to-text-enhanced#ran","syntology_url":"https://syntology.ai/paper/2309.11206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11206"}},"official":{"repos":["wuyike2000/retrieve-rewrite-answer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/grasp-anything-large-scale-grasp-dataset-from","slug":"grasp-anything-large-scale-grasp-dataset-from","title":"Grasp-Anything: Large-scale Grasp Dataset from Foundation Models","date":"2023-09-18","arxiv_id":"2309.09818","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/grasp-anything-large-scale-grasp-dataset-from#ran","syntology_url":"https://syntology.ai/paper/2309.09818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.09818"}},"official":{"repos":["andvg3/Grasp-Anything"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2309-05936","slug":"2309-05936","title":"Do PLMs Know and Understand Ontological Knowledge?","date":"2023-09-12","arxiv_id":"2309.05936","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/2309-05936#ran","syntology_url":"https://syntology.ai/paper/2309.05936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05936"}},"official":{"repos":["vickywu1022/ontoprobe-plms"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/expel-llm-agents-are-experiential-learners","slug":"expel-llm-agents-are-experiential-learners","title":"ExpeL: LLM Agents Are Experiential Learners","date":"2023-08-20","arxiv_id":"2308.10144","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/expel-llm-agents-are-experiential-learners#ran","syntology_url":"https://syntology.ai/paper/2308.10144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10144"}},"official":{"repos":["LeapLabTHU/ExpeL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/head-to-tail-how-knowledgeable-are-large","slug":"head-to-tail-how-knowledgeable-are-large","title":"Head-to-Tail: How Knowledgeable are Large Language Models (LLMs)? A.K.A. Will LLMs Replace Knowledge Graphs?","date":"2023-08-20","arxiv_id":"2308.10168","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/head-to-tail-how-knowledgeable-are-large#ran","syntology_url":"https://syntology.ai/paper/2308.10168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10168"}},"official":{"repos":["facebookresearch/head-to-tail"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/investigating-the-factual-knowledge-boundary","slug":"investigating-the-factual-knowledge-boundary","title":"Investigating the Factual Knowledge Boundary of Large Language Models with Retrieval Augmentation","date":"2023-07-20","arxiv_id":"2307.11019","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/investigating-the-factual-knowledge-boundary#ran","syntology_url":"https://syntology.ai/paper/2307.11019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11019"}},"official":{"repos":["rucaibox/llm-knowledge-boundary"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reta-llm-a-retrieval-augmented-large-language","slug":"reta-llm-a-retrieval-augmented-large-language","title":"RETA-LLM: A Retrieval-Augmented Large Language Model Toolkit","date":"2023-06-08","arxiv_id":"2306.05212","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reta-llm-a-retrieval-augmented-large-language#ran","syntology_url":"https://syntology.ai/paper/2306.05212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05212"}},"official":{"repos":["ruc-gsai/yulan-ir"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"b950442825504b8b140d8f5d4c5acbe7b028ce808d7cf4af304d2a92d4e5562b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}