{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/position-wise-feed-forward-layer/papers/79","list_of":"/method/position-wise-feed-forward-layer","method":"Position-Wise Feed-Forward Layer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":79,"pages_in_order":139,"rows_per_page":100,"rows":[7801,7900],"of":13895,"counts":{"archive_papers_tagged":13895,"with_a_code_link":6514,"where_syntology_ran_a_sample":2229,"not_listed_spam_title":0,"listed":13895,"listed_where_code_ran":2229,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1902,"every_run_a_failure_of_syntologys_instrument":327,"listed_with_a_run_with_no_instrument_failure":1902,"listed_every_run_a_failure_of_syntologys_instrument":327,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/position-wise-feed-forward-layer","prev":"/method/position-wise-feed-forward-layer/papers/78","next":"/method/position-wise-feed-forward-layer/papers/80","papers":[{"paper":null,"slug":"human-like-intuitive-behavior-and-reasoning","title":"Human-Like Intuitive Behavior and Reasoning Biases Emerged in Language Models -- and Disappeared in GPT-4","date":"2023-06-13","arxiv_id":"2306.07622","n_code_links":0,"syntology":null},{"paper":null,"slug":"molcap-molecular-chemical-reactivity","title":"MolCAP: Molecular Chemical reActivity pretraining and prompted-finetuning enhanced molecular representation learning","date":"2023-06-13","arxiv_id":"2306.09187","n_code_links":0,"syntology":null},{"paper":null,"slug":"vector-quantized-graph-auto-encoder","title":"Discrete Graph Auto-Encoder","date":"2023-06-13","arxiv_id":"2306.07735","n_code_links":0,"syntology":null},{"paper":"/paper/xraygpt-chest-radiographs-summarization-using","slug":"xraygpt-chest-radiographs-summarization-using","title":"XrayGPT: Chest Radiographs Summarization using Medical Vision-Language Models","date":"2023-06-13","arxiv_id":"2306.07971","n_code_links":1,"syntology":null},{"paper":"/paper/aerialformer-multi-resolution-transformer-for","slug":"aerialformer-multi-resolution-transformer-for","title":"AerialFormer: Multi-resolution Transformer for Aerial Image Segmentation","date":"2023-06-12","arxiv_id":"2306.06842","n_code_links":1,"syntology":null},{"paper":null,"slug":"cd-ctfm-a-lightweight-cnn-transformer-network","title":"CD-CTFM: A Lightweight CNN-Transformer Network for Remote Sensing Cloud Detection Fusing Multiscale Features","date":"2023-06-12","arxiv_id":"2306.07186","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-covid-19-diagnosis-through-vision","title":"Enhancing COVID-19 Diagnosis through Vision Transformer-Based Analysis of Chest X-ray Images","date":"2023-06-12","arxiv_id":"2306.06914","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-attention-mechanisms-for-multimodal","title":"Exploring Attention Mechanisms for Multimodal Emotion Recognition in an Emergency Call Center Corpus","date":"2023-06-12","arxiv_id":"2306.07115","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-and-non-linguistic","title":"Large language models and (non-)linguistic recursion","date":"2023-06-12","arxiv_id":"2306.07195","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-tax-attorneys-a-case","title":"Large Language Models as Tax Attorneys: A Case Study in Legal Capabilities Emergence","date":"2023-06-12","arxiv_id":"2306.07075","n_code_links":0,"syntology":null},{"paper":"/paper/learning-multilingual-sentence","slug":"learning-multilingual-sentence","title":"Learning Multilingual Sentence Representations with Cross-lingual Consistency Regularization","date":"2023-06-12","arxiv_id":"2306.06919","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-mask-and-permute-visual-tokens","slug":"learning-to-mask-and-permute-visual-tokens","title":"Learning to Mask and Permute Visual Tokens for Vision Transformer Pre-Training","date":"2023-06-12","arxiv_id":"2306.07346","n_code_links":1,"syntology":null},{"paper":null,"slug":"leveraging-skill-to-skill-supervision-for","title":"Leveraging Skill-to-Skill Supervision for Knowledge Tracing","date":"2023-06-12","arxiv_id":"2306.06841","n_code_links":0,"syntology":null},{"paper":null,"slug":"lost-in-translation-large-language-models-in","title":"Lost in Translation: Large Language Models in Non-English Content Analysis","date":"2023-06-12","arxiv_id":"2306.07377","n_code_links":0,"syntology":null},{"paper":"/paper/mitigating-transformer-overconfidence-via","slug":"mitigating-transformer-overconfidence-via","title":"Mitigating Transformer Overconfidence via Lipschitz Regularization","date":"2023-06-12","arxiv_id":"2306.06849","n_code_links":1,"syntology":null},{"paper":null,"slug":"network-robustness-learning-via-graph","title":"A Graph Transformer-Driven Approach for Network Robustness Learning","date":"2023-06-12","arxiv_id":"2306.06913","n_code_links":0,"syntology":null},{"paper":null,"slug":"npvforensics-jointing-non-critical-phonemes","title":"NPVForensics: Jointing Non-critical Phonemes and Visemes for Deepfake Detection","date":"2023-06-12","arxiv_id":"2306.06885","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-based-extraction-of-social","title":"Prompt-based Extraction of Social Determinants of Health Using Few-shot Learning","date":"2023-06-12","arxiv_id":"2306.07170","n_code_links":0,"syntology":null},{"paper":"/paper/e-2-equivariant-vision-transformer","slug":"e-2-equivariant-vision-transformer","title":"$E(2)$-Equivariant Vision Transformer","date":"2023-06-11","arxiv_id":"2306.06722","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zjucdsyangkaifan/gevit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/inductive-reasoning-in-humans-and-large","slug":"inductive-reasoning-in-humans-and-large","title":"Inductive reasoning in humans and large language models","date":"2023-06-11","arxiv_id":"2306.06548","n_code_links":1,"syntology":null},{"paper":"/paper/local-to-global-perspectives-on-graph-neural","slug":"local-to-global-perspectives-on-graph-neural","title":"Local-to-global Perspectives on Graph Neural Networks","date":"2023-06-11","arxiv_id":"2306.06547","n_code_links":1,"syntology":null},{"paper":null,"slug":"robertweet-a-bert-language-model-for-romanian","title":"RoBERTweet: A BERT Language Model for Romanian Tweets","date":"2023-06-11","arxiv_id":"2306.06598","n_code_links":0,"syntology":null},{"paper":"/paper/vpuformer-visual-prompt-unified-transformer","slug":"vpuformer-visual-prompt-unified-transformer","title":"PVPUFormer: Probabilistic Visual Prompt Unified Transformer for Interactive Image Segmentation","date":"2023-06-11","arxiv_id":"2306.06656","n_code_links":2,"syntology":null},{"paper":"/paper/multi-modal-pre-training-for-medical-vision","slug":"multi-modal-pre-training-for-medical-vision","title":"Multi-modal Pre-training for Medical Vision-language Understanding and Generation: An Empirical Study with A New Benchmark","date":"2023-06-10","arxiv_id":"2306.06494","n_code_links":1,"syntology":null},{"paper":null,"slug":"shuffled-autoregression-for-motion","title":"Shuffled Autoregression For Motion Interpolation","date":"2023-06-10","arxiv_id":"2306.06367","n_code_links":0,"syntology":null},{"paper":null,"slug":"vista-morph-unsupervised-image-registration","title":"Vista-Morph: Unsupervised Image Registration of Visible-Thermal Facial Pairs","date":"2023-06-10","arxiv_id":"2306.06505","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-can-an-accent-identifier-learn-probing","title":"What Can an Accent Identifier Learn? Probing Phonetic and Prosodic Information in a Wav2vec2-based Accent Identification Model","date":"2023-06-10","arxiv_id":"2306.06524","n_code_links":0,"syntology":null},{"paper":"/paper/14-examples-of-how-llms-can-transform","slug":"14-examples-of-how-llms-can-transform","title":"14 Examples of How LLMs Can Transform Materials Science and Chemistry: A Reflection on a Large Language Model Hackathon","date":"2023-06-09","arxiv_id":"2306.06283","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["qai222/llm_organic_synthesis","doncamilom/bollama"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-dual-source-attention-transformer-for-multi","title":"A Gated Attention Transformer for Multi-Person Pose Tracking","date":"2023-06-09","arxiv_id":"2306.05807","n_code_links":0,"syntology":null},{"paper":null,"slug":"boosting-fast-and-high-quality-speech","title":"Boosting Fast and High-Quality Speech Synthesis with Linear Diffusion","date":"2023-06-09","arxiv_id":"2306.05708","n_code_links":0,"syntology":null},{"paper":null,"slug":"customizing-general-purpose-foundation-models","title":"Customizing General-Purpose Foundation Models for Medical Report Generation","date":"2023-06-09","arxiv_id":"2306.05642","n_code_links":0,"syntology":null},{"paper":"/paper/everybody-compose-deep-beats-to-music","slug":"everybody-compose-deep-beats-to-music","title":"Everybody Compose: Deep Beats To Music","date":"2023-06-09","arxiv_id":"2306.06284","n_code_links":1,"syntology":null},{"paper":"/paper/illumination-controllable-dehazing-network","slug":"illumination-controllable-dehazing-network","title":"Illumination Controllable Dehazing Network based on Unsupervised Retinex Embedding","date":"2023-06-09","arxiv_id":"2306.05675","n_code_links":1,"syntology":null},{"paper":"/paper/judging-llm-as-a-judge-with-mt-bench-and-1","slug":"judging-llm-as-a-judge-with-mt-bench-and-1","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","date":"2023-06-09","arxiv_id":"2306.05685","n_code_links":11,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["lm-sys/fastchat"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"lightweight-monocular-depth-estimation-via","title":"Lightweight Monocular Depth Estimation via Token-Sharing Transformer","date":"2023-06-09","arxiv_id":"2306.05682","n_code_links":0,"syntology":null},{"paper":"/paper/modet-learning-deformable-image-registration","slug":"modet-learning-deformable-image-registration","title":"ModeT: Learning Deformable Image Registration via Motion Decomposition Transformer","date":"2023-06-09","arxiv_id":"2306.05688","n_code_links":1,"syntology":null},{"paper":"/paper/morphosyntactic-probing-of-multilingual-bert","slug":"morphosyntactic-probing-of-multilingual-bert","title":"Morphosyntactic probing of multilingual BERT models","date":"2023-06-09","arxiv_id":"2306.06205","n_code_links":1,"syntology":null},{"paper":"/paper/poet-a-generative-model-of-protein-families","slug":"poet-a-generative-model-of-protein-families","title":"PoET: A generative model of protein families as sequences-of-sequences","date":"2023-06-09","arxiv_id":"2306.06156","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["OpenProteinAI/PoET"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/rankformer-listwise-learning-to-rank-using","slug":"rankformer-listwise-learning-to-rank-using","title":"RankFormer: Listwise Learning-to-Rank Using Listwide Labels","date":"2023-06-09","arxiv_id":"2306.05808","n_code_links":1,"syntology":null},{"paper":"/paper/reconstructing-human-expressiveness-in-piano","slug":"reconstructing-human-expressiveness-in-piano","title":"Reconstructing Human Expressiveness in Piano Performances with a Transformer Network","date":"2023-06-09","arxiv_id":"2306.06040","n_code_links":1,"syntology":null},{"paper":null,"slug":"virtual-node-tuning-for-few-shot-node","title":"Virtual Node Tuning for Few-shot Node Classification","date":"2023-06-09","arxiv_id":"2306.06063","n_code_links":0,"syntology":null},{"paper":null,"slug":"city-wide-origin-destination-matrix","title":"Complexity-aware Large Scale Origin-Destination Network Generation via Diffusion Model","date":"2023-06-08","arxiv_id":"2306.04873","n_code_links":0,"syntology":null},{"paper":"/paper/daccord-un-jeu-de-donnees-pour-la-detection","slug":"daccord-un-jeu-de-donnees-pour-la-detection","title":"DACCORD : un jeu de données pour la Détection Automatique d'énonCés COntRaDictoires en français","date":"2023-06-08","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"deep-learning-method-for-object-tracking","title":"Deep Learning Method for Cell-Wise Object Tracking, Velocity Estimation and Projection of Sensor Data over Time","date":"2023-06-08","arxiv_id":"2306.06126","n_code_links":0,"syntology":null},{"paper":"/paper/invpt-inverted-pyramid-multi-task-transformer","slug":"invpt-inverted-pyramid-multi-task-transformer","title":"InvPT++: Inverted Pyramid Multi-Task Transformer for Visual Scene Understanding","date":"2023-06-08","arxiv_id":"2306.04842","n_code_links":1,"syntology":null},{"paper":"/paper/large-scale-dataset-pruning-with-dynamic","slug":"large-scale-dataset-pruning-with-dynamic","title":"Large-scale Dataset Pruning with Dynamic Uncertainty","date":"2023-06-08","arxiv_id":"2306.05175","n_code_links":2,"syntology":{"ran":11,"of":11,"n_ran_checked":8,"n_instrument":3,"unverified":0,"pointer_only":7,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["baai-dcai/dataset-pruning"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/m3exam-a-multilingual-multimodal-multilevel","slug":"m3exam-a-multilingual-multimodal-multilevel","title":"M3Exam: A Multilingual, Multimodal, Multilevel Benchmark for Examining Large Language Models","date":"2023-06-08","arxiv_id":"2306.05179","n_code_links":1,"syntology":null},{"paper":null,"slug":"mapping-the-challenges-of-hci-an-application","title":"Mapping the Challenges of HCI: An Application and Evaluation of ChatGPT and GPT-4 for Mining Insights at Scale","date":"2023-06-08","arxiv_id":"2306.05036","n_code_links":0,"syntology":null},{"paper":"/paper/mixture-of-domain-adapters-decoupling-and","slug":"mixture-of-domain-adapters-decoupling-and","title":"Mixture-of-Domain-Adapters: Decoupling and Injecting Domain Knowledge to Pre-trained Language Models Memories","date":"2023-06-08","arxiv_id":"2306.05406","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["amano-aki/mixture-of-domain-adapters"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-level-multiple-instance-learning-with","slug":"multi-level-multiple-instance-learning-with","title":"Multi-level Multiple Instance Learning with Transformer for Whole Slide Image Classification","date":"2023-06-08","arxiv_id":"2306.05029","n_code_links":1,"syntology":null},{"paper":null,"slug":"muti-scale-and-token-mergence-make-your-vit","title":"Multi-Scale And Token Mergence: Make Your ViT More Efficient","date":"2023-06-08","arxiv_id":"2306.04897","n_code_links":0,"syntology":null},{"paper":"/paper/point-lgmask-local-and-global-contexts","slug":"point-lgmask-local-and-global-contexts","title":"Point-LGMask: Local and Global Contexts Embedding for Point Cloud Pre-training with Multi-Ratio Masking","date":"2023-06-08","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"rate-forecaster-based-energy-aware-band","title":"Rate Forecaster based Energy Aware Band Assignment in Multiband Networks","date":"2023-06-08","arxiv_id":"2306.05369","n_code_links":0,"syntology":null},{"paper":"/paper/toolalpaca-generalized-tool-learning-for","slug":"toolalpaca-generalized-tool-learning-for","title":"ToolAlpaca: Generalized Tool Learning for Language Models with 3000 Simulated Cases","date":"2023-06-08","arxiv_id":"2306.05301","n_code_links":3,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["tangqiaoyu/ToolAlpaca"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/2d-object-detection-with-transformers-a","slug":"2d-object-detection-with-transformers-a","title":"Object Detection with Transformers: A Review","date":"2023-06-07","arxiv_id":"2306.04670","n_code_links":2,"syntology":null},{"paper":"/paper/good-data-large-data-or-no-data-comparing","slug":"good-data-large-data-or-no-data-comparing","title":"Good Data, Large Data, or No Data? Comparing Three Approaches in Developing Research Aspect Classifiers for Biomedical Papers","date":"2023-06-07","arxiv_id":"2306.04820","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-self-supervision-for-a-better-data","title":"GPT Self-Supervision for a Better Data Annotator","date":"2023-06-07","arxiv_id":"2306.04349","n_code_links":0,"syntology":null},{"paper":"/paper/how-far-can-camels-go-exploring-the-state-of","slug":"how-far-can-camels-go-exploring-the-state-of","title":"How Far Can Camels Go? Exploring the State of Instruction Tuning on Open Resources","date":"2023-06-07","arxiv_id":"2306.04751","n_code_links":4,"syntology":{"ran":6,"of":12,"n_ran_checked":2,"n_instrument":4,"unverified":6,"pointer_only":2,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","official":{"repos":["allenai/open-instruct"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/instructeval-towards-holistic-evaluation-of","slug":"instructeval-towards-holistic-evaluation-of","title":"INSTRUCTEVAL: Towards Holistic Evaluation of Instruction-Tuned Large Language Models","date":"2023-06-07","arxiv_id":"2306.04757","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["declare-lab/instruct-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/normalization-layers-are-all-that-sharpness-1","slug":"normalization-layers-are-all-that-sharpness-1","title":"Normalization Layers Are All That Sharpness-Aware Minimization Needs","date":"2023-06-07","arxiv_id":"2306.04226","n_code_links":1,"syntology":null},{"paper":"/paper/self-supervised-audio-teacher-student","slug":"self-supervised-audio-teacher-student","title":"Self-supervised Audio Teacher-Student Transformer for Both Clip-level and Frame-level Tasks","date":"2023-06-07","arxiv_id":"2306.04186","n_code_links":2,"syntology":null},{"paper":"/paper/tec-net-vision-transformer-embrace","slug":"tec-net-vision-transformer-embrace","title":"TEC-Net: Vision Transformer Embrace Convolutional Neural Networks for Medical Image Segmentation","date":"2023-06-07","arxiv_id":"2306.04086","n_code_links":1,"syntology":null},{"paper":"/paper/the-two-word-test-a-semantic-benchmark-for","slug":"the-two-word-test-a-semantic-benchmark-for","title":"The Two Word Test: A Semantic Benchmark for Large Language Models","date":"2023-06-07","arxiv_id":"2306.04610","n_code_links":1,"syntology":null},{"paper":"/paper/change-diffusion-change-detection-map","slug":"change-diffusion-change-detection-map","title":"GCD-DDPM: A Generative Change Detection Model Based on Difference-Feature Guided DDPM","date":"2023-06-06","arxiv_id":"2306.03424","n_code_links":1,"syntology":null},{"paper":"/paper/cit-net-convolutional-neural-networks-hand-in","slug":"cit-net-convolutional-neural-networks-hand-in","title":"CiT-Net: Convolutional Neural Networks Hand in Hand with Vision Transformers for Medical Image Segmentation","date":"2023-06-06","arxiv_id":"2306.03373","n_code_links":1,"syntology":{"ran":15,"of":23,"n_ran_checked":15,"n_instrument":0,"unverified":8,"pointer_only":23,"phrase":"15 ran (of which 15 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified; every one of the 15 samples that ran constructed an object rather than computing a result","official":{"repos":["sr0920/cit-net"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":15,"n_ran_no_instrument_failure":15,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":"/paper/long-term-5g-network-traffic-forecasting-via","slug":"long-term-5g-network-traffic-forecasting-via","title":"Long term 5G network traffic forecasting via modeling non-stationarity with deep learning","date":"2023-06-06","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/pgformer-proxy-bridged-game-transformer-for","slug":"pgformer-proxy-bridged-game-transformer-for","title":"PGformer: Proxy-Bridged Game Transformer for Multi-Person Highly Interactive Extreme Motion Prediction","date":"2023-06-06","arxiv_id":"2306.03374","n_code_links":0,"syntology":null},{"paper":"/paper/sgat4pass-spherical-geometry-aware","slug":"sgat4pass-spherical-geometry-aware","title":"SGAT4PASS: Spherical Geometry-Aware Transformer for PAnoramic Semantic Segmentation","date":"2023-06-06","arxiv_id":"2306.03403","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":5,"n_instrument":4,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["tencentarc/sgat4pass"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"textformer-a-query-based-end-to-end-text","title":"TextFormer: A Query-based End-to-End Text Spotter with Mixed Supervision","date":"2023-06-06","arxiv_id":"2306.03377","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-preference-learning-for-offline-rl","title":"PEARL: Zero-shot Cross-task Preference Alignment and Robust Reward Learning for Robotic Manipulation","date":"2023-06-06","arxiv_id":"2306.03615","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-large-language-models-on-cmexam","slug":"benchmarking-large-language-models-on-cmexam","title":"Benchmarking Large Language Models on CMExam -- A Comprehensive Chinese Medical Exam Dataset","date":"2023-06-05","arxiv_id":"2306.03030","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["williamliujl/cmexam"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/cross-drone-transformer-network-for-robust","slug":"cross-drone-transformer-network-for-robust","title":"Cross-Drone Transformer Network for Robust Single Object Tracking","date":"2023-06-05","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-gpt-model-pre-training-using-tensor","title":"Efficient GPT Model Pre-training using Tensor Train Matrix Representation","date":"2023-06-05","arxiv_id":"2306.02697","n_code_links":0,"syntology":null},{"paper":"/paper/orca-progressive-learning-from-complex","slug":"orca-progressive-learning-from-complex","title":"Orca: Progressive Learning from Complex Explanation Traces of GPT-4","date":"2023-06-05","arxiv_id":"2306.02707","n_code_links":4,"syntology":null},{"paper":null,"slug":"permutation-decision-trees","title":"Permutation Decision Trees","date":"2023-06-05","arxiv_id":"2306.02617","n_code_links":0,"syntology":null},{"paper":null,"slug":"selfevolve-a-code-evolution-framework-via","title":"SelfEvolve: A Code Evolution Framework via Large Language Models","date":"2023-06-05","arxiv_id":"2306.02907","n_code_links":0,"syntology":null},{"paper":"/paper/solving-np-hard-min-max-routing-problems-as","slug":"solving-np-hard-min-max-routing-problems-as","title":"Equity-Transformer: Solving NP-hard Min-Max Routing Problems as Sequential Generation with Equity Context","date":"2023-06-05","arxiv_id":"2306.02689","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kaist-silab/equity-transformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"transformer-based-unet-with-multi-headed","title":"Transformer-Based UNet with Multi-Headed Cross-Attention Skip Connections to Eliminate Artifacts in Scanned Documents","date":"2023-06-05","arxiv_id":"2306.02815","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-mathematical-abstraction-for-balancing-the","title":"A Mathematical Abstraction for Balancing the Trade-off Between Creativity and Reality in Large Language Models","date":"2023-06-04","arxiv_id":"2306.02295","n_code_links":0,"syntology":null},{"paper":"/paper/auto-gpt-for-online-decision-making","slug":"auto-gpt-for-online-decision-making","title":"Auto-GPT for Online Decision Making: Benchmarks and Additional Opinions","date":"2023-06-04","arxiv_id":"2306.02224","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["younghuman/llmagent"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/graph-transformer-for-recommendation","slug":"graph-transformer-for-recommendation","title":"Graph Transformer for Recommendation","date":"2023-06-04","arxiv_id":"2306.02330","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hkuds/gformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"modular-transformers-compressing-transformers","title":"Modular Transformers: Compressing Transformers into Modularized Layers for Flexible Efficient Inference","date":"2023-06-04","arxiv_id":"2306.02379","n_code_links":0,"syntology":null},{"paper":null,"slug":"gat-gan-a-graph-attention-based-time-series","title":"GAT-GAN : A Graph-Attention-based Time-Series Generative Adversarial Network","date":"2023-06-03","arxiv_id":"2306.01999","n_code_links":0,"syntology":null},{"paper":null,"slug":"lightweight-structure-aware-transformer","title":"Lightweight Structure-aware Transformer Network for VHR Remote Sensing Image Change Detection","date":"2023-06-03","arxiv_id":"2306.01988","n_code_links":0,"syntology":null},{"paper":"/paper/memorization-capacity-of-multi-head-attention","slug":"memorization-capacity-of-multi-head-attention","title":"Memorization Capacity of Multi-Head Attention in Transformers","date":"2023-06-03","arxiv_id":"2306.02010","n_code_links":1,"syntology":null},{"paper":null,"slug":"question-context-alignment-and-answer-context","title":"Question-Context Alignment and Answer-Context Dependencies for Effective Answer Sentence Selection","date":"2023-06-03","arxiv_id":"2306.02196","n_code_links":0,"syntology":null},{"paper":"/paper/transrupnet-for-improved-out-of-distribution","slug":"transrupnet-for-improved-out-of-distribution","title":"TransRUPNet for Improved Polyp Segmentation","date":"2023-06-03","arxiv_id":"2306.02176","n_code_links":1,"syntology":null},{"paper":null,"slug":"unsupervised-low-light-image-enhancement","title":"Unsupervised Low Light Image Enhancement Using SNR-Aware Swin Transformer","date":"2023-06-03","arxiv_id":"2306.02082","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-on-challenging-math","slug":"an-empirical-study-on-challenging-math","title":"MathChat: Converse to Tackle Challenging Math Problems with LLM Agents","date":"2023-06-02","arxiv_id":"2306.01337","n_code_links":2,"syntology":null},{"paper":null,"slug":"can-llms-like-gpt-4-outperform-traditional-ai","title":"Can LLMs like GPT-4 outperform traditional AI tools in dementia diagnosis? Maybe, but not today","date":"2023-06-02","arxiv_id":"2306.01499","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-language-models-for-mathematics","slug":"evaluating-language-models-for-mathematics","title":"Evaluating Language Models for Mathematics through Interactions","date":"2023-06-02","arxiv_id":"2306.01694","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["collinskatie/checkmate"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/explainability-of-speech-recognition","slug":"explainability-of-speech-recognition","title":"Explainability of Speech Recognition Transformers via Gradient-based Attention Visualization","date":"2023-06-02","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/generalist-equivariant-transformer-towards-3d","slug":"generalist-equivariant-transformer-towards-3d","title":"Generalist Equivariant Transformer Towards 3D Molecular Interaction Learning","date":"2023-06-02","arxiv_id":"2306.01474","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thunlp-mt/get"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improved-training-for-end-to-end-streaming","title":"Improved Training for End-to-End Streaming Automatic Speech Recognition Model with Punctuation","date":"2023-06-02","arxiv_id":"2306.01296","n_code_links":0,"syntology":null},{"paper":"/paper/nnmobile-net-rethinking-cnn-design-for-deep","slug":"nnmobile-net-rethinking-cnn-design-for-deep","title":"nnMobileNet: Rethinking CNN for Retinopathy Research","date":"2023-06-02","arxiv_id":"2306.01289","n_code_links":2,"syntology":null},{"paper":null,"slug":"rita-group-attention-is-all-you-need-for","title":"RITA: Group Attention is All You Need for Timeseries Analytics","date":"2023-06-02","arxiv_id":"2306.01926","n_code_links":0,"syntology":null},{"paper":"/paper/towards-robust-fastspeech-2-by-modelling","slug":"towards-robust-fastspeech-2-by-modelling","title":"Towards Robust FastSpeech 2 by Modelling Residual Multimodality","date":"2023-06-02","arxiv_id":"2306.01442","n_code_links":1,"syntology":null},{"paper":"/paper/transformer-based-annotation-bias-aware","slug":"transformer-based-annotation-bias-aware","title":"Transformer-based Annotation Bias-aware Medical Image Segmentation","date":"2023-06-02","arxiv_id":"2306.01340","n_code_links":1,"syntology":null},{"paper":"/paper/a-novel-driver-distraction-behavior-detection","slug":"a-novel-driver-distraction-behavior-detection","title":"A Novel Driver Distraction Behavior Detection Method Based on Self-supervised Learning with Masked Image Modeling","date":"2023-06-01","arxiv_id":"2306.00543","n_code_links":1,"syntology":null},{"paper":null,"slug":"accelerated-fingerprint-enhancement-a-gpu","title":"Accelerated Fingerprint Enhancement: A GPU-Optimized Mixed Architecture Approach","date":"2023-06-01","arxiv_id":"2306.00272","n_code_links":0,"syntology":null}],"record_sha256":"b6b2a74406a2621a8e352d6fea394037d973b992476d34fe9652e4bc2d84e7dd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}