{"id":37540,"date":"2025-01-30T16:01:17","date_gmt":"2025-01-30T08:01:17","guid":{"rendered":"https:\/\/17aitech.com\/?p=37540"},"modified":"2025-01-30T16:01:17","modified_gmt":"2025-01-30T08:01:17","slug":"%e6%af%8f%e6%9c%88%e9%83%bd%e6%9c%89%e9%87%8d%e7%a3%85%e7%a0%94%e7%a9%b6%ef%bc%8c2024%e5%85%a8%e5%b9%b4%e5%80%bc%e5%be%97%e4%b8%80%e8%af%bb%e7%9a%84%e8%ae%ba%e6%96%87%e9%83%bd%e5%9c%a8%e8%bf%99","status":"publish","type":"post","link":"https:\/\/17aitech.com\/?p=37540","title":{"rendered":"\u6bcf\u6708\u90fd\u6709\u91cd\u78c5\u7814\u7a76\uff0c2024\u5168\u5e74\u503c\u5f97\u4e00\u8bfb\u7684\u8bba\u6587\u90fd\u5728\u8fd9\u4e86"},"content":{"rendered":"<p>\u6587\u7ae0\u6765\u6e90\u4e8e\u4e92\u8054\u7f51:<a href=\"https:\/\/www.jiqizhixin.com\/articles\/2025-01-01-2\" target=\"_blank\">\u6bcf\u6708\u90fd\u6709\u91cd\u78c5\u7814\u7a76\uff0c2024\u5168\u5e74\u503c\u5f97\u4e00\u8bfb\u7684\u8bba\u6587\u90fd\u5728\u8fd9\u4e86<\/a><\/p>\n<p>2024 \u5e74\uff0c\u662f AI \u9886\u57df\u8ba9\u4eba\u5174\u594b\u7684\u4e00\u5e74\u3002\u5728\u8fd9\u4e00\u5e74\u4e2d\uff0c\u5404\u5927\u79d1\u6280\u516c\u53f8\u3001\u673a\u6784\u53d1\u5e03\u4e86\u6570\u4e0d\u80dc\u6570\u7684\u7814\u7a76\u3002<\/p>\n<p>\u4ece\u5e74\u521d\u7684 Sora\uff0c\u5230\u5e74\u5c3e DeepSeek-V3\uff0c\u6211\u4eec\u89c1\u8bc1\u4e86 AI \u4e00\u8f6e\u53c8\u4e00\u8f6e\u7684\u8f70\u70b8\uff0cAI\u7ed9\u6211\u4eec\u5e26\u6765\u4e86\u610f\u60f3\u4e0d\u5230\u7684\u60ca\u559c\u3002<\/p>\n<p>\u5728\u8fd9\u4e00\u5e74\u4e2d\uff0cAI \u8bba\u6587\u88ab\u6e90\u6e90\u4e0d\u65ad\u7684\u4ea7\u51fa\u3002\u5bf9\u4e8e\u521a\u521a\u8fc7\u53bb\u7684 2024 \u5e74\uff0c\u6709\u54ea\u4e9b\u8bba\u6587\u503c\u5f97\u53cd\u590d\u9605\u8bfb\uff1f\u77e5\u540d\u673a\u5668\u5b66\u4e60\u4e0e AI \u7814\u7a76\u8005 Sebastian Raschka \u6574\u7406\u4e86\u4e00\u4efd\u5173\u4e8eLLM \u7684\u9605\u8bfb\u6e05\u5355\uff0c\u6e05\u5355\u8be6\u7ec6\u4ecb\u7ecd\u4e86\u6bcf\u4e2a\u6708\u90fd\u6709\u54ea\u4e9b\u91cd\u8981\u8bba\u6587\u4ea7\u51fa\u3002<\/p>\n<p><a href=\"https:\/\/17aitech.com\/wp-content\/uploads\/2025\/01\/frc-461b2de4507e35e20473b9f31252bcce.png\" data-fancybox=\"images\" data-fancybox=\"gallery\"><img decoding=\"async\" src=\"https:\/\/17aitech.com\/wp-content\/uploads\/2025\/01\/frc-461b2de4507e35e20473b9f31252bcce.png\"><\/a><\/p>\n<p>\u539f\u6587\u94fe\u63a5\uff1ahttps:\/\/sebastianraschka.com\/blog\/2024\/llm-research-papers-the-2024-list.html<\/p>\n<p><strong>\u4e00\u6708\u8bba\u6587<\/strong><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAstraios: Parameter-Efficient Instruction Tuning Code Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.00788<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Comprehensive Study of Knowledge Editing for Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.01286<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLM Maybe LongLM: Self-Extend LLM Context Window Without Tuning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.01325<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650903457&amp;idx=4&amp;sn=05f3f84d0a233df66ed5288f0f1e38af&amp;scene=21#wechat_redirect\" target=\"_blank\">Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.01335<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLaMA Beyond English: An Empirical Study on Language Capability Transfer<\/p>\n<p>\u8bba\u6587\u94fe\u63a5 https:\/\/arxiv.org\/abs\/2401.01055<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Mechanistic Understanding of Alignment Algorithms: A Case Study on DPO and Toxicity<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.01967<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLaMA Pro: Progressive LLaMA with Block Expansion<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.02415<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLM Augmented LLMs: Expanding Capabilities through Composition<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.02412<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a <a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650904746&amp;idx=3&amp;sn=a2fea6977ccf2f44c02fd66e28b726a4&amp;scene=21#wechat_redirect\" target=\"_blank\">Blending Is All You Need: Cheaper, Better Alternative to Trillion-Parameters LLM<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2401.02994<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDeepSeek LLM: Scaling Open-Source Language Models with Longtermism<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.02954<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDenoising Vision Transformers<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.02957<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLong Context Compression with Activation Beacon<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.03462<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650904018&amp;idx=4&amp;sn=dc55a2e3c3837a2c3ab68ae09a506a78&amp;scene=21#wechat_redirect\" target=\"_blank\">Mixtral of Experts<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2401.04088<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650905254&amp;idx=3&amp;sn=71a840e160797a8052a8d6ea4fcf4dec&amp;scene=21#wechat_redirect\" target=\"_blank\">MoE-Mamba: Efficient Selective State Space Models with Mixture of Experts<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.04081<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Minimaximalist Approach to Reinforcement Learning from Human Feedback<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.04056<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRoSA: Accurate Parameter-Efficient Fine-Tuning via Robust Adaptation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2401.04679<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.05566<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTransformers are Multi-State RNNs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.06104<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Closer Look at AUROC and AUPRC under Class Imbalance<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.06091<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAn Experimental Design Framework for Label-Efficient Supervised Finetuning of Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.06692<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTuning Language Models by Proxy<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2401.08565<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScalable Pre-training of Large Autoregressive Image Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5 https:\/\/arxiv.org\/abs\/2401.08541<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCode Generation with AlphaCodium: From Prompt Engineering to Flow Engineering<\/p>\n<p>\u8bba\u6587\u94fe\u63a5https:\/\/arxiv.org\/abs\/2401.08500<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRAG vs Fine-tuning: Pipelines, Tradeoffs, and a Case Study on Agriculture<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2401.08406<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aReFT: Reasoning with Reinforced Fine-Tuning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2401.08967<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDiffusionGPT: LLM-Driven Text-to-Image Generation System<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2401.10061<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSelf-Rewarding Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.10020<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650905098&amp;idx=4&amp;sn=aef64ada941b8dde43eee63ca49c9741&amp;scene=21#wechat_redirect\" target=\"_blank\">VMamba: Visual State Space Model<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2401.10166<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aKnowledge Fusion of Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2401.10491<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSpatialVLM: Endowing Vision-Language Models with Spatial Reasoning Capabilities<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.12168<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWARM: On the Benefits of Weight Averaged Reward Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2401.12187<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a Spotting LLMs With Binoculars: Zero-Shot Detection of Machine-Generated Text<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2401.12070<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650906843&amp;idx=3&amp;sn=a2c32347f2cb12b81d3b284aaa974b3b&amp;scene=21#wechat_redirect\" target=\"_blank\">MambaByte: Token-free Selective State Space Model<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.13660<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSpacTor-T5: Pre-training T5 Models with Span Corruption and Replaced Token Detection<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.13160<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRethinking Patch Dependence for Masked Autoencoders<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.14391<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPix2gestalt: Amodal Segmentation by Synthesizing Wholes<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.14398<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMultimodal Pathway: Improve Transformers with Irrelevant Data from Other Modalities<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.14405<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.15077<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650906368&amp;idx=3&amp;sn=de29015cd3874bf021926bafa358f5c5&amp;scene=21#wechat_redirect\" target=\"_blank\">MoE-LLaVA: Mixture of Experts for Large Vision-Language Models<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.15947<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRephrasing the Web: A Recipe for Compute and Data-Efficient Language Modeling<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2401.16380<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aKVQuant: Towards 10 Million Context Length LLM Inference with KV Cache Quantization<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2401.18079<\/p>\n<p><strong>\u4e8c\u6708\u8bba\u6587<\/strong><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEfficient Exploration for LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.00396<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aOLMo: Accelerating the Science of Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.00838<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTiny Titans: Can Smaller Large Language Models Punch Above Their Weight in the Real World for Meeting Summarization?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.00841<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRepeat After Me: Transformers are Better than State Space Models at Copying<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.01032<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLiPO: Listwise Preference Optimization through Learning-to-Rank<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.01878<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aFindingEmo: An Image Dataset for Emotion Recognition in the Wild<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.01355<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650908860&amp;idx=5&amp;sn=57ab71abc8918f4558af9044cb9ed0c3&amp;scene=21#wechat_redirect\" target=\"_blank\">More Agents Is All You Need<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.05120<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.03300<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMobileVLM V2: Faster and Stronger Baseline for Vision Language Model<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.03766<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Phase Transition Between Positional and Semantic Learning in a Solvable Model of Dot-Product Attention<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.03902<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScaling Laws for Downstream Task Performance of Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.04177<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMOMENT: A Family of Open Time-series Foundation Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.03885<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aVision Superalignment: Weak-to-Strong Generalization for Vision Foundation Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.03749<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSelf-Discover: Large Language Models Self-Compose Reasoning Structures<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.03620<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aGrandmaster-Level Chess Without Search<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.04494<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDirect Language Model Alignment from Online AI Feedback<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.04792<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBuffer Overflow in Mixture of Experts<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.05526<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe Boundary of Neural Network Trainability is Fractal<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.06184<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aODIN: Disentangled Reward Mitigates Hacking in RLHF<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.07319<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPolicy Improvement using Language Feedback Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.07876<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScaling Laws for Fine-Grained Mixture of Experts<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.07871<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAya Model: An Instruction Finetuned Open-Access Multilingual Language Model<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.07610<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aStep-On-Feet Tuning: Scaling Self-Alignment of LLMs via Bootstrapping<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.07610<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSuppressing Pink Elephants with Direct Principle Feedback<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.07896<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWorld Model on Million-Length Video And Language With RingAttention<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.08268<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMixtures of Experts Unlock Parameter Scaling for Deep RL<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.08609<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDoRA: Weight-Decomposed Low-Rank Adaptation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.09353<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTransformers Can Achieve Length Generalization But Not Robustly<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.09371<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBASE TTS: Lessons From Building a Billion-Parameter Text-to-Speech Model on 100K Hours of Data<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.08093<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRecovering the Pre-Fine-Tuning Weights of Generative Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.10208<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aGenerative Representational Instruction Tuning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.09906<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aFinTral: A Family of GPT-4 Level Multimodal Financial Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.10986<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650909283&amp;idx=4&amp;sn=ed23c7765fc418fc8d768230c1fb2167&amp;scene=21#wechat_redirect\" target=\"_blank\">OneBit: Towards Extremely Low-bit Large Language Models<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.11295<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLongAgent: Scaling Language Models to 128k Context through Multi-Agent Collaboration<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.11550<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aReformatted Alignment<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.12219<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650909335&amp;idx=4&amp;sn=52666111800aab01aeb55710f4b294ee&amp;scene=21#wechat_redirect\" target=\"_blank\">AnyGPT: Unified Multimodal LLM with Discrete Sequence Modeling<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.12226<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTowards Cross-Tokenizer Distillation: the Universal Logit Distillation Loss for LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.12030<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLoRA+: Efficient Low Rank Adaptation of Large Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.12354<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650908519&amp;idx=4&amp;sn=12740683e7cf0ac0d9a1c661e2b4308f&amp;scene=21#wechat_redirect\" target=\"_blank\">Neural Network Diffusion<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2402.13144<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650908360&amp;idx=3&amp;sn=5c630c4e75d3393fbfc4b684cf8de78d&amp;scene=21#wechat_redirect\" target=\"_blank\">YOLOv9: Learning What You Want to Learn Using Programmable Gradient Information<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.13616<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLongRoPE: Extending LLM Context Window Beyond 2 Million Tokens<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1ahttps:\/\/arxiv.org\/abs\/2402.13753<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLarge Language Models for Data Annotation: A Survey<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.13446<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTinyLLaVA: A Framework of Small-scale Large Multimodal Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.14289<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBack to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.14740<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650908691&amp;idx=2&amp;sn=95082998c75df28509f4a837a573d020&amp;scene=21#wechat_redirect\" target=\"_blank\">\u00a0Genie: Generative Interactive Environments<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.15391<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCARTE: Pretraining and Transfer for Tabular Learning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.16785<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650908962&amp;idx=3&amp;sn=9966e7954b48ab986149307a590617a8&amp;scene=21#wechat_redirect\" target=\"_blank\">The Era of 1-bit LLMs: All Large Language Models are in 1.58 Bits<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.17764<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSora Generates Videos with Stunning Geometrical Consistency<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.17403<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWhen Scaling Meets LLM Finetuning: The Effect of Data, Model and Finetuning Method<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.17193<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650909283&amp;idx=2&amp;sn=90da9c570137489f5311a5ec56f36763&amp;scene=21#wechat_redirect\" target=\"_blank\">Griffin: Mixing Gated Linear Recurrences with Local Attention for Efficient Language Models<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2402.19427<\/p>\n<p><strong>\u4e09\u6708\u8bba\u6587<\/strong><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLearning and Leveraging World Models in Visual Representation Learning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.00504<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aImproving LLM Code Generation with Grammar Augmentation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.01632<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe Hidden Attention of Mamba Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.01590<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTraining-Free Pretrained Model Merging<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.01753<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aVision-RWKV: Efficient and Scalable Visual Perception with RWKV-Like Architectures<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.02308<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe WMDP Benchmark: Measuring and Reducing Malicious Use With Unlearning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.03218<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEvolution Transformer: In-Context Evolutionary Optimization<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.02985<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEnhancing Vision-Language Pre-training with Rich Supervisions<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.03346<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScaling Rectified Flow Transformers for High-Resolution Image Synthesis<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.03206<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDesign2Code: How Far Are We From Automating Front-End Engineering?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.03163<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aShortGPT: Layers in Large Language Models are More Redundant Than You Expect<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.03853<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBacktracing: Retrieving the Cause of the Query<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.03956<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLearning to Decode Collaboratively with Multiple Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.03870<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSaulLM-7B: A pioneering Large Language Model for Law<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.03883<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAre Language Models Puzzle Prodigies? Algorithmic Puzzles Unveil Serious Challenges in Multimodal Reasoning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.03864<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a3D Diffusion Policy<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.03954<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMedMamba: Vision Mamba for Medical Image Classification<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.03849<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650910100&amp;idx=4&amp;sn=5a8ea9ea1c6d8ac9f7ef1589730bb1be&amp;scene=21#wechat_redirect\" target=\"_blank\">GaLore: Memory-Efficient LLM Training by Gradient Low-Rank Projection<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.03507<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aStop Regressing: Training Value Functions via Classification for Scalable Deep RL<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.03950<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aHow Far Are We from Intelligent Visual Deductive Reasoning?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.04732<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCommon 7B Language Models Already Possess Strong Math Capabilities<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.04706<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650918321&amp;idx=2&amp;sn=95646308a9cb687ba37c58e31dac1b09&amp;scene=21#wechat_redirect\" target=\"_blank\">Gemini 1.5: Unlocking Multimodal Understanding Across Millions of Tokens of Context<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.05530<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aIs Cosine-Similarity of Embeddings Really About Similarity?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.05440<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLM4Decompile: Decompiling Binary Code with Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.05286<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAlgorithmic Progress in Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.05812<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aStealing Part of a Production Language Model<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.06634<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aChronos: Learning the Language of Time Series<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.07815<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSimple and Scalable Strategies to Continually Pre-train Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.08763<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLanguage Models Scale Reliably With Over-Training and on Downstream Tasks<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.08540<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBurstAttention: An Efficient Distributed Attention Framework for Extremely Long Sequences<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.09347<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a LocalMamba: Visual State Space Model with Windowed Selective Scan<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.09338<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aGiT: Towards Generalist Vision Transformer through Universal Language Interface<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.09394<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMM1: Methods, Analysis &amp; Insights from Multimodal LLM Pre-training<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.09611<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a RAFT: Adapting Language Model to Domain Specific RAG<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.10131<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTnT-LLM: Text Mining at Scale with Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.12173<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a Decoding Compressed Trust: Scrutinizing the Trustworthiness of Efficient LLMs Under Compression<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.15447<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a PERL: Parameter Efficient Reinforcement Learning from Human Feedback<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.10704<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRewardBench: Evaluating Reward Models for Language Modeling<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.13787<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLlamaFactory: Unified Efficient Fine-Tuning of 100+ Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.13372<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRakutenAI-7B: Extending Large Language Models for Japanese<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.15484<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSiMBA: Simplified Mamba-Based Architecture for Vision and Multivariate Time Series<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.15360<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCan Large Language Models Explore In-Context?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.15371<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLM2LLM: Boosting LLMs with Novel Iterative Data Enhancement<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.15042<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a LLM Agent Operating System<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.16971<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe Unreasonable Ineffectiveness of the Deeper Layers<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.17887<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBioMedLM: A 2.7B Parameter Language Model Trained On Biomedical Text<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.18421<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aViTAR: Vision Transformer with Any Resolution<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.18361<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLong-form Factuality in Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.18802<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMini-Gemini: Mining the Potential of Multi-modality Vision Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2403.18814<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLISA: Layerwise Importance Sampling for Memory-Efficient Large Language Model Fine-Tuning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.17919<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMechanistic Design and Scaling of Hybrid Architectures<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.17844<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMagicLens: Self-Supervised Image Retrieval with Open-Ended Instructions<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.19651<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aModel Stock: All We Need Is Just a Few Fine-Tuned Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2403.19522<\/p>\n<p><strong>\u56db\u6708\u8bba\u6587<\/strong><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a Do Language Models Plan Ahead for Future Tokens?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.00859<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBigger is not Always Better: Scaling Properties of Latent Diffusion Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2404.01367<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe Fine Line: Navigating Large Language Model Pretraining with Down-streaming Capability Analysis<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.01204<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDiffusion-RWKV: Scaling RWKV-Like Architectures for Diffusion Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2404.04478<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMixture-of-Depths: Dynamically Allocating Compute in Transformer-Based Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2404.02258<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLong-context LLMs Struggle with Long In-context Learning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2404.02060<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEmergent Abilities in Reduced-Scale Generative Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.02204<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aJailbreaking Leading Safety-Aligned LLMs with Simple Adaptive Attacks<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.02151<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aOn the Scalability of Diffusion-based Text-to-Image Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.02883<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBAdam: A Memory Efficient Full Parameter Training Method for Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.02827<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCross-Attention Makes Inference Cumbersome in Text-to-Image Diffusion Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.02747<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDirect Nash Optimization: Teaching Language Models to Self-Improve with General Preferences<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.02151<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTraining LLMs over Neurally Compressed Text<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.03626<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCantTalkAboutThis: Aligning Language Models to Stay on Topic in Dialogues<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.03820<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aReFT: Representation Finetuning for Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.03592<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aVerifiable by Design: Aligning Language Models to Quote from Pre-Training Data<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.03862<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSigma: Siamese Mamba Network for Multi-Modal Semantic Segmentation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.04256<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAutoCodeRover: Autonomous Program Improvement<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.05427<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEagle and Finch: RWKV with Matrix-Valued States and Dynamic Recurrence<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.05892<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCodecLM: Aligning Language Models with Tailored Synthetic Data<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.05875<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMiniCPM: Unveiling the Potential of Small Language Models with Scalable Training Strategies<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.06395<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aElephants Never Forget: Memorization and Learning of Tabular Data in Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.06209<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLM2Vec: Large Language Models Are Secretly Powerful Text Encoders<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.05961<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAdapting LLaMA Decoder to Vision Transformer<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.06773<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a Leave No Context Behind: Efficient Infinite Context Transformers with Infini-attention<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.07143<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLoCO: Learning Long Contexts Offline<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.07979<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aJetMoE: Reaching Llama2 Performance with 0.1M Dollars<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.07413<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a Best Practices and Lessons Learned on Synthetic Data for Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.07503<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRho-1: Not All Tokens Are What You Need<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.07965<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPre-training Small Base LMs with Fewer Tokens<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.08634<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDataset Reset Policy Optimization for RLHF<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.08495<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLM In-Context Recall is Prompt Dependent<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.08865<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aState Space Model for New-Generation Network Alternative to Transformers: A Survey<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.09516<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aChinchilla Scaling: A Replication Attempt<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.10102<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLearn Your Reference Model for Real Good Alignment<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.09656<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aIs DPO Superior to PPO for LLM Alignment? A Comprehensive Study<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.10719<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScaling (Down) CLIP: A Comprehensive Analysis of Data, Architecture, and Training Strategies<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.08197<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aHow Faithful Are RAG Models? Quantifying the Tug-of-War Between RAG and LLMs\u2019 Internal Prior<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.10198<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Survey on Retrieval-Augmented Text Generation for Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2404.10981<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWhen LLMs are Unfit Use FastFit: Fast and Effective Text Classification with Many Classes<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.12365<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aToward Self-Improvement of LLMs via Imagination, Searching, and Criticizing<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.12253<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aOpenBezoar: Small, Cost-Effective and Open Models Trained on Mixes of Instruction Data<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.12195<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.13208<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAn Empirical Study of LLaMA3 Quantization: From LLMs to MLLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.14047<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650915526&amp;idx=3&amp;sn=55b4a4cb01b7dcc4b71044cc415877b0&amp;scene=21#wechat_redirect\" target=\"_blank\">Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.14219<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a <a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650915761&amp;idx=3&amp;sn=d233f9cb4e687f98a70dd834a7635ec4&amp;scene=21#wechat_redirect\" target=\"_blank\">OpenELM: An Efficient Language Model Family with Open-source Training and Inference Framework<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.14619<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a A Survey on Self-Evolution of Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.14662<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a Multi-Head Mixture-of-Experts<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.15045<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aNExT: Teaching Large Language Models to Reason about Code Execution<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.14662<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aGraph Machine Learning in the Era of Large Language Models (LLMs)<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.14928<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRetrieval Head Mechanistically Explains Long-Context Factuality<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.15574<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLayer Skip: Enabling Early Exit Inference and Self-Speculative Decoding<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.16710<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMake Your LLM Fully Utilize the Context<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2404.16811<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLoRA Land: 310 Fine-tuned LLMs that Rival GPT-4, A Technical Report<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.00732<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBetter &amp; Faster Large Language Models via Multi-token Prediction<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.19737<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRAG and RAU: A Survey on Retrieval-Augmented Language Model in Natural Language Processing<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.19543<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Primer on the Inner Workings of Transformer-based Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.00208<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWhen to Retrieve: Teaching LLMs to Utilize Information Retrieval Effectively<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2404.19705<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650916710&amp;idx=2&amp;sn=28024640e081228cc5887ad8c4aae8a3&amp;scene=21#wechat_redirect\" target=\"_blank\">KAN: Kolmogorov\u2013Arnold Networks<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2404.19756<\/p>\n<p><strong>\u4e94\u6708\u8bba\u6587<\/strong><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aIs Bigger Edit Batch Size Always Better? An Empirical Study on Model Editing with Llama-3<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.00664<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650917605&amp;idx=4&amp;sn=1be7b2a4993007794054100ceef5606f&amp;scene=21#wechat_redirect\" target=\"_blank\">Self-Play Preference Optimization for Language Model Alignment<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.00675<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Careful Examination of Large Language Model Performance on Grade School Arithmetic<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.00332<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPrometheus 2: An Open Source Language Model Specialized in Evaluating Other Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.01535<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWhat Matters When Building Vision-Language Models?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.02246<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aIs Flash Attention Stable?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.02803<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1avAttention: Dynamic Memory Management for Serving LLMs without PagedAttention<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.04437<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650917394&amp;idx=2&amp;sn=be53466538b409ccf9980d85eff3fd9d&amp;scene=21#wechat_redirect\" target=\"_blank\">xLSTM: Extended Long Short-Term Memory<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.04517<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aYou Only Cache Once: Decoder-Decoder Architectures for Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.05254<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650917057&amp;idx=4&amp;sn=2d3225438117cdea15e417b6aa6b3d1e&amp;scene=21#wechat_redirect\" target=\"_blank\">DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.04434<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aFishing for Magikarp: Automatically Detecting Under-trained Tokens in Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.05417<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDoes Fine-Tuning LLMs on New Knowledge Encourage Hallucinations?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.05904<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aValue Augmented Sampling for Language Model Alignment and Personalization<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a https:\/\/arxiv.org\/abs\/2405.06639<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPHUDGE: Phi-3 as Scalable Judge<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.08029<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRLHF Workflow: From Reward Modeling to Online RLHF<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.07863<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLoRA Learns Less and Forgets Less<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.09673<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aXmodel-VLM: A Simple Baseline for Multimodal Vision Language Model<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.09215<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aChameleon: Mixed-Modal Early-Fusion Foundation Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.09818<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTowards Modular LLMs by Building and Reusing a Library of LoRAs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.11157<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSLAB: Efficient Transformers with Simplified Linear Attention and Progressive Re-parameterized Batch Normalization<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.11582<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMoRA: High-Rank Updating for Parameter-Efficient Fine-Tuning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.12130<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650919194&amp;idx=2&amp;sn=e305b94dc2f62184f4f8e427d17aaf77&amp;scene=21#wechat_redirect\" target=\"_blank\">Attention as an RNN<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.13956<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDense Connector for MLLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.13800<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAlignGPT: Multi-modal Large Language Models with Adaptive Alignment Capability<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.14129<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a SimPO: Simple Preference Optimization with a Reference-Free Reward<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.14734<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aInstruction Tuning With Loss Over Instructions<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.14394<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe Road Less Scheduled<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.15682<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aStacking Your Transformers: A Closer Look at Model Growth for Efficient LLM Pre-Training<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.15319<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1agzip Predicts Data-dependent Scaling Laws<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.16684<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTrans-LoRA: Towards Data-free Transferable Parameter Efficient Finetuning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.17258<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aVeLoRA: Memory Efficient Training using Rank-1 Sub-Token Projections<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.17991<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLaMA-NAS: Efficient Neural Architecture Search for Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2405.18377<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aContextual Position Encoding: Learning to Count What\u2019s Important<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2405.18719<\/p>\n<p><strong>\u516d\u6708\u8bba\u6587<\/strong><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aShow, Don\u2019t Tell: Aligning Language Models with Demonstrated Feedback<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.00888<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSkywork-MoE: A Deep Dive into Training Techniques for Mixture-of-Experts Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.06563<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aOLoRA: Orthonormal Low-Rank Adaptation of Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.01775<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe Geometry of Categorical and Hierarchical Concepts in Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.01506<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTowards Scalable Automated Alignment of LLMs: A Survey<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.01252<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScalable MatMul-free Language Modeling<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.02528<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBlock Transformer: Global-to-Local Language Modeling for Fast Inference<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.02657<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBuffer of Thoughts: Thought-Augmented Reasoning with Large Language Models<\/p>\n<p>\u00a0\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.04271<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe Prompt Report: A Systematic Survey of Prompting Techniques<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.06608<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTransformers Need Glasses! Information Over-Squashing in Language Tasks<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.04267<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAre We Done with MMLU?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.04127<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aStep-aware Preference Optimization: Aligning Preference with Denoising Performance at Each Step<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.04314<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBoosting Large-scale Parallel Training Efficiency with C4: A Communication-Driven Approach<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.04594<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCRAG \u2013 Comprehensive RAG Benchmark<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.04744<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.04770<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMixture-of-Agents Enhances Large Language Model Capabilities<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.04692<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBERTs are Generative In-Context Learners<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.04823<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a3D-GRAND: A Million-Scale Dataset for 3D-LLMs with Better Grounding and Less Hallucination<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.05132<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCreativity Has Left the Chat: The Price of Debiasing Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.05587<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAutoregressive Model Beats Diffusion: Llama for Scalable Image Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.06525<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMargin-aware Preference Optimization for Aligning Diffusion Models Without Reference<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.06424<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aHusky: A Unified, Open-Source Language Agent for Multi-Step Reasoning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.06469<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a Turbo Sparse: Achieving LLM SOTA Performance with Minimal Activated Parameters<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.05955<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSelf-Tuning: Instructing LLMs to Effectively Acquire New Knowledge through Self-Teaching<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.06326<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAn Image is Worth 32 Tokens for Reconstruction and Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.07550<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTextGrad: Automatic \u201cDifferentiation\u201d via Text<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.07496<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSimple and Effective Masked Diffusion Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.07524<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aNever Miss A Beat: An Efficient Recipe for Context Window Extension of Large Language Models with Consistent \u201cMiddle\u201d Enhancement<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.07138<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSamba: Simple Hybrid State Space Models for Efficient Unlimited Context Language Modeling<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.07522<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMagpie: Alignment Data Synthesis from Scratch by Prompting Aligned LLMs with Nothing<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.08464<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWhat If We Recaption Billions of Web Images with LLaMA-3?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.08478<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLarge Language Model Unlearning via Embedding-Corrupted Prompts<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.07933<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLarge Language Models Must Be Taught to Know What They Don\u2019t Know<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.08391<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAn Empirical Study of Mamba-based Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.07887<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a Discovering Preference Optimization Algorithms with and for Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.08414<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTransformers Meet Neural Algorithmic Reasoners<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.09308<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMLKV: Multi-Layer Key-Value Heads for Memory Efficient Transformer Decoding<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.09297<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAn Image is Worth More Than 16&#215;16 Patches: Exploring Transformers on Individual Pixels<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.09415<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aFouRA: Fourier Low Rank Adaptation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.08798<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a Bootstrapping Language Models with DPO Implicit Rewards<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.09760<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBe like a Goldfish, Don\u2019t Memorize! Mitigating Memorization in Generative LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.10209<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRegularizing Hidden States Enables Learning Generalizable Reward Model for LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.10216<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTHEANINE: Revisiting Memory Management in Long-term Conversations with Timeline-augmented Response Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.10996<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTask Me Anything<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.11775<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aHow Do Large Language Models Acquire Factual Knowledge During Pretraining?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.11813<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1amDPO: Conditional Preference Optimization for Multimodal Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.11839<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650921965&amp;idx=3&amp;sn=d20e5e456077bfef06e7d6d2c656aab5&amp;scene=21#wechat_redirect\" target=\"_blank\">Nemotron-4 340B Technical Report<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.11704<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDataComp-LM: In Search of the Next Generation of Training Sets for Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.11794<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTokenization Falling Short: The Curse of Tokenization<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.11687<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a DeepSeek-Coder-V2: Breaking the Barrier of Closed-Source Models in Code Intelligence<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.11931<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aUnveiling Encoder-Free Vision-Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.11832<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aIterative Length-Regularized Direct Preference Optimization: A Case Study on Improving 7B Language Models to GPT-4 Level<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.11817<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aHARE: HumAn pRiors, a key to small language model Efficiency<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.11410<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMeasuring memorization in RLHF for code completion<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.11715<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSelf-MoE: Towards Compositional Large Language Models with Self-Specialized Experts<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.12034<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aFrom RAGs to Rich Parameters: Probing How Language Models Utilize External Knowledge Over Parametric Information for Factual Queries<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.12824<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aJudging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.12624<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCan Long-Context Language Models Subsume Retrieval, RAG, SQL, and More?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.13121<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aInstruction Pre-Training: Language Models are Supervised Multitask Learners<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.14491<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCan LLMs Learn by Teaching? A Preliminary Study<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.14629<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Tale of Trust and Accuracy: Base vs. Instruct LLMs in RAG Systems<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.14972<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a LongRAG: Enhancing Retrieval-Augmented Generation with Long-context LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.15319<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMoA: Mixture of Sparse Attention for Automatic Large Language Model Compression<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.14909<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEfficient Continual Pre-training by Mitigating the Stability Gap<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.14833<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSparser is Faster and Less is More: Efficient Sparse Attention for Long-Range Transformers<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.16747<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWARP: On the Benefits of Weight Averaged Rewarded Policies<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.16768<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAdam-mini: Use Fewer Learning Rates To Gain More<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.16793<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe FineWeb Datasets: Decanting the Web for the Finest Text Data at Scale<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.17557<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLongIns: A Challenging Long-context Instruction-based Exam for LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.17588<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aFollowing Length Constraints in Instructions<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.17744<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Closer Look into Mixture-of-Experts in Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2406.18219<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a RouteLLM: Learning to Route LLMs with Preference Data<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.18665<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aStep-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.18629<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDataset Size Recovery from LoRA Weights<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.19395<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aFrom Artificial Needles to Real Haystacks: Improving Retrieval Capabilities in LLMs by Finetuning on Synthetic Data<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.19292<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aChanging Answer Order Can Decrease MMLU Accuracy<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.19470<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDirect Preference Knowledge Distillation for Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.19774<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLM Critics Help Catch LLM Bugs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.00215<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScaling Synthetic Data Creation with 1,000,000,000 Personas<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1a https:\/\/arxiv.org\/abs\/2406.20094<\/p>\n<p><strong>\u4e03\u6708\u8bba\u6587<\/strong><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLM See, LLM Do: Guiding Data Generation to Target Non-Differentiable Objectives<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.01490<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSearching for Best Practices in Retrieval-Augmented Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.01219<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLet the Expert Stick to His Last: Expert-Specialized Fine-Tuning for Sparse Architectural Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.01906<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650927161&amp;idx=4&amp;sn=90a44edfc4182de251e9862861c52f4d&amp;scene=21#wechat_redirect\" target=\"_blank\">Diffusion Forcing: Next-token Prediction Meets Full-Sequence Diffusion<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.01392<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEliminating Position Bias of Language Models: A Mechanistic Approach<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.01100<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aJMInference 1.0: Accelerating Pre-filling for Long-Context LLMs via Dynamic Sparse Attention<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.02490<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTokenPacker: Efficient Visual Projector for Multimodal LLM<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.02392<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aReasoning in Large Language Models: A Geometric Perspective<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.02678<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRankRAG: Unifying Context Ranking with Retrieval-Augmented Generation in LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.02485<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAgentInstruct: Toward Generative Teaching with Agentic Flows<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.03502<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aHEMM: Holistic Evaluation of Multimodal Foundation Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.03418<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650925512&amp;idx=3&amp;sn=7a737e23afe4f1f14659fc07baa8c2b1&amp;scene=21#wechat_redirect\" target=\"_blank\">Mixture of A Million Experts<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.04153<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650925292&amp;idx=1&amp;sn=4ab40e0e0500187e329ff7a5c9904092&amp;scene=21#wechat_redirect\" target=\"_blank\">Learning to (Learn at Test Time): RNNs with Expressive Hidden States<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.04620<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650925762&amp;idx=3&amp;sn=a4b23bbed70ef4be3c810251f688f339&amp;scene=21#wechat_redirect\" target=\"_blank\">Vision Language Models Are Blind<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.06581<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSelf-Recognition in Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.06946<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aInference Performance Optimization for Large Language Models on CPUs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.07304<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aGradient Boosting Reinforcement Learning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.08250<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650925868&amp;idx=3&amp;sn=fae16a9ac8df3d922cdd4ff5d3674c7d&amp;scene=21#wechat_redirect\" target=\"_blank\">FlashAttention-3: Fast and Accurate Attention with Asynchrony and Low-precision<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.08608<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSpreadsheetLLM: Encoding Spreadsheets for Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.09025<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aNew Desiderata for Direct Preference Optimization<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.09072<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aContext Embeddings for Efficient Answer Generation in RAG<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.09252<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aQwen2 Technical Report<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.10671<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe Good, The Bad, and The Greedy: Evaluation of LLMs Should Not Ignore Non-Determinism<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.10457<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aFrom GaLore to WeLore: How Low-Rank Weights Non-uniformly Emerge from Low-Rank Gradients<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.11239<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aGoldFinch: High Performance RWKV\/Transformer Hybrid with Linear Pre-Fill and Extreme KV-Cache Compression<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.12077<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScaling Diffusion Transformers to 16 Billion Parameters<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.11633<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aNeedleBench: Can LLMs Do Retrieval and Reasoning in 1 Million Context Window?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.11963<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPatch-Level Training for Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.12665<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650930994&amp;idx=5&amp;sn=ff9035eca7c46f3347dca4519adfd5ff&amp;scene=21#wechat_redirect\" target=\"_blank\">LMMs-Eval: Reality Check on the Evaluation of Large Multimodal Models<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.12772<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Survey of Prompt Engineering Methods in Large Language Models for Different NLP Tasks<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.12994<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSpectra: A Comprehensive Study of Ternary, Quantized, and FP16 Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.12327<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAttention Overflow: Language Model Input Blur during Long-Context Missing Items Recommendation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.13481<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWeak-to-Strong Reasoning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.13647<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aUnderstanding Reference Policies in Direct Preference Optimization<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.13709<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScaling Laws with Vocabulary: Larger Models Deserve Larger Vocabularies<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.13623<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBOND: Aligning LLMs with Best-of-N Distillation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.14622<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCompact Language Models via Pruning and Knowledge Distillation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.14679<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650928796&amp;idx=4&amp;sn=5d50fd86ed5f09846ec50ef5212f19d0&amp;scene=21#wechat_redirect\" target=\"_blank\">LazyLLM: Dynamic Token Pruning for Efficient Long Context LLM Inference<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.14057<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMini-Sequence Transformer: Optimizing Intermediate Memory for Long Sequences Training<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.15892<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDDK: Distilling Domain Knowledge for Efficient Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.16154<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aGeneration Constraint Scaling Can Mitigate Hallucination<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.16908<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRetrieval Augmented Generation or Long-Context LLMs? A Comprehensive Study and Hybrid Approach<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.16833<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCourse-Correction: Safety Alignment Using Synthetic Preferences<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.16637<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aData Mixture Inference: What do BPE Tokenizers Reveal about their Training Data?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.16607<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650904974&amp;idx=1&amp;sn=08dcf899578eda25f48d0b571b9ad41a&amp;scene=21#wechat_redirect\" target=\"_blank\">Meta-Rewarding Language Models: Self-Improving Alignment with LLM-as-a-Meta-Judge<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.19594<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aImproving Retrieval Augmented Language Model with Self-Reasoning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.19813<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650928422&amp;idx=3&amp;sn=3b6ed65b6670eb27aa702f5c7ff950f1&amp;scene=21#wechat_redirect\" target=\"_blank\">Apple Intelligence Foundation Language Models<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.21075<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThinK: Thinner Key Cache by Query-Driven Pruning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.21018<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650927368&amp;idx=2&amp;sn=9d82c1f832143fda2c8cf473ba9fa424&amp;scene=21#wechat_redirect\" target=\"_blank\">The Llama 3 Herd of Models<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2407.21783<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650923969&amp;idx=4&amp;sn=0179e0823077c84f1f22f06d7ed4650b&amp;scene=21#wechat_redirect\" target=\"_blank\">Gemma 2: Improving Open Language Models at a Practical Size<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.00118<\/p>\n<p><strong>\u516b\u6708\u8bba\u6587<\/strong><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650928134&amp;idx=1&amp;sn=7855c453083236c5c5d121d3b4fb955e&amp;scene=21#wechat_redirect\" target=\"_blank\">SAM 2: Segment Anything in Images and Videos<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.00714<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPOA: Pre-training Once for Models of All Sizes<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.01031<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRAGEval: Scenario Specific RAG Evaluation Dataset Generation Framework<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.01262<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Survey of Mamba<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.01129<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMiniCPM-V: A GPT-4V Level MLLM on Your Phone<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.01800<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRAG Foundry: A Framework for Enhancing LLMs for Retrieval Augmented Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.02545<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSelf-Taught Evaluators<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.02666<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBioMamba: A Pre-trained Biomedical Language Representation Model Leveraging Mamba<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.02600<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEXAONE 3.0 7.8B Instruction Tuned Language Model<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.03541<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a1.5-Pints Technical Report: Pretraining in Days, Not Months \u2013 Your Language Model Thrives on Quality Data<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.03506<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aConversational Prompt Engineering<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.04560<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTrans-Tokenization and Cross-lingual Vocabulary Transfers: Language Adaptation of LLMs for Low-Resource NLP<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.04303<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650948841&amp;idx=1&amp;sn=f6c29781a9e1e66f3733d7c02d8c62f4&amp;scene=21#wechat_redirect\" target=\"_blank\">The AI Scientist: Towards Fully Automated Open-Ended Scientific Discovery<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.06292<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aHermes 3 Technical Report<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.12570<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCustomizing Language Models with Instance-wise LoRA for Sequential Recommendation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.10159<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEnhancing Robustness in Large Language Models: Prompting for Mitigating the Impact of Irrelevant Information<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.10615<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650931294&amp;idx=3&amp;sn=efa889f29c973d6c91f164bccc785791&amp;scene=21#wechat_redirect\" target=\"_blank\">To Code, or Not To Code? Exploring Impact of Code in Pre-training<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.10914<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLM Pruning and Distillation in Practice: The Minitron Approach<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.11796<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aJamba-1.5: Hybrid Transformer-Mamba Models at Scale<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.12570<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aControllable Text Generation for Large Language Models: A Survey<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.12599<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMulti-Layer Transformers Gradient Can be Approximated in Almost Linear Time<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.13233<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Practitioner&#8217;s Guide to Continual Multimodal Pretraining<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.14471<\/p>\n<p><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBuilding and better understanding vision-language models: insights and future directions<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.12637<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCURLoRA: Stable LLM Continual Fine-Tuning and Catastrophic Forgetting Mitigation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.14572<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650932383&amp;idx=2&amp;sn=7600df2e659e20f361626d354c424e8e&amp;scene=21#wechat_redirect\" target=\"_blank\">The Mamba in the Llama: Distilling and Accelerating Hybrid Models<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.15237<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aReMamba: Equip Mamba with Effective Long-Sequence Modeling<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.15496<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSmaller, Weaker, Yet Better: Training LLM Reasoners via Compute-Optimal Sampling<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2408.16737<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLongRecipe: Recipe for Efficient Long Context Generalization in Large Languge Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.00509<\/p>\n<p><strong>\u4e5d\u6708\u8bba\u6587<\/strong><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650932950&amp;idx=1&amp;sn=c90c67a7e31bde900a9f8c4a7c215a0b&amp;scene=21#wechat_redirect\" target=\"_blank\">OLMoE: Open Mixture-of-Experts Language Models<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.02060<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aIn Defense of RAG in the Era of Long-Context Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.01666<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAttention Heads of Large Language Models: A Survey<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.03752<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLongCite: Enabling LLMs to Generate Fine-grained Citations in Long-context QA<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.02897<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aHow Do Your Code LLMs Perform? Empowering Code Instruction Tuning with High-Quality Data<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.03810<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTheory, Analysis, and Best Practices for Sigmoid Self-Attention<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.04431<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLaMA-Omni: Seamless Speech Interaction with Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.06666<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWhat is the Role of Small Models in the LLM Era: A Survey<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.06857<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPolicy Filtration in RLHF to Fine-Tune LLM for Code Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.06957<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRetrievalAttention: Accelerating Long-Context LLM Inference via Vector Retrieval<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.10516<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650934959&amp;idx=1&amp;sn=28be321f470ab4df52dcc12035494a16&amp;scene=21#wechat_redirect\" target=\"_blank\">Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.12122<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650934959&amp;idx=1&amp;sn=28be321f470ab4df52dcc12035494a16&amp;scene=21#wechat_redirect\" target=\"_blank\">Qwen2.5-Coder Technical Report<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.12186<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aInstruction Following without Instruction Tuning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.14254<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aIs Preference Alignment Always the Best Option to Enhance LLM-Based Translation? An Empirical Analysis<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.20059<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe Perfect Blend: Redefining RLHF with Mixture of Judges<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2409.20370<\/p>\n<p><strong>\u5341\u6708\u8bba\u6587<\/strong><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAddition is All You Need for Energy-efficient Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.00907<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aQuantifying Generalization Complexity for Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.01769<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWhen a language model is optimized for reasoning, does it still show embers of autoregression? An analysis of OpenAI o1<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.01792<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650938369&amp;idx=3&amp;sn=c0fc8dd15e96a254baaec4d834e75ca3&amp;scene=21#wechat_redirect\" target=\"_blank\">Were RNNs All We Needed?<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.01201<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSelective Attention Improves Transformer<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.02703<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLMs Know More Than They Show: On the Intrinsic Representation of LLM Hallucinations<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.02707<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650938369&amp;idx=4&amp;sn=ee475dc1e7f2180a303d8a09e4cdaa23&amp;scene=21#wechat_redirect\" target=\"_blank\">LLaVA-Critic: Learning to Evaluate Multimodal Models<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.02712<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDifferential Transformer<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.05258<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aGSM-Symbolic: Understanding the Limitations of Mathematical Reasoning in Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.05229<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aARIA: An Open Multimodal Native Mixture-of-Experts Model<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.05993<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650937232&amp;idx=1&amp;sn=b3e99fc44adaf14fe85b47d52c4ff03f&amp;scene=21#wechat_redirect\" target=\"_blank\">O1 Replication Journey: A Strategic Progress Report \u2013 Part 1<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.18982<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLong-Context LLMs Meet RAG: Overcoming Challenges for Long Inputs in RAG<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.05983<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aFrom Generalist to Specialist: Adapting Vision Language Models via Task-Specific Visual Instruction Tuning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.06456<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aKV Prediction for Improved Time to First Token<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.08391<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBaichuan-Omni Technical Report<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.08565<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMMIE: Massive Multimodal Interleaved Comprehension Benchmark for Large Vision-Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.10139<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLOKI: A Comprehensive Synthetic Data Detection Benchmark using Large Multimodal Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.09732<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAFlow: Automating Agentic Workflow Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.10762<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aToward General Instruction-Following Alignment for Retrieval-Augmented Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.09584<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPre-training Distillation for Large Language Models: A Design Space Exploration<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.16215<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMIA-DPO: Multi-Image Augmented Direct Preference Optimization For Large Vision-Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.17637<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScalable Ranked Preference Optimization for Text-to-Image Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.18013<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScaling Diffusion Language Models via Adaptation from Autoregressive Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.17891<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aHybrid Preferences: Learning to Route Instances for Human vs. AI Feedback<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.19133<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCounting Ability of Large Language Models and Impact of Tokenization<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.19730<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Survey of Small Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.20011<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAccelerating Direct Preference Optimization with Prefix Sharing<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.20305<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMind Your Step (by Step): Chain-of-Thought can Reduce Performance on Tasks where Thinking Makes Humans Worse<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.21333<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLongReward: Improving Long-context Large Language Models with AI Feedback<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.21252<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aShadowKV: KV Cache in Shadows for High-Throughput Long-Context LLM Inference<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.21465<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBeyond Text: Optimizing RAG with Multimodal Inputs for Industrial Applications<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.21943<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCORAL: Benchmarking Multi-turn Conversational Retrieval-Augmentation Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.23090<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWhat Happened in LLMs Layers when Trained for Fast vs. Slow Thinking: A Gradient Perspective<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.23743<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aGPT or BERT: why not both?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.24159<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLanguage Models can Self-Lengthen to Generate Long Texts<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2410.23933<\/p>\n<p><strong>\u5341\u4e00\u6708\u8bba\u6587<\/strong><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAdding Error Bars to Evals: A Statistical Approach to Language Model Evaluations<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.00640<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAdapting While Learning: Grounding LLMs for Scientific Problems with Intelligent Tool Usage Adaptation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.00412<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMulti-expert Prompting Improves Reliability, Safety, and Usefulness of Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.00492<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSample-Efficient Alignment for LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.01493<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Comprehensive Survey of Small Language Models in the Era of Large Language Models: Techniques, Enhancements, Applications, Collaboration with LLMs, and Trustworthiness<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.03350<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a&#8221;Give Me BF16 or Give Me Death&#8221;? Accuracy-Performance Trade-Offs in LLM Quantization<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.02355<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aParameter-Efficient Fine-Tuning of Large Language Models for Unit Test Generation: An Empirical Study<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.02462<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aHtmlRAG: HTML is Better Than Plain Text for Modeling Retrieved Knowledge in RAG Systems<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.02959<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBoth Text and Images Leaked! A Systematic Analysis of Multimodal LLM Data Contamination<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.03823<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLanguage Models are Hidden Reasoners: Unlocking Latent Reasoning Capabilities via Self-Rewarding<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.04282<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aNumber Cookbook: Number Understanding of Language Models and How to Improve It<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.03766<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMixture-of-Transformers: A Sparse and Scalable Architecture for Multi-Modal Foundation Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.04996<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBitNet a4.8: 4-bit Activations for 1-bit LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.04965<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScaling Laws for Precision<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.04330<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEnergy Efficient Protein Language Models: Leveraging Small Language Models with LoRA for Controllable Protein Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.05966<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBalancing Pipeline Parallelism with Vocabulary Parallelism<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.05288<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aToward Optimal Search and Retrieval for RAG<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.07396<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLarge Language Models Can Self-Improve in Long-context Reasoning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.08147<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aStronger Models are NOT Stronger Teachers for Instruction Tuning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.07133<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDirect Preference Optimization Using Sparse Feature-Level Constraints<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.07618<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCut Your Losses in Large-Vocabulary Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.09009<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDoes Prompt Formatting Have Any Impact on LLM Performance?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.10541<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSymDPO: Boosting In-Context Learning of Large Multimodal Models with Symbol Demonstration Direct Preference Optimization<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.11909<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650949183&amp;idx=5&amp;sn=35f18a03be42b0006c83de4d150cb906&amp;scene=21#wechat_redirect\" target=\"_blank\">SageAttention2 Technical Report: Accurate 4 Bit Attention for Plug-and-play Inference Acceleration<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.10958<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBi-Mamba: Towards Accurate 1-Bit State Space Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.11843<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRedPajama: an Open Dataset for Training Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.12372<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aHymba: A Hybrid-head Architecture for Small Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.13676<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLoss-to-Loss Prediction: Scaling Laws for All Datasets<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.12925<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aWhen Precision Meets Position: BFloat16 Breaks Down RoPE in Long-Context Training<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.13476<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMultimodal Autoregressive Pre-training of Large Vision Encoders<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.14402<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aNatural Language Reinforcement Learning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.14251<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLarge Multi-modal Models Can Interpret Features in Large Multi-modal Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.14982<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1a<a data-itemshowtype=\"0\" data-linktype=\"2\" href=\"https:\/\/mp.weixin.qq.com\/s?__biz=MzA3MzI4MjgzMw==&amp;mid=2650944022&amp;idx=1&amp;sn=c9ab438daa0a315f9a395878a317edfe&amp;scene=21#wechat_redirect\" target=\"_blank\">T\u00dcLU 3: Pushing Frontiers in Open Language Model Post-Training<\/a><\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.15124<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.15296<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLMs Do Not Think Step-by-step In Implicit Reasoning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.15862<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aO1 Replication Journey \u2013 Part 2: Surpassing O1-preview through Simple Distillation, Big Progress or Bitter Lesson?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.16489<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aStar Attention: Efficient LLM Inference over Long Sequences<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.17116<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLow-Bit Quantization Favors Undertrained LLMs: Scaling Laws for Quantized LLMs with 100T Training Tokens<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.17691<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRethinking Token Reduction in MLLMs: Towards a Unified Paradigm for Training-Free Acceleration<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.17686<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aReverse Thinking Makes LLMs Stronger Reasoners<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.19865<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCritical Tokens Matter: Token-Level Contrastive Estimation Enhances LLM&#8217;s Reasoning Capability<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2411.19943<\/p>\n<p><strong>\u5341\u4e8c\u6708\u8bba\u6587<\/strong><\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDesigning Scale-Wise Transformers for Text-to-Image Synthesis<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.01819<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aX-Prompt: Towards Universal In-Context Image Generation in Auto-Regressive Vision Language Foundation Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.01824<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aFree Process Rewards without Process Labels<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.01981<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aScaling Image Tokenizers with Grouped Spherical Quantization<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.02632<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aRARE: Retrieval-Augmented Reasoning Enhancement for Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.02830<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPerception Tokens Enhance Visual Reasoning in Multimodal Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.03548<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEvaluating Language Models as Synthetic Data Generators<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.03679<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aBest-of-N Jailbreaking<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.03556<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPaliGemma 2: A Family of Versatile VLMs for Transfer<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.03555<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aVisionZip: Longer is Better but Not Necessary in Vision Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.04467<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aEvaluating and Aligning CodeLLMs on Human Preference<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.05210<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMAmmoTH-VL: Eliciting Multimodal Reasoning with Instruction Tuning at Scale<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.05237<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aExpanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.05271<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.05579<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDoes RLHF Scale? Exploring the Impacts From Data, Model, and Method<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.06000<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aUnraveling the Complexity of Memory in RL Agents: An Approach for Classification and Evaluation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.06531<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aTraining Large Language Models to Reason in a Continuous Latent Space<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.06769<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAutoReason: Automatic Few-Shot Reasoning Decomposition<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.06975<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLarge Concept Models: Language Modeling in a Sentence Representation Space<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.08821<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPhi-4 Technical Report<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.08905<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aByte Latent Transformer: Patches Scale Better Than Tokens<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.09871<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSCBench: A KV Cache-Centric Analysis of Long-Context Methods<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.10319<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aCultural Evolution of Cooperation among LLM Agents<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.10270<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aDeepSeek-VL2: Mixture-of-Experts Vision-Language Models for Advanced Multimodal Understanding<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.10302<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aNo More Adam: Learning Rate Scaling at Initialization is All You Need<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.11768<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aPrecise Length Control in Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.11937<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aThe Open Source Advantage in Large Language Models (LLMs)<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.12004<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aA Survey of Mathematical Reasoning in the Era of Multimodal Large Language Model: Benchmark, Method &amp; Challenges<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.11936<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAre Your LLMs Capable of Stable Reasoning?<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.13147<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLLM Post-Training Recipes, Improving Reasoning in LLMs<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.14135<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aHansel: Output Length Controlling Framework for Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.14033<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMind Your Theory: Theory of Mind Goes Deeper Than Reasoning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.1363<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aAlignment Faking in Large Language Models<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.14093<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aSCOPE: Optimizing Key-Value Cache Compression in Long-Context Generation<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.13649<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aLongBench v2: Towards Deeper Understanding and Reasoning on Realistic Long-Context Multitasks<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.15204<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aOffline Reinforcement Learning for LLM Multi-Step Reasoning<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.16145<\/p>\n<p>\u8bba\u6587\u6807\u9898\uff1aMulberry: Empowering MLLM with O1-like Reasoning and Reflection via Collective Monte Carlo Tree Search<\/p>\n<p>\u8bba\u6587\u94fe\u63a5\uff1ahttps:\/\/arxiv.org\/abs\/2412.18319<\/p>\n<p>\u6587\u7ae0\u6765\u6e90\u4e8e\u4e92\u8054\u7f51:<a href=\"https:\/\/www.jiqizhixin.com\/articles\/2025-01-01-2\" target=\"_blank\">\u6bcf\u6708\u90fd\u6709\u91cd\u78c5\u7814\u7a76\uff0c2024\u5168\u5e74\u503c\u5f97\u4e00\u8bfb\u7684\u8bba\u6587\u90fd\u5728\u8fd9\u4e86<\/a><\/p>\n","protected":false},"excerpt":{"rendered":"<p>\u6587\u7ae0\u6765\u6e90\u4e8e\u4e92\u8054\u7f51:\u6bcf\u6708\u90fd\u6709\u91cd\u78c5\u7814\u7a76\uff0c20 [&hellip;]<\/p>\n","protected":false},"author":3,"featured_media":0,"comment_status":"open","ping_status":"","sticky":false,"template":"","format":"standard","meta":{"site-sidebar-layout":"default","site-content-layout":"","ast-site-content-layout":"","site-content-style":"default","site-sidebar-style":"default","ast-global-header-display":"","ast-banner-title-visibility":"","ast-main-header-display":"","ast-hfb-above-header-display":"","ast-hfb-below-header-display":"","ast-hfb-mobile-header-display":"","site-post-title":"","ast-breadcrumbs-content":"","ast-featured-img":"","footer-sml-layout":"","theme-transparent-header-meta":"","adv-header-id-meta":"","stick-header-meta":"","header-above-stick-meta":"","header-main-stick-meta":"","header-below-stick-meta":"","astra-migrate-meta-layouts":"default","ast-page-background-enabled":"default","ast-page-background-meta":{"desktop":{"background-color":"var(--ast-global-color-4)","background-image":"","background-repeat":"repeat","background-position":"center center","background-size":"auto","background-attachment":"scroll","background-type":"","background-media":"","overlay-type":"","overlay-color":"","overlay-opacity":"","overlay-gradient":""},"tablet":{"background-color":"","background-image":"","background-repeat":"repeat","background-position":"center center","background-size":"auto","background-attachment":"scroll","background-type":"","background-media":"","overlay-type":"","overlay-color":"","overlay-opacity":"","overlay-gradient":""},"mobile":{"background-color":"","background-image":"","background-repeat":"repeat","background-position":"center center","background-size":"auto","background-attachment":"scroll","background-type":"","background-media":"","overlay-type":"","overlay-color":"","overlay-opacity":"","overlay-gradient":""}},"ast-content-background-meta":{"desktop":{"background-color":"var(--ast-global-color-5)","background-image":"","background-repeat":"repeat","background-position":"center center","background-size":"auto","background-attachment":"scroll","background-type":"","background-media":"","overlay-type":"","overlay-color":"","overlay-opacity":"","overlay-gradient":""},"tablet":{"background-color":"var(--ast-global-color-5)","background-image":"","background-repeat":"repeat","background-position":"center center","background-size":"auto","background-attachment":"scroll","background-type":"","background-media":"","overlay-type":"","overlay-color":"","overlay-opacity":"","overlay-gradient":""},"mobile":{"background-color":"var(--ast-global-color-5)","background-image":"","background-repeat":"repeat","background-position":"center center","background-size":"auto","background-attachment":"scroll","background-type":"","background-media":"","overlay-type":"","overlay-color":"","overlay-opacity":"","overlay-gradient":""}},"footnotes":""},"categories":[27],"tags":[70,71,55],"class_list":["post-37540","post","type-post","status-publish","format-standard","hentry","category-news","tag-agent","tag-rag","tag-55"],"yoast_head":"<!-- This site is optimized with the Yoast SEO plugin v27.9 - https:\/\/yoast.com\/product\/yoast-seo-wordpress\/ -->\n<title>\u6bcf\u6708\u90fd\u6709\u91cd\u78c5\u7814\u7a76\uff0c2024\u5168\u5e74\u503c\u5f97\u4e00\u8bfb\u7684\u8bba\u6587\u90fd\u5728\u8fd9\u4e86 - \u4e00\u8d77AI\u6280\u672f<\/title>\n<meta name=\"robots\" content=\"index, follow, max-snippet:-1, max-image-preview:large, max-video-preview:-1\" \/>\n<link rel=\"canonical\" href=\"https:\/\/17aitech.com\/?p=37540\" \/>\n<script type=\"application\/ld+json\" class=\"yoast-schema-graph\">{\"@context\":\"https:\\\/\\\/schema.org\",\"@graph\":[{\"@type\":\"Article\",\"@id\":\"https:\\\/\\\/17aitech.com\\\/?p=37540#article\",\"isPartOf\":{\"@id\":\"https:\\\/\\\/17aitech.com\\\/?p=37540\"},\"author\":{\"name\":\"AI\u5c0f\u52a9\u624b\",\"@id\":\"https:\\\/\\\/17aitech.com\\\/#\\\/schema\\\/person\\\/60225458499e817ae0af73e67e440b9d\"},\"headline\":\"\u6bcf\u6708\u90fd\u6709\u91cd\u78c5\u7814\u7a76\uff0c2024\u5168\u5e74\u503c\u5f97\u4e00\u8bfb\u7684\u8bba\u6587\u90fd\u5728\u8fd9\u4e86\",\"datePublished\":\"2025-01-30T08:01:17+00:00\",\"mainEntityOfPage\":{\"@id\":\"https:\\\/\\\/17aitech.com\\\/?p=37540\"},\"wordCount\":6615,\"commentCount\":0,\"image\":{\"@id\":\"https:\\\/\\\/17aitech.com\\\/?p=37540#primaryimage\"},\"thumbnailUrl\":\"https:\\\/\\\/17aitech.com\\\/wp-content\\\/uploads\\\/2025\\\/01\\\/frc-461b2de4507e35e20473b9f31252bcce.png\",\"keywords\":[\"Agent\",\"RAG\",\"\u673a\u5668\u5b66\u4e60\"],\"articleSection\":[\"\u884c\u4e1a\u8d44\u8baf\"],\"inLanguage\":\"zh-Hans\",\"potentialAction\":[{\"@type\":\"CommentAction\",\"name\":\"Comment\",\"target\":[\"https:\\\/\\\/17aitech.com\\\/?p=37540#respond\"]}]},{\"@type\":\"WebPage\",\"@id\":\"https:\\\/\\\/17aitech.com\\\/?p=37540\",\"url\":\"https:\\\/\\\/17aitech.com\\\/?p=37540\",\"name\":\"\u6bcf\u6708\u90fd\u6709\u91cd\u78c5\u7814\u7a76\uff0c2024\u5168\u5e74\u503c\u5f97\u4e00\u8bfb\u7684\u8bba\u6587\u90fd\u5728\u8fd9\u4e86 - \u4e00\u8d77AI\u6280\u672f\",\"isPartOf\":{\"@id\":\"https:\\\/\\\/17aitech.com\\\/#website\"},\"primaryImageOfPage\":{\"@id\":\"https:\\\/\\\/17aitech.com\\\/?p=37540#primaryimage\"},\"image\":{\"@id\":\"https:\\\/\\\/17aitech.com\\\/?p=37540#primaryimage\"},\"thumbnailUrl\":\"https:\\\/\\\/17aitech.com\\\/wp-content\\\/uploads\\\/2025\\\/01\\\/frc-461b2de4507e35e20473b9f31252bcce.png\",\"datePublished\":\"2025-01-30T08:01:17+00:00\",\"author\":{\"@id\":\"https:\\\/\\\/17aitech.com\\\/#\\\/schema\\\/person\\\/60225458499e817ae0af73e67e440b9d\"},\"breadcrumb\":{\"@id\":\"https:\\\/\\\/17aitech.com\\\/?p=37540#breadcrumb\"},\"inLanguage\":\"zh-Hans\",\"potentialAction\":[{\"@type\":\"ReadAction\",\"target\":[\"https:\\\/\\\/17aitech.com\\\/?p=37540\"]}]},{\"@type\":\"ImageObject\",\"inLanguage\":\"zh-Hans\",\"@id\":\"https:\\\/\\\/17aitech.com\\\/?p=37540#primaryimage\",\"url\":\"https:\\\/\\\/17aitech.com\\\/wp-content\\\/uploads\\\/2025\\\/01\\\/frc-461b2de4507e35e20473b9f31252bcce.png\",\"contentUrl\":\"https:\\\/\\\/17aitech.com\\\/wp-content\\\/uploads\\\/2025\\\/01\\\/frc-461b2de4507e35e20473b9f31252bcce.png\"},{\"@type\":\"BreadcrumbList\",\"@id\":\"https:\\\/\\\/17aitech.com\\\/?p=37540#breadcrumb\",\"itemListElement\":[{\"@type\":\"ListItem\",\"position\":1,\"name\":\"\u9996\u9875\",\"item\":\"https:\\\/\\\/17aitech.com\\\/\"},{\"@type\":\"ListItem\",\"position\":2,\"name\":\"\u6bcf\u6708\u90fd\u6709\u91cd\u78c5\u7814\u7a76\uff0c2024\u5168\u5e74\u503c\u5f97\u4e00\u8bfb\u7684\u8bba\u6587\u90fd\u5728\u8fd9\u4e86\"}]},{\"@type\":\"WebSite\",\"@id\":\"https:\\\/\\\/17aitech.com\\\/#website\",\"url\":\"https:\\\/\\\/17aitech.com\\\/\",\"name\":\"\u4e00\u8d77AI\u6280\u672f\",\"description\":\"\u8ba9AI\u77e5\u8bc6\u89e6\u624b\u53ef\u53ca\",\"alternateName\":\"\u4e00\u8d77AI\u6280\u672f\",\"potentialAction\":[{\"@type\":\"SearchAction\",\"target\":{\"@type\":\"EntryPoint\",\"urlTemplate\":\"https:\\\/\\\/17aitech.com\\\/?s={search_term_string}\"},\"query-input\":{\"@type\":\"PropertyValueSpecification\",\"valueRequired\":true,\"valueName\":\"search_term_string\"}}],\"inLanguage\":\"zh-Hans\"},{\"@type\":\"Person\",\"@id\":\"https:\\\/\\\/17aitech.com\\\/#\\\/schema\\\/person\\\/60225458499e817ae0af73e67e440b9d\",\"name\":\"AI\u5c0f\u52a9\u624b\",\"image\":{\"@type\":\"ImageObject\",\"inLanguage\":\"zh-Hans\",\"@id\":\"\\\/\\\/17aitech.com\\\/wp-content\\\/uploads\\\/2024\\\/04\\\/robot_3.png\",\"url\":\"\\\/\\\/17aitech.com\\\/wp-content\\\/uploads\\\/2024\\\/04\\\/robot_3.png\",\"contentUrl\":\"\\\/\\\/17aitech.com\\\/wp-content\\\/uploads\\\/2024\\\/04\\\/robot_3.png\",\"caption\":\"AI\u5c0f\u52a9\u624b\"},\"description\":\"\u8fd9\u4e2a\u4eba\u5f88\u61d2\uff0c\u4ec0\u4e48\u90fd\u6ca1\u6709\u7559\u4e0b\uff5e\",\"url\":\"https:\\\/\\\/17aitech.com\\\/?page_id=33738&user=3\"}]}<\/script>\n<!-- \/ Yoast SEO plugin. -->","yoast_head_json":{"title":"\u6bcf\u6708\u90fd\u6709\u91cd\u78c5\u7814\u7a76\uff0c2024\u5168\u5e74\u503c\u5f97\u4e00\u8bfb\u7684\u8bba\u6587\u90fd\u5728\u8fd9\u4e86 - \u4e00\u8d77AI\u6280\u672f","robots":{"index":"index","follow":"follow","max-snippet":"max-snippet:-1","max-image-preview":"max-image-preview:large","max-video-preview":"max-video-preview:-1"},"canonical":"https:\/\/17aitech.com\/?p=37540","schema":{"@context":"https:\/\/schema.org","@graph":[{"@type":"Article","@id":"https:\/\/17aitech.com\/?p=37540#article","isPartOf":{"@id":"https:\/\/17aitech.com\/?p=37540"},"author":{"name":"AI\u5c0f\u52a9\u624b","@id":"https:\/\/17aitech.com\/#\/schema\/person\/60225458499e817ae0af73e67e440b9d"},"headline":"\u6bcf\u6708\u90fd\u6709\u91cd\u78c5\u7814\u7a76\uff0c2024\u5168\u5e74\u503c\u5f97\u4e00\u8bfb\u7684\u8bba\u6587\u90fd\u5728\u8fd9\u4e86","datePublished":"2025-01-30T08:01:17+00:00","mainEntityOfPage":{"@id":"https:\/\/17aitech.com\/?p=37540"},"wordCount":6615,"commentCount":0,"image":{"@id":"https:\/\/17aitech.com\/?p=37540#primaryimage"},"thumbnailUrl":"https:\/\/17aitech.com\/wp-content\/uploads\/2025\/01\/frc-461b2de4507e35e20473b9f31252bcce.png","keywords":["Agent","RAG","\u673a\u5668\u5b66\u4e60"],"articleSection":["\u884c\u4e1a\u8d44\u8baf"],"inLanguage":"zh-Hans","potentialAction":[{"@type":"CommentAction","name":"Comment","target":["https:\/\/17aitech.com\/?p=37540#respond"]}]},{"@type":"WebPage","@id":"https:\/\/17aitech.com\/?p=37540","url":"https:\/\/17aitech.com\/?p=37540","name":"\u6bcf\u6708\u90fd\u6709\u91cd\u78c5\u7814\u7a76\uff0c2024\u5168\u5e74\u503c\u5f97\u4e00\u8bfb\u7684\u8bba\u6587\u90fd\u5728\u8fd9\u4e86 - \u4e00\u8d77AI\u6280\u672f","isPartOf":{"@id":"https:\/\/17aitech.com\/#website"},"primaryImageOfPage":{"@id":"https:\/\/17aitech.com\/?p=37540#primaryimage"},"image":{"@id":"https:\/\/17aitech.com\/?p=37540#primaryimage"},"thumbnailUrl":"https:\/\/17aitech.com\/wp-content\/uploads\/2025\/01\/frc-461b2de4507e35e20473b9f31252bcce.png","datePublished":"2025-01-30T08:01:17+00:00","author":{"@id":"https:\/\/17aitech.com\/#\/schema\/person\/60225458499e817ae0af73e67e440b9d"},"breadcrumb":{"@id":"https:\/\/17aitech.com\/?p=37540#breadcrumb"},"inLanguage":"zh-Hans","potentialAction":[{"@type":"ReadAction","target":["https:\/\/17aitech.com\/?p=37540"]}]},{"@type":"ImageObject","inLanguage":"zh-Hans","@id":"https:\/\/17aitech.com\/?p=37540#primaryimage","url":"https:\/\/17aitech.com\/wp-content\/uploads\/2025\/01\/frc-461b2de4507e35e20473b9f31252bcce.png","contentUrl":"https:\/\/17aitech.com\/wp-content\/uploads\/2025\/01\/frc-461b2de4507e35e20473b9f31252bcce.png"},{"@type":"BreadcrumbList","@id":"https:\/\/17aitech.com\/?p=37540#breadcrumb","itemListElement":[{"@type":"ListItem","position":1,"name":"\u9996\u9875","item":"https:\/\/17aitech.com\/"},{"@type":"ListItem","position":2,"name":"\u6bcf\u6708\u90fd\u6709\u91cd\u78c5\u7814\u7a76\uff0c2024\u5168\u5e74\u503c\u5f97\u4e00\u8bfb\u7684\u8bba\u6587\u90fd\u5728\u8fd9\u4e86"}]},{"@type":"WebSite","@id":"https:\/\/17aitech.com\/#website","url":"https:\/\/17aitech.com\/","name":"\u4e00\u8d77AI\u6280\u672f","description":"\u8ba9AI\u77e5\u8bc6\u89e6\u624b\u53ef\u53ca","alternateName":"\u4e00\u8d77AI\u6280\u672f","potentialAction":[{"@type":"SearchAction","target":{"@type":"EntryPoint","urlTemplate":"https:\/\/17aitech.com\/?s={search_term_string}"},"query-input":{"@type":"PropertyValueSpecification","valueRequired":true,"valueName":"search_term_string"}}],"inLanguage":"zh-Hans"},{"@type":"Person","@id":"https:\/\/17aitech.com\/#\/schema\/person\/60225458499e817ae0af73e67e440b9d","name":"AI\u5c0f\u52a9\u624b","image":{"@type":"ImageObject","inLanguage":"zh-Hans","@id":"\/\/17aitech.com\/wp-content\/uploads\/2024\/04\/robot_3.png","url":"\/\/17aitech.com\/wp-content\/uploads\/2024\/04\/robot_3.png","contentUrl":"\/\/17aitech.com\/wp-content\/uploads\/2024\/04\/robot_3.png","caption":"AI\u5c0f\u52a9\u624b"},"description":"\u8fd9\u4e2a\u4eba\u5f88\u61d2\uff0c\u4ec0\u4e48\u90fd\u6ca1\u6709\u7559\u4e0b\uff5e","url":"https:\/\/17aitech.com\/?page_id=33738&user=3"}]}},"_links":{"self":[{"href":"https:\/\/17aitech.com\/index.php?rest_route=\/wp\/v2\/posts\/37540","targetHints":{"allow":["GET"]}}],"collection":[{"href":"https:\/\/17aitech.com\/index.php?rest_route=\/wp\/v2\/posts"}],"about":[{"href":"https:\/\/17aitech.com\/index.php?rest_route=\/wp\/v2\/types\/post"}],"author":[{"embeddable":true,"href":"https:\/\/17aitech.com\/index.php?rest_route=\/wp\/v2\/users\/3"}],"replies":[{"embeddable":true,"href":"https:\/\/17aitech.com\/index.php?rest_route=%2Fwp%2Fv2%2Fcomments&post=37540"}],"version-history":[{"count":0,"href":"https:\/\/17aitech.com\/index.php?rest_route=\/wp\/v2\/posts\/37540\/revisions"}],"wp:attachment":[{"href":"https:\/\/17aitech.com\/index.php?rest_route=%2Fwp%2Fv2%2Fmedia&parent=37540"}],"wp:term":[{"taxonomy":"category","embeddable":true,"href":"https:\/\/17aitech.com\/index.php?rest_route=%2Fwp%2Fv2%2Fcategories&post=37540"},{"taxonomy":"post_tag","embeddable":true,"href":"https:\/\/17aitech.com\/index.php?rest_route=%2Fwp%2Fv2%2Ftags&post=37540"}],"curies":[{"name":"wp","href":"https:\/\/api.w.org\/{rel}","templated":true}]}}