[{"data":1,"prerenderedAt":317},["ShallowReactive",2],{"content-query-rThhHT0MLo":3},{"_path":4,"_dir":5,"_draft":6,"_partial":6,"_locale":7,"title":8,"description":9,"date":10,"cover":11,"type":12,"category":13,"body":14,"_type":311,"_id":312,"_source":313,"_file":314,"_stem":315,"_extension":316},"\u002Ftechnology-blogs\u002Fzh\u002F2026-9-13","zh",false,"","MindSpore HyperParallel一键开箱DeepSeek-V4.1-Flash","新模型采用全新的Causal Encoder-Decoder架构，在模型能力、推理速度和吞吐量等方面进一步提升，并具备原生多模态视觉理解能力。","2026-9-13","https:\u002F\u002Fobs-mindspore-file.obs.cn-north-4.myhuaweicloud.com\u002Ffile\u002F2024\u002F11\u002F28\u002F8e0e0150508a4c5ba4287fa3bec8ea3f.png","technology-blogs","技术解读",{"type":15,"children":16,"toc":299},"root",[17,25,32,37,42,53,59,75,80,85,90,95,100,106,111,118,123,133,139,151,159,165,170,178,184,189,196,201,207,212,219,225,230,235,242,247,254,259,264,271,277,282,287,294],{"type":18,"tag":19,"props":20,"children":21},"element","p",{},[22],{"type":23,"value":24},"text","9月10日，DeepSeek正式发布DeepSeek-V4.1-Flash。新模型采用全新的Causal Encoder-Decoder架构，在模型能力、推理速度和吞吐量等方面进一步提升，并具备原生多模态视觉理解能力。\nMindSpore HyperParallel基于Hugging Face同步完成DeepSeek-V4.1-Flash核心训练路径适配，实现快速适配。",{"type":18,"tag":26,"props":27,"children":29},"h1",{"id":28},"_01-模型介绍",[30],{"type":23,"value":31},"01 模型介绍",{"type":18,"tag":19,"props":33,"children":34},{},[35],{"type":23,"value":36},"DeepSeek-V4.1-Flash是DeepSeek最新发布的多模态MoE模型，拥有552B Backbone参数和196B Engram参数，支持最长1M tokens的上下文，并原生支持图像和文本输入。每个MoE层配置1个Shared Expert和384个Routed Experts，每个Token激活6个Routed Experts。",{"type":18,"tag":19,"props":38,"children":39},{},[40],{"type":23,"value":41},"模型采用全新的Causal Encoder-Decoder架构，Language Backbone共40层，由20层Causal Encoder和20层Decoder组成。Prefill阶段每个Token激活8B参数，Decode阶段每个Token激活16B参数，在保持大模型容量的同时降低计算成本。",{"type":18,"tag":43,"props":44,"children":46},"div",{"style":45},"text-align: center;",[47],{"type":18,"tag":48,"props":49,"children":52},"img",{"src":50,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-13\u002F1.jpg","display: block;margin: 0 auto;max-width:60%",[],{"type":18,"tag":26,"props":54,"children":56},{"id":55},"_02-deepseek-v41-flash快速适配",[57],{"type":23,"value":58},"02 DeepSeek-V4.1-Flash快速适配",{"type":18,"tag":19,"props":60,"children":61},{},[62,64,73],{"type":23,"value":63},"HyperParallel基于DeepSeek官方开源模型（",{"type":18,"tag":65,"props":66,"children":70},"a",{"href":67,"rel":68},"https:\u002F\u002Fhuggingface.co\u002Fdeepseek-ai\u002FDeepSeek-V4.1-Flash%EF%BC%89%EF%BC%8C%E9%80%9A%E8%BF%87Model",[69],"nofollow",[71],{"type":23,"value":72},"https:\u002F\u002Fhuggingface.co\u002Fdeepseek-ai\u002FDeepSeek-V4.1-Flash），通过Model",{"type":23,"value":74}," Adapter接入FSDP、TP、CP和EP等多维并行能力，并复用通用Trainer、数据处理组件及Muon + AdamW优化器链路，完成DeepSeek-V4.1-Flash核心训练路径适配。",{"type":18,"tag":19,"props":76,"children":77},{},[78],{"type":23,"value":79},"HyperParallel通过声明式Model Adapter定义模型适配规则，由Sharding Planner结合模型结构与并行配置生成切分方案，使新模型能够复用已有并行训练能力，减少重复开发，支撑快速适配。",{"type":18,"tag":19,"props":81,"children":82},{},[83],{"type":23,"value":84},"本次适配主要包括：",{"type":18,"tag":19,"props":86,"children":87},{},[88],{"type":23,"value":89},"接入DeepSeek-ViT、3×3 pixel-unshuffle、两层Aligner与图文路由对应的多模态训练链路；",{"type":18,"tag":19,"props":91,"children":92},{},[93],{"type":23,"value":94},"为Single-Pass mHC的跨子层状态传递、Engram分布式查表以及CSA2的Full \u002F Reindex \u002F Reuse模式补充并行适配；",{"type":18,"tag":19,"props":96,"children":97},{},[98],{"type":23,"value":99},"通过声明式并行计划组织TP、CP、EP与FSDP，打通Online多模态数据处理、模型前向、反向传播及优化器更新流程。",{"type":18,"tag":26,"props":101,"children":103},{"id":102},"_03-快速开始",[104],{"type":23,"value":105},"03 快速开始",{"type":18,"tag":19,"props":107,"children":108},{},[109],{"type":23,"value":110},"HyperParallel提供DeepSeek-V4.1-Flash轻量化训练验证样例。完成HyperParallel环境安装并准备好模型配置、Tokenizer和多模态数据后，即可启动训练。",{"type":18,"tag":112,"props":113,"children":115},"h2",{"id":114},"_31-获取并安装适配代码",[116],{"type":23,"value":117},"3.1 获取并安装适配代码",{"type":18,"tag":19,"props":119,"children":120},{},[121],{"type":23,"value":122},"DeepSeek-V4.1-Flash适配样例当前位于deepseek-v4.1-preview分支。克隆该分支并安装HyperParallel：",{"type":18,"tag":124,"props":125,"children":127},"pre",{"code":126},"git clone -b deepseek-v4.1-preview https:\u002F\u002Fgitcode.com\u002Fmindspore\u002Fhyper-parallel.git\ncd hyper-parallel\npip install -e .\n",[128],{"type":18,"tag":129,"props":130,"children":131},"code",{"__ignoreMap":7},[132],{"type":23,"value":126},{"type":18,"tag":112,"props":134,"children":136},{"id":135},"_32-准备模型配置",[137],{"type":23,"value":138},"3.2 准备模型配置",{"type":18,"tag":19,"props":140,"children":141},{},[142,144],{"type":23,"value":143},"从DeepSeek Hugging Face官方仓库（",{"type":18,"tag":65,"props":145,"children":148},{"href":146,"rel":147},"https:\u002F\u002Fhuggingface.co\u002Fdeepseek-ai\u002FDeepSeek-V4.1-Flash%EF%BC%89%E8%8E%B7%E5%8F%96config.json%E3%80%81tokenizer.json%E5%92%8Ctokenizer_config.json%EF%BC%8C%E5%B9%B6%E5%B0%86MODEL_PATH%E8%AE%BE%E7%BD%AE%E4%B8%BA%E6%96%87%E4%BB%B6%E6%89%80%E5%9C%A8%E7%9B%AE%E5%BD%95%E3%80%82%E8%AF%A5%E9%AA%8C%E8%AF%81%E6%A0%B7%E4%BE%8B%E9%87%87%E7%94%A8%E9%9A%8F%E6%9C%BA%E5%88%9D%E5%A7%8B%E5%8C%96%E7%9A%84%E6%A8%A1%E5%9E%8B%E6%9D%83%E9%87%8D%EF%BC%8C%E6%97%A0%E9%9C%80%E4%B8%8B%E8%BD%BD%E5%AE%8C%E6%95%B4%E6%A8%A1%E5%9E%8B%E6%9D%83%E9%87%8D%E3%80%82",[69],[149],{"type":23,"value":150},"https:\u002F\u002Fhuggingface.co\u002Fdeepseek-ai\u002FDeepSeek-V4.1-Flash）获取config.json、tokenizer.json和tokenizer_config.json，并将MODEL_PATH设置为文件所在目录。该验证样例采用随机初始化的模型权重，无需下载完整模型权重。",{"type":18,"tag":124,"props":152,"children":154},{"code":153},"MODEL_PATH=\u002Fpath\u002Fto\u002FDeepSeek-V4.1-Flash\n",[155],{"type":18,"tag":129,"props":156,"children":157},{"__ignoreMap":7},[158],{"type":23,"value":153},{"type":18,"tag":112,"props":160,"children":162},{"id":161},"_33-准备多模态数据",[163],{"type":23,"value":164},"3.3 准备多模态数据",{"type":18,"tag":19,"props":166,"children":167},{},[168],{"type":23,"value":169},"训练数据采用OpenAI Messages格式的JSONL文件，图像通过image_url指定。将DATA_PATH设置为训练数据文件路径，并确保其中引用的图像可访问。",{"type":18,"tag":124,"props":171,"children":173},{"code":172},"DATA_PATH=\u002Fpath\u002Fto\u002Fdeepseek_v41_messages\u002Ftrain.jsonl\n",[174],{"type":18,"tag":129,"props":175,"children":176},{"__ignoreMap":7},[177],{"type":23,"value":172},{"type":18,"tag":112,"props":179,"children":181},{"id":180},"_34-确认训练配置",[182],{"type":23,"value":183},"3.4 确认训练配置",{"type":18,"tag":19,"props":185,"children":186},{},[187],{"type":23,"value":188},"样例默认使用16个die，序列长度为4K。发布版配置与本次轻量化验证配置对比如下：",{"type":18,"tag":43,"props":190,"children":191},{"style":45},[192],{"type":18,"tag":48,"props":193,"children":195},{"src":194,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-13\u002F2.jpg",[],{"type":18,"tag":19,"props":197,"children":198},{},[199],{"type":23,"value":200},"启动进程数应与设备网格及TP、CP、EP、FSDP等并行配置相匹配；调整并行策略时，应同步检查启动参数与对应配置。",{"type":18,"tag":112,"props":202,"children":204},{"id":203},"_35-启动训练",[205],{"type":23,"value":206},"3.5 启动训练",{"type":18,"tag":19,"props":208,"children":209},{},[210],{"type":23,"value":211},"本样例基于以下环境完成验证：",{"type":18,"tag":43,"props":213,"children":214},{"style":45},[215],{"type":18,"tag":48,"props":216,"children":218},{"src":217,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-13\u002F3.jpg",[],{"type":18,"tag":26,"props":220,"children":222},{"id":221},"_04-hyperparallel训练性能实测",[223],{"type":23,"value":224},"04 HyperParallel训练性能实测",{"type":18,"tag":19,"props":226,"children":227},{},[228],{"type":23,"value":229},"HyperParallel基于Atlas 800T A3，在16个die上完成了DeepSeek-V4.1-Flash轻量化多模态模型连续100 Step的训练验证，覆盖DeepSeek-ViT、3×3 pixel-unshuffle、两层Aligner、Single-Pass mHC、Engram、CSA2、图文路由以及Online数据处理链路。",{"type":18,"tag":112,"props":231,"children":233},{"id":232},"实验配置",[234],{"type":23,"value":232},{"type":18,"tag":43,"props":236,"children":237},{"style":45},[238],{"type":18,"tag":48,"props":239,"children":241},{"src":240,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-13\u002F4.jpg",[],{"type":18,"tag":112,"props":243,"children":245},{"id":244},"训练性能",[246],{"type":23,"value":244},{"type":18,"tag":43,"props":248,"children":249},{"style":45},[250],{"type":18,"tag":48,"props":251,"children":253},{"src":252,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-13\u002F5.jpg",[],{"type":18,"tag":19,"props":255,"children":256},{},[257],{"type":23,"value":258},"在TP1 + CP1 + EP16 + FSDP16并行策略下，训练连续完成100 Step，loss和grad norm均未出现NaN或Inf。稳态平均单步耗时为4.9602 s，按全局Batch Size 16、序列长度4096折算的全局吞吐约为13212 tokens\u002Fs；100 Step纯训练总耗时为504.47 s。",{"type":18,"tag":112,"props":260,"children":262},{"id":261},"多维并行能力",[263],{"type":23,"value":261},{"type":18,"tag":43,"props":265,"children":266},{"style":45},[267],{"type":18,"tag":48,"props":268,"children":270},{"src":269,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-13\u002F6.jpg",[],{"type":18,"tag":26,"props":272,"children":274},{"id":273},"_05-总-结",[275],{"type":23,"value":276},"05 总 结",{"type":18,"tag":19,"props":278,"children":279},{},[280],{"type":23,"value":281},"HyperParallel通过Model Adapter、高性能组件和多维并行能力，完成DeepSeek-V4.1-Flash核心训练路径适配，并在Atlas 800T A3上使用16个die完成轻量化模型训练验证。",{"type":18,"tag":19,"props":283,"children":284},{},[285],{"type":23,"value":286},"下一步，HyperParallel将推进DeepSeek-V4.1-Flash完整规模模型的训练适配与验证，并持续跟进前沿模型架构与训练技术，进一步优化训练性能、显存效率和大规模扩展能力。",{"type":18,"tag":43,"props":288,"children":289},{"style":45},[290],{"type":18,"tag":48,"props":291,"children":293},{"src":292,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-13\u002F7.jpg",[],{"type":18,"tag":19,"props":295,"children":296},{},[297],{"type":23,"value":298},"无论是代码实现、文档完善、示例补充还是Bug反馈，您的每一份贡献都将帮助更多研究者和开发者受益。",{"title":7,"searchDepth":300,"depth":300,"links":301},4,[302,304,305,306,307,308,309,310],{"id":114,"depth":303,"text":117},2,{"id":135,"depth":303,"text":138},{"id":161,"depth":303,"text":164},{"id":180,"depth":303,"text":183},{"id":203,"depth":303,"text":206},{"id":232,"depth":303,"text":232},{"id":244,"depth":303,"text":244},{"id":261,"depth":303,"text":261},"markdown","content:technology-blogs:zh:2026-9-13.md","content","technology-blogs\u002Fzh\u002F2026-9-13.md","technology-blogs\u002Fzh\u002F2026-9-13","md",1789904230807]