[{"data":1,"prerenderedAt":138},["ShallowReactive",2],{"content-query-Tu0thtRaQa":3},{"_path":4,"_dir":5,"_draft":6,"_partial":6,"_locale":7,"title":8,"description":9,"date":10,"cover":11,"type":12,"body":13,"_type":132,"_id":133,"_source":134,"_file":135,"_stem":136,"_extension":137},"\u002Fnews\u002Fzh\u002F2026-9-9","zh",false,"","HyperParallel亮相PTC 2026：专为超节点打造的分布式并行组件","介绍了专为超节点打造的分布式并行组件——HyperParallel，展示了HyperParallel在昇腾Atlas 800 A3风冷超节点上的卓越训练加速能力。","2026-9-4","https:\u002F\u002Fobs-mindspore-file.obs.cn-north-4.myhuaweicloud.com\u002Ffile\u002F2025\u002F07\u002F25\u002F199b735845bf4106b44b2035dc97bd39.png","news",{"type":14,"children":15,"toc":129},"root",[16,30,41,48,53,60,66,71,77,82,87,92,99,105,110,117,122],{"type":17,"tag":18,"props":19,"children":20},"element","p",{},[21,28],{"type":17,"tag":22,"props":23,"children":24},"span",{},[25],{"type":26,"value":27},"text","中国，上海，2026年9月9日",{"type":26,"value":29}," 在 PyTorch Conference China 2026 大会上，华为AI框架首席技术专家苏腾发表主题演讲，介绍了专为超节点打造的分布式并行组件——HyperParallel，并由Linux基金会云与基础设施执行董事、OpenInfra Foundation执行董事兼CNCF执行董事Jonathan Bryce启动Live Demo演示，展示了HyperParallel在昇腾Atlas 800 A3风冷超节点上的卓越训练加速能力。",{"type":17,"tag":31,"props":32,"children":34},"div",{"style":33},"text-align: center;",[35],{"type":17,"tag":36,"props":37,"children":40},"img",{"src":38,"style":39,"alt":7},"\u002Fcategory\u002Finformation\u002Fnews\u002Fbanner\u002F2026-9-9\u002F1.jpg","display: block;margin: 0 auto;max-width:60%",[],{"type":17,"tag":42,"props":43,"children":45},"h1",{"id":44},"从原生支持到生态共建昇腾与pytorch合作迈入新阶段",[46],{"type":26,"value":47},"从原生支持到生态共建：昇腾与PyTorch合作迈入新阶段",{"type":17,"tag":18,"props":49,"children":50},{},[51],{"type":26,"value":52},"首先回顾大会首日KN提到昇腾与PyTorch的生态合作已迈入新里程碑，询问昇腾新成立的Ascend for PyTorch开源社区进展，昇腾AI框架生态负责人王神迪介绍，Ascend for PyTorch社区（基于TorchNPU的中国社区）于今年5月正式成立，目前已汇聚超过1,000名贡献者，外部贡献占比超过50%，是中国发展最快的AI社区之一，社区正持续推动中国AI开发者深度参与PyTorch上游贡献。",{"type":17,"tag":31,"props":54,"children":55},{"style":33},[56],{"type":17,"tag":36,"props":57,"children":59},{"src":58,"style":39,"alt":7},"\u002Fcategory\u002Finformation\u002Fnews\u002Fbanner\u002F2026-9-9\u002F2.jpg",[],{"type":17,"tag":42,"props":61,"children":63},{"id":62},"hyperparallel昇腾超节点亲和加速新模型结构和新训推范式创新",[64],{"type":26,"value":65},"HyperParallel：昇腾超节点亲和，加速新模型结构和新训推范式创新",{"type":17,"tag":18,"props":67,"children":68},{},[69],{"type":26,"value":70},"随着大模型向十万亿参数、全模态、Agent化方向演进，训推共存、Agentic RL等新训练范式持续涌现，为超节点时代的分布式训练带来全新挑战。HyperParallel深度适配昇腾超节点的内存统一编址、超大带宽、超低时延等核心特性，通过硬件感知的编排机制，将超节点视为单一逻辑计算机进行全局调度，能够显著降低并行编程与系统调优开销，同时大幅提升训练效率。\nHyperParallel兼容主流模型生态，无缝对接HuggingFace、LlamaFactory等开源生态库，提供大模型分布式训练全栈能力，并集成混合精度、通信优化、显存管理和DCP，实现从单卡到超大规模分布式训练的平滑扩展。",{"type":17,"tag":42,"props":72,"children":74},{"id":73},"live-demo演示qwen3-30b-a3b训练吞吐量提升超200",[75],{"type":26,"value":76},"Live Demo演示：Qwen3-30B-A3B训练吞吐量提升超200%",{"type":17,"tag":18,"props":78,"children":79},{},[80],{"type":26,"value":81},"Jonathan Bryce现场演示了Live Demo，在昇腾Atlas 800 A3风冷超节点上使用HyperParallel+PyTorch训练Qwen3-30B-A3B模型。在相同配置下，HyperParallel-FSDP2+Muon相比PyTorch-FSDP2+Muon，在保持loss收敛一致性（即模型训练效果完全一致）的前提下，训练吞吐提升幅度超过200%。",{"type":17,"tag":18,"props":83,"children":84},{},[85],{"type":26,"value":86},"这一性能提升主要源于HyperParallel的两大核心优化：\n高性能昇腾亲和module：在保持入参、出参一致的前提下，重写优化\u002F替换包括attention、rmsnorm、muon等模块，如muon优化器通过HSDP&DP去冗余计算，以分组调度匹配超节点层次结构，达成主流模型生态开箱即优的效果。",{"type":17,"tag":18,"props":88,"children":89},{},[90],{"type":26,"value":91},"FSDP2零拷贝通信：支持逐参数通信，避免了分组通信带来的数据拷贝（copy in\u002Fout）开销，实现了计算与通信的高效重叠，充分发挥了超节点的高带宽优势。",{"type":17,"tag":31,"props":93,"children":94},{"style":33},[95],{"type":17,"tag":36,"props":96,"children":98},{"src":97,"style":39,"alt":7},"\u002Fcategory\u002Finformation\u002Fnews\u002Fbanner\u002F2026-9-9\u002F3.jpg",[],{"type":17,"tag":42,"props":100,"children":102},{"id":101},"hyperparallel持续深耕技术创新回馈pytorch开源生态",[103],{"type":26,"value":104},"HyperParallel持续深耕技术创新，回馈PyTorch开源生态",{"type":17,"tag":18,"props":106,"children":107},{},[108],{"type":26,"value":109},"HyperParallel会持续围绕多模态、长序列、高稀疏MoE模型等领域做深度优化，进一步支持torch compile图模式，支持好自动策略搜索、强化学习、世界模型等特性。秉持开放协作的理念，HyperParallel将持续把经过大规模生产验证的创新技术贡献给PyTorch等开源社区，与全球开发者共同推动分布式训练技术的演进。",{"type":17,"tag":31,"props":111,"children":112},{"style":33},[113],{"type":17,"tag":36,"props":114,"children":116},{"src":115,"style":39,"alt":7},"\u002Fcategory\u002Finformation\u002Fnews\u002Fbanner\u002F2026-9-9\u002F4.jpg",[],{"type":17,"tag":18,"props":118,"children":119},{},[120],{"type":26,"value":121},"欢迎全球开发者共同参与HyperParallel关键技术共建！",{"type":17,"tag":31,"props":123,"children":124},{"style":33},[125],{"type":17,"tag":36,"props":126,"children":128},{"src":127,"style":39,"alt":7},"\u002Fcategory\u002Finformation\u002Fnews\u002Fbanner\u002F2026-9-9\u002F5.jpg",[],{"title":7,"searchDepth":130,"depth":130,"links":131},4,[],"markdown","content:news:zh:2026-9-9.md","content","news\u002Fzh\u002F2026-9-9.md","news\u002Fzh\u002F2026-9-9","md",1789162114921]