[{"data":1,"prerenderedAt":428},["ShallowReactive",2],{"content-query-cgEZeoPkvR":3},{"_path":4,"_dir":5,"_draft":6,"_partial":6,"_locale":7,"title":8,"description":9,"date":10,"cover":11,"type":12,"category":13,"body":14,"_type":422,"_id":423,"_source":424,"_file":425,"_stem":426,"_extension":427},"\u002Ftechnology-blogs\u002Fzh\u002F2026-09-23","zh",false,"","昇思 HyperParallel DCP 关键技术拆解：异步持久化、细粒度去冗余、跨切片加载、分片广播","检查点重构为\"规划（Planner）—存储（Storage）—元数据（Metadata）\"三层解耦架构，构建四大关键技术解决以上四大成本，实现零冗余落盘、零阻塞存盘、零重复读盘、零离线转换。","2026-9-23","https:\u002F\u002Fobs-mindspore-file.obs.cn-north-4.myhuaweicloud.com\u002Ffile\u002F2024\u002F11\u002F28\u002F8e0e0150508a4c5ba4287fa3bec8ea3f.png","technology-blogs","技术解读",{"type":15,"children":16,"toc":411},"root",[17,25,30,35,42,53,59,66,71,78,85,90,97,102,109,114,124,129,137,142,148,155,160,165,173,178,184,191,196,204,210,215,223,228,236,243,248,254,260,265,272,278,283,303,309,314,321,327,332,337,342,347,352,357,363,368,373,378,383,388,393,399,404],{"type":18,"tag":19,"props":20,"children":21},"element","p",{},[22],{"type":23,"value":24},"text","千亿至万亿参数模型训练已进入万卡规模，单次 checkpoint 可达数千 GB，断点续训效率成为长稳训练关键瓶颈。传统 checkpoint 存在四大成本：存盘阻塞、落盘冗余、读盘重复、并行策略转换。",{"type":18,"tag":19,"props":26,"children":27},{},[28],{"type":23,"value":29},"昇思 HyperParallel DCP（Distributed Checkpoint） 将检查点重构为\"规划（Planner）—存储（Storage）—元数据（Metadata）\"三层解耦架构，构建四大关键技术解决以上四大成本，实现零冗余落盘、零阻塞存盘、零重复读盘、零离线转换。",{"type":18,"tag":19,"props":31,"children":32},{},[33],{"type":23,"value":34},"本文聚焦原理分析，重点阐述四大关键技术。至于接口如何调用、断点续训如何配置等相关操作详见本系列后续实操篇。",{"type":18,"tag":36,"props":37,"children":39},"h1",{"id":38},"_01-传统-checkpoint-四大成本",[40],{"type":23,"value":41},"01 传统 checkpoint 四大成本",{"type":18,"tag":43,"props":44,"children":46},"div",{"style":45},"text-align: center;",[47],{"type":18,"tag":48,"props":49,"children":52},"img",{"src":50,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-23\u002F1.jpg","display: block;margin: 0 auto;max-width:60%",[],{"type":18,"tag":36,"props":54,"children":56},{"id":55},"_02-hyperparallel-dcp-四大技术",[57],{"type":23,"value":58},"02 HyperParallel DCP 四大技术",{"type":18,"tag":43,"props":60,"children":61},{"style":45},[62],{"type":18,"tag":48,"props":63,"children":65},{"src":64,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-23\u002F2.jpg",[],{"type":18,"tag":19,"props":67,"children":68},{},[69],{"type":23,"value":70},"传统方案类似于“先拼回完整图像，再重新切分”：保存时要先把各卡分片合并成完整张量，加载时再按新并行策略重新切分。DCP 每张卡仅保存自身部分分片，并附加一份全局元数据（.metadata，相当于一张“座位表”），记录每个分片在完整张量中的坐标位置，该元数据即全局契约。并行策略变化时，无需恢复完整张量，只需根据新分片区域查询元数据求交。",{"type":18,"tag":43,"props":72,"children":73},{"style":45},[74],{"type":18,"tag":48,"props":75,"children":77},{"src":76,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-23\u002F3.jpg",[],{"type":18,"tag":79,"props":80,"children":82},"h2",{"id":81},"_21-异步持久化同步-staging-异步落盘",[83],{"type":23,"value":84},"2.1 异步持久化：同步 staging + 异步落盘",{"type":18,"tag":19,"props":86,"children":87},{},[88],{"type":23,"value":89},"build_staged_state_dict 将权重D2H拷贝到Host，之后构建plan、集群协同、写盘全交给子进程。在 async_save返回后，训练就可以继续执行。",{"type":18,"tag":43,"props":91,"children":92},{"style":45},[93],{"type":18,"tag":48,"props":94,"children":96},{"src":95,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-23\u002F4.jpg",[],{"type":18,"tag":19,"props":98,"children":99},{},[100],{"type":23,"value":101},"难点在于子进程做集群协同对齐plan，训练侧HCCL\u002FNCCL 通信域活在 C++ 全局状态里，fork 之后就失效了，子进程不能直接复用用。HyperParallel DCP 为此给了两种协同策略，按环境二选一：",{"type":18,"tag":43,"props":103,"children":104},{"style":45},[105],{"type":18,"tag":48,"props":106,"children":108},{"src":107,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-23\u002F5.jpg",[],{"type":18,"tag":19,"props":110,"children":111},{},[112],{"type":23,"value":113},"第二种尤其实用：集群本来就有共享文件系统，不必再为 checkpoint 单开一个通信域。在每个 pkl 写完后末尾追加 _COMPLETE_FLAG 标记，读端只认带标记的文件，没有标记就先跳过去读下一个，然后轮询重试，从未保证读到文件都是完整可靠的。",{"type":18,"tag":115,"props":116,"children":118},"pre",{"code":117},"# hyper_parallel\u002Fcore\u002Fdistributed_checkpoint\u002Fapi.py:223\nuse_gloo=True         → 重建独立的 CPU gloo 通信域\nuse_gloo=False        → 通过共享文件存储交换 plan 与 write result\nuse_collectives=False → 降级选项：完全不跨卡，每 rank 各写各的 {rank}.metadata\n",[119],{"type":18,"tag":120,"props":121,"children":122},"code",{"__ignoreMap":7},[123],{"type":23,"value":117},{"type":18,"tag":19,"props":125,"children":126},{},[127],{"type":23,"value":128},"子进程还会把 plan 缓存带回主进程。规划结果存在 StandardSavePlanner.cached_save_result 这个类级字典里，可算出它的是子进程，子进程一退出，缓存就跟着没了，下一次 async_save 又得从头 all_gather 一遍。DCP 的做法是让子进程写完盘后，把缓存连同 metadata 一起塞回 result queue，由主进程的 join 线程 merge 进自己的缓存：",{"type":18,"tag":115,"props":130,"children":132},{"code":131},"# hyper_parallel\u002Fcore\u002Fdistributed_checkpoint\u002Fasync_persist.py:389 \u002F 437   子进程侧\nresult_queue.put((AsyncPersistStatus.SUCCESS, (meta, StandardSavePlanner.cached_save_result)))\n\n# hyper_parallel\u002Fcore\u002Fdistributed_checkpoint\u002Fasync_persist.py:480        主进程的 join 线程\nStandardSavePlanner.cached_save_result.update(payload[1])\n",[133],{"type":18,"tag":120,"props":134,"children":135},{"__ignoreMap":7},[136],{"type":23,"value":131},{"type":18,"tag":19,"props":138,"children":139},{},[140],{"type":23,"value":141},"两条协同策略（gloo \u002F 共享文件存储）都会回传。下一次  async_save fork 出的子进程，继承到的就是这份已经填好的缓存。这意味着同结构的checkpoint 只在第一次做规划，之后每次异步\u002F同步保存都复用缓存。",{"type":18,"tag":79,"props":143,"children":145},{"id":144},"_22-细粒度去冗余分片级去重与负载均衡",[146],{"type":23,"value":147},"2.2 细粒度去冗余：分片级去重与负载均衡",{"type":18,"tag":43,"props":149,"children":150},{"style":45},[151],{"type":18,"tag":48,"props":152,"children":154},{"src":153,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-23\u002F6.jpg",[],{"type":18,"tag":19,"props":156,"children":157},{},[158],{"type":23,"value":159},"去重的粒度不是\"张量\"而是分片：MetadataIndex(fqn, offsets) 唯一标识\"哪个参数的哪一块\"，坐标相同的分片按定义就是同一份副本，无论被多少张rank持有，全局只有一个 rank 会写它。DP 复制维度、TP 上的 Replicate()参数、fully_shard 的重叠副本，都只会落盘一份。从而保证checkpoint的体积等于模型参数量本身，而不是参数量 × 复制倍数。",{"type":18,"tag":19,"props":161,"children":162},{},[163],{"type":23,"value":164},"同时去冗余不止是简单丢掉多余副本，它也是一次写侧的负载均衡：",{"type":18,"tag":115,"props":166,"children":168},{"code":167},"# hyper_parallel\u002Fcore\u002Fdistributed_checkpoint\u002Futil.py:216\nfor item_key, containing_plans in multi_plan_items:\n    if save_to_minimum_rank:\n            target_plan = min(containing_plans)\n    else:\n         target_plan = min(containing_plans, key=lambda p_idx: storage_sizes[p_idx])    \n    entry = item_registry[item_key]\n    storage_sizes[target_plan] += entry.tensor_storage_size() or 1    \n    for p_idx in containing_plans - {target_plan}:     # 其余 plan 丢掉这个副本\n        remaining_items[p_idx].discard(item_key)\n",[169],{"type":18,"tag":120,"props":170,"children":171},{"__ignoreMap":7},[172],{"type":23,"value":167},{"type":18,"tag":19,"props":174,"children":175},{},[176],{"type":23,"value":177},"先把\"只有一个 rank 持有\"的分片入账，再把有副本的分片派给当前写入量最小的那个 rank 。如果一律交给最小号rank，副本都会堆到 DP0 rank上，整体写盘时间就会被这部分卡拖住。",{"type":18,"tag":79,"props":179,"children":181},{"id":180},"_23-分片广播最小-rank-读-组内广播",[182],{"type":23,"value":183},"2.3 分片广播：最小 rank 读 + 组内广播",{"type":18,"tag":43,"props":185,"children":186},{"style":45},[187],{"type":18,"tag":48,"props":188,"children":190},{"src":189,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-23\u002F7.jpg",[],{"type":18,"tag":19,"props":192,"children":193},{},[194],{"type":23,"value":195},"去冗余在读侧有一个对偶操作：",{"type":18,"tag":115,"props":197,"children":199},{"code":198},"# hyper_parallel\u002Fcore\u002Fdistributed_checkpoint\u002Fstandard_planner.py:492\ngroup_ranks = infer_same_shard_ranks_for_dtensor(tensor)   # 由 mesh_shape + tensor_map 推同分片组\nload_rank = min(group_ranks)\nif len(group_ranks) > 1:\n    setattr(tensor, BROADCAST_INFO, BroadcastInfo(group_ranks, load_rank))\n    return self.rank == load_rank\nDP=4 的复制维度上，磁盘读量从 4× 降到 1×，由最小号rank从磁盘读，然后广播给组内其他3张rank，用D2D高速通信替换存储IO。\n",[200],{"type":18,"tag":120,"props":201,"children":202},{"__ignoreMap":7},[203],{"type":23,"value":198},{"type":18,"tag":79,"props":205,"children":207},{"id":206},"_24-跨切片加载n-维区间求交",[208],{"type":23,"value":209},"2.4 跨切片加载：N 维区间求交",{"type":18,"tag":19,"props":211,"children":212},{},[213],{"type":23,"value":214},"\"拿着新区域去查座位表\"：每个张量分片都抽象成 N 维半开区间 [start, end)，查表即求交，两个区间交集就是要搬运的那片。load时重切分都归结到这里。",{"type":18,"tag":115,"props":216,"children":218},{"code":217},"# hyper_parallel\u002Fcore\u002Fdistributed_checkpoint\u002Freshard.py:88\ndef infer_intersection(area_a, area_b):\n    intersection = []\n    for axis_range_a, axis_range_b in zip(area_a, area_b):    \n        left = max(axis_range_a[0], axis_range_b[0])    \n        right = min(axis_range_a[1], axis_range_b[1])     \n        if left >= right:          # 任一维不相交 → 整体不相交     \n               return None        \n        intersection.append((left, right))    \n    return tuple(intersection)\n",[219],{"type":18,"tag":120,"props":220,"children":221},{"__ignoreMap":7},[222],{"type":23,"value":217},{"type":18,"tag":19,"props":224,"children":225},{},[226],{"type":23,"value":227},"\"我要的\"与\"磁盘上存的\"求交，算出的偏移量直接变成读请求：",{"type":18,"tag":115,"props":229,"children":231},{"code":230},"# hyper_parallel\u002Fcore\u002Fdistributed_checkpoint\u002Fstandard_planner.py:396\noverlap = infer_intersection(local_area, saved_area)\nif overlap is None:\n    continue\ndest_offsets    = tuple(overlap[i][0] - local_chunk.offsets[i]   for i in range(len(overlap)))\nstorage_offsets = tuple(overlap[i][0] - storage_chunk.offsets[i] for i in range(len(overlap)))\nlengths         = tuple(overlap[i][1] - overlap[i][0]            for i in range(len(overlap)))\n",[232],{"type":18,"tag":120,"props":233,"children":234},{"__ignoreMap":7},[235],{"type":23,"value":230},{"type":18,"tag":43,"props":237,"children":238},{"style":45},[239],{"type":18,"tag":48,"props":240,"children":242},{"src":241,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-23\u002F8.jpg",[],{"type":18,"tag":19,"props":244,"children":245},{},[246],{"type":23,"value":247},"save 端只记录每片全局坐标，load 端只关心\"我的新区域压到了哪些旧片\"。TP8 存、TP4 读，或者 fully_shard 的分片换一个 mesh，走的都是这套设计逻辑。\n除以上四大技术外，HyperParallel DCP还提供 huggingface safetensors ⇄ DCP 离线互转配套工具（offline_transform），用于冷启动加载开源权重、训练结束后导出发布。具体用法见本系列实操篇。",{"type":18,"tag":36,"props":249,"children":251},{"id":250},"_03-万卡集群分钟级恢复",[252],{"type":23,"value":253},"03 万卡集群分钟级恢复",{"type":18,"tag":79,"props":255,"children":257},{"id":256},"_31-恢复时间不随卡数线性增长",[258],{"type":23,"value":259},"3.1 恢复时间不随卡数线性增长",{"type":18,"tag":19,"props":261,"children":262},{},[263],{"type":23,"value":264},"传统checkpoint 恢复耗时有几项随集群规模和拓扑变更成倍放大，DCP逐个消掉：",{"type":18,"tag":43,"props":266,"children":267},{"style":45},[268],{"type":18,"tag":48,"props":269,"children":271},{"src":270,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-23\u002F9.jpg",[],{"type":18,"tag":79,"props":273,"children":275},{"id":274},"_32-万卡集群实测",[276],{"type":23,"value":277},"3.2 万卡集群实测",{"type":18,"tag":19,"props":279,"children":280},{},[281],{"type":23,"value":282},"在盘古 505B 模型万卡集群训练场景下，实测 HyperParallel DCP：",{"type":18,"tag":284,"props":285,"children":286},"ul",{},[287,293,298],{"type":18,"tag":288,"props":289,"children":290},"li",{},[291],{"type":23,"value":292},"异步保存性能：5s；",{"type":18,"tag":288,"props":294,"children":295},{},[296],{"type":23,"value":297},"断点续训状态恢复耗时：290s；",{"type":18,"tag":288,"props":299,"children":300},{},[301],{"type":23,"value":302},"当前持续优化中，预计恢复耗时可进一步压缩至 3 分钟内。",{"type":18,"tag":36,"props":304,"children":306},{"id":305},"_04-业界方案对比",[307],{"type":23,"value":308},"04 业界方案对比",{"type":18,"tag":19,"props":310,"children":311},{},[312],{"type":23,"value":313},"这套三层解耦思路业界也有类似方案，PyTorch 的 torch.distributed.checkpoint 、Megatron-LM 的 dist_checkpointing 、字节的 ByteCheckpoint 都基于这条路线构建。",{"type":18,"tag":43,"props":315,"children":316},{"style":45},[317],{"type":18,"tag":48,"props":318,"children":320},{"src":319,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-23\u002F10.jpg",[],{"type":18,"tag":79,"props":322,"children":324},{"id":323},"hyperparallel-dcp-三项核心竞争力",[325],{"type":23,"value":326},"HyperParallel DCP 三项核心竞争力",{"type":18,"tag":19,"props":328,"children":329},{},[330],{"type":23,"value":331},"内置广播能力：：",{"type":18,"tag":19,"props":333,"children":334},{},[335],{"type":23,"value":336},"HyperParallel DCP 根据 DTensor本地推导同一分片副本rank，只让其中一张rank读盘，再广播给其余副本。对照之下：PyTorch DCP 未做读侧去重；Megatron 必须先 all_gather 分片元数据才能去重；ByteCheckpoint 也还未实现该能力。",{"type":18,"tag":19,"props":338,"children":339},{},[340],{"type":23,"value":341},"异步落盘整链路搬离主进程：",{"type":18,"tag":19,"props":343,"children":344},{},[345],{"type":23,"value":346},"HyperParallel DCP 把规划、通信、写盘都交给子进程，并提供两种跨卡协同策略：重建CPU gloo 组和共享文件系统通信。大集群本来就有共享存储，选后者就不必为 checkpoint 另开通信域。PyTorch DCP子进程只支持重建CPU gloo；Megatron 子进程只负责写盘，集合通信与 .metadata 落盘退回主进程的 finalize；ByteCheckpoint 则在主进程完成规划。",{"type":18,"tag":19,"props":348,"children":349},{},[350],{"type":23,"value":351},"一份实现覆盖双框架多硬件：",{"type":18,"tag":19,"props":353,"children":354},{},[355],{"type":23,"value":356},"HyperParallel DCP 架构不绑定框架和硬件，torch\u002FMindSpore + NPU\u002FGPU 共用。业界其他架构主要支持torch + GPU。",{"type":18,"tag":36,"props":358,"children":360},{"id":359},"_06-结-语",[361],{"type":23,"value":362},"06 结 语",{"type":18,"tag":19,"props":364,"children":365},{},[366],{"type":23,"value":367},"DCP 是 HyperParallel\"故障快速恢复\"这条主线的第一块拼图，它先把\"存得快、存得省、读得省、切分可变\"做扎实。后续故障快恢、临终遗言、SDC 检测等都会基于这套架构构建，只要每片数据全局坐标是可描述的，恢复就是一次区间求交。",{"type":18,"tag":19,"props":369,"children":370},{},[371],{"type":23,"value":372},"DCP 后续规划集中在以下两个关键事项上。",{"type":18,"tag":19,"props":374,"children":375},{},[376],{"type":23,"value":377},"一、极致性能",{"type":18,"tag":19,"props":379,"children":380},{},[381],{"type":23,"value":382},"存侧性能已近乎极致，读侧还有优化空间。其一是读侧负载均衡：同一分片有多个副本 rank 时，\"谁去读\"可以按rank读取量来均衡；其二是读取和广播异步流水化：读完的分片立刻广播，与后面分片的读盘重叠。这两步做完，恢复耗时还能再压一个台阶。",{"type":18,"tag":19,"props":384,"children":385},{},[386],{"type":23,"value":387},"二、独立交付：",{"type":18,"tag":19,"props":389,"children":390},{},[391],{"type":23,"value":392},"DCP 是一个相对独立模块，不与 HyperParallel 并行实现绑定，只依赖\"张量分片报出自己全局坐标\"这一个约定。因此，上层不管是 HyperParallel 还是其他基于torch生态构建的加速方案，都可以接入HyperParallel DCP，享受关键技术优势。",{"type":18,"tag":36,"props":394,"children":396},{"id":395},"_07-欢迎加入-hyperparallel",[397],{"type":23,"value":398},"07 欢迎加入 HyperParallel",{"type":18,"tag":19,"props":400,"children":401},{},[402],{"type":23,"value":403},"我们诚挚邀请各位开发者、研究者加入HyperParallel。无论是贡献代码、完善文档，还是提出改进建议，您的参与都将推动大模型分布式并行技术的边界。让我们一起，让大模型训练更简单、更快速、更智能！",{"type":18,"tag":43,"props":405,"children":406},{"style":45},[407],{"type":18,"tag":48,"props":408,"children":410},{"src":409,"style":51,"alt":7},"\u002Fcategory\u002Finformation\u002Ftechnology-blogs\u002Fbanner\u002F2026-9-23\u002F11.jpg",[],{"title":7,"searchDepth":412,"depth":412,"links":413},4,[414,416,417,418,419,420,421],{"id":81,"depth":415,"text":84},2,{"id":144,"depth":415,"text":147},{"id":180,"depth":415,"text":183},{"id":206,"depth":415,"text":209},{"id":256,"depth":415,"text":259},{"id":274,"depth":415,"text":277},{"id":323,"depth":415,"text":326},"markdown","content:technology-blogs:zh:2026-09-23.md","content","technology-blogs\u002Fzh\u002F2026-09-23.md","technology-blogs\u002Fzh\u002F2026-09-23","md",1791540832245]