vf.load_unalign_init#
产品支持情况#
Ascend 950PR/Ascend 950DT:支持
Atlas A3 训练系列产品/Atlas A3 推理系列产品:不支持
Atlas A2 训练系列产品/Atlas A2 推理系列产品:不支持
功能说明#
为非对齐搬入分配非对齐寄存器(UnalignRegForLoad)。该寄存器作为临时缓存区,用于暂存跨对齐边界的数据,贯穿后续的vf.load_unalign_pre / vf.load_unalign调用链。
非对齐搬入的完整流程为:先调用vf.load_unalign_init分配寄存器,再调用vf.load_unalign_pre进行初始化,最后调用vf.load_unalign进行数据搬入。
函数原型#
load_unalign_init() -> ureg
参数说明#
无
约束说明#
无。
返回值说明#
返回ureg目标reg_tensor。
调用示例#
import os
import pypto_pro.language as pl
import torch
import torch_npu
@pl.vector_function
def example_vf(src_tile, dst_tile):
ureg = vf.load_unalign_init()
vf.load_unalign_pre(ureg, src_tile)
src_reg = vf.load_unalign(ureg, src_tile, post_update=True)
store_ureg = vf.unalign_reg_for_store()
vf.store_unalign(dst_tile, src_reg, store_ureg, 64, post_update=True)
vf.store_unalign_post(dst_tile, store_ureg, 0, post_update=True)
@pl.jit()
def example_kernel(
a: pl.Tensor[[pl.DYNAMIC, pl.DYNAMIC], pl.DT_FP32],
out: pl.Tensor[[pl.DYNAMIC, pl.DYNAMIC], pl.DT_FP32],
):
tf = pl.TileType(shape=[1, 64], dtype=pl.DT_FP32, target_memory=pl.MemorySpace.Vec)
in_a_grp = pl.make_tile_group(type=tf, addrs=0x0, mutex_ids=[0])
in_a = in_a_grp.current()
t_out_grp = pl.make_tile_group(type=tf, addrs=0x100, mutex_ids=[1])
t_out = t_out_grp.current()
with pl.section_vector():
pl.load(in_a, a, [0, 0])
example_vf(in_a, t_out)
pl.store(out, t_out, [0, 0])
def test_example():
device_id = int(os.environ.get("TILE_FWK_DEVICE_ID", 0))
device = f"npu:{device_id}"
core_nums = 1
torch.npu.set_device(device)
a = torch.randn([1, 64], device=device, dtype=torch.float32)
out = torch.empty([1, 64], device=device, dtype=torch.float32)
example_kernel[None, core_nums](a, out)
torch.npu.synchronize()
torch.testing.assert_close(out, a, rtol=1e-5, atol=1e-5)
if __name__ == "__main__":
test_example()
print("PASSED")