#!/usr/bin/env python3 """Helion v1.0 tiled matmul boundary from the official project example.""" import helion import helion.language as hl import torch @helion.kernel() def matmul(x: torch.Tensor, y: torch.Tensor) -> torch.Tensor: m, k = x.size() _k, n = y.size() output = torch.empty([m, n], dtype=x.dtype, device=x.device) for tile_m, tile_n in hl.tile([m, n]): accumulator = hl.zeros([tile_m, tile_n], dtype=torch.float32) for tile_k in hl.tile(k): accumulator = torch.addmm( accumulator, x[tile_m, tile_k], y[tile_k, tile_n] ) output[tile_m, tile_n] = accumulator return output if __name__ == "__main__": raise SystemExit("parser_only: GPU execution and reference comparison not captured")