#!/usr/bin/env python3 """Smallest official Gluon boundary: Python launcher to one GPU kernel.""" import torch from triton.experimental import gluon from triton.experimental.gluon import language as gl @gluon.jit def copy_scalar_kernel(input_pointer, output_pointer): value = gl.load(input_pointer) gl.store(output_pointer, value) def run() -> torch.Tensor: source = torch.tensor([1.0], device="cuda") destination = torch.empty_like(source) copy_scalar_kernel[(1,)](source, destination) return destination if __name__ == "__main__": print(run())