#!/usr/bin/env python3 """Compare eager and torch.compile without claiming a GLM-5.2 receipt.""" from __future__ import annotations import json import time import torch class TinyMoE(torch.nn.Module): def __init__(self, hidden: int = 256, experts: int = 4) -> None: super().__init__() self.router = torch.nn.Linear(hidden, experts, bias=False) self.experts = torch.nn.ModuleList( [torch.nn.Linear(hidden, hidden, bias=False) for _ in range(experts)] ) def forward(self, x: torch.Tensor) -> torch.Tensor: # The discrete route intentionally creates a graph-break risk. The # compile explanation should show that a graph break is evidence, not a # silent compiler failure. route = int(self.router(x).mean(dim=0).argmax().item()) return self.experts[route](x) def timed(fn, x: torch.Tensor) -> tuple[torch.Tensor, float]: if x.is_cuda: torch.cuda.synchronize() start = time.perf_counter() y = fn(x) if x.is_cuda: torch.cuda.synchronize() return y, (time.perf_counter() - start) * 1_000 def main() -> None: device = "cuda" if torch.cuda.is_available() else "cpu" model = TinyMoE().to(device).eval() x = torch.randn(32, 256, device=device) eager, eager_ms = timed(model, x) compiled_model = torch.compile(model, backend="inductor") compiled, compile_and_first_ms = timed(compiled_model, x) compiled_steady, steady_ms = timed(compiled_model, x) print( json.dumps( { "device": device, "torch": torch.__version__, "eager_ms": eager_ms, "compile_and_first_ms": compile_and_first_ms, "compiled_steady_ms": steady_ms, "max_abs_error": float((eager - compiled).abs().max()), "steady_matches_first": bool(torch.allclose(compiled, compiled_steady)), "receipt_scope": "toy_operator_path_not_glm_5_2", }, indent=2, ) ) if __name__ == "__main__": main()