#!/usr/bin/env python3 """Small FP8 and quantization preparation probes. The Model Optimizer route is opt-in because calibration changes model state. Neither branch exports or claims a GLM-5.2 artifact. """ from __future__ import annotations import argparse import json import torch def transformer_engine_probe() -> dict[str, object]: import transformer_engine.pytorch as te from transformer_engine.common.recipe import DelayedScaling layer = te.Linear(128, 128, bias=False).cuda().eval() x = torch.randn(16, 128, device="cuda", dtype=torch.float16) with torch.no_grad(), te.fp8_autocast(enabled=True, fp8_recipe=DelayedScaling()): output = layer(x) return {"shape": list(output.shape), "dtype": str(output.dtype)} def model_optimizer_probe(execute: bool) -> dict[str, object]: import modelopt.torch.quantization as mtq result: dict[str, object] = { "config": "NVFP4_DEFAULT_CFG", "execute": execute, "warning": "preparation_only_not_a_glm_5_2_export", } if not execute: return result model = torch.nn.Linear(128, 128, bias=False).cuda().eval() def forward_loop(candidate: torch.nn.Module) -> None: with torch.no_grad(): for _ in range(4): candidate(torch.randn(8, 128, device="cuda")) mtq.quantize(model, mtq.NVFP4_DEFAULT_CFG, forward_loop=forward_loop) result["quantized"] = True return result def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("--execute-modelopt", action="store_true") args = parser.parse_args() if not torch.cuda.is_available(): raise SystemExit("CUDA GPU required") print( json.dumps( { "transformer_engine": transformer_engine_probe(), "model_optimizer": model_optimizer_probe(args.execute_modelopt), }, indent=2, ) ) if __name__ == "__main__": main()