"""Convert SAM3 ONNX components from fp32 to fp16. Halves file size with effectively zero accuracy loss on most modern hardware (GPU, CPU with AVX-512, all modern phones, WebGPU). This is the safest quantization step — no calibration data required, no risk of correctness drift. Produces per-component fp16 files alongside the originals: sam3-onnx-test/vision_encoder.onnx (fp32) sam3-onnx-test/vision_encoder_fp16.onnx (fp16, ~half size) ... etc. Run validate_sam3_e2e.py with --variant fp16 afterward to confirm correctness. """ from pathlib import Path import time import onnx from onnxconverter_common import float16 OUTPUT_DIR = Path("sam3-onnx-test") # (input_file, output_file, output_data_file_name_if_external) COMPONENTS = [ ("vision_encoder.onnx", "vision_encoder_fp16.onnx", "vision_encoder_fp16.onnx.data"), ("text_encoder.onnx", "text_encoder_fp16.onnx", "text_encoder_fp16.onnx.data"), ("decoder.onnx", "decoder_fp16.onnx", "decoder_fp16.onnx.data"), ] def convert(in_name: str, out_name: str, data_name: str) -> None: in_path = OUTPUT_DIR / in_name out_path = OUTPUT_DIR / out_name print(f"\n[{in_name}] → [{out_name}]") print(f" Loading {in_path} ...") t0 = time.time() model = onnx.load(str(in_path)) in_size = in_path.stat().st_size in_data = OUTPUT_DIR / f"{in_name}.data" if in_data.exists(): in_size += in_data.stat().st_size print(f" Source size: {in_size / 1024 / 1024:.1f} MB total") print(f" Converting to fp16 (keep_io_types=True so inputs/outputs stay fp32) ...") # keep_io_types=True means the model accepts fp32 inputs and returns fp32 outputs # but uses fp16 internally. This avoids needing to convert tensors before feeding. model_fp16 = float16.convert_float_to_float16( model, keep_io_types=True, disable_shape_infer=False, ) print(f" Saving to {out_path} ...") # If the model is large, ONNX will spill weights into an external .data file # automatically. Force external data for the big components to keep .onnx # files under the 2 GB protobuf limit and to mirror the original structure. is_big = in_size > 100 * 1024 * 1024 if is_big: # Remove existing external data files to ensure clean save for p in [out_path, OUTPUT_DIR / data_name]: if p.exists(): p.unlink() onnx.save( model_fp16, str(out_path), save_as_external_data=True, all_tensors_to_one_file=True, location=data_name, size_threshold=1024, # spill any tensor > 1 KB ) else: onnx.save(model_fp16, str(out_path)) out_size = out_path.stat().st_size out_data = OUTPUT_DIR / data_name if out_data.exists(): out_size += out_data.stat().st_size elapsed = time.time() - t0 print(f" Output size: {out_size / 1024 / 1024:.1f} MB total") print(f" Reduction: {in_size / out_size:.2f}x ({elapsed:.1f}s)") def main() -> None: print("SAM3 fp32 → fp16 quantization\n" + "=" * 40) for in_name, out_name, data_name in COMPONENTS: if not (OUTPUT_DIR / in_name).exists(): print(f"\n[SKIP] {in_name} not found, did you run the export scripts?") continue try: convert(in_name, out_name, data_name) except Exception as e: print(f" ❌ FAILED: {type(e).__name__}: {e}") print("\n" + "=" * 40) print("Done. Files in output dir:") for f in sorted(OUTPUT_DIR.iterdir()): size = f.stat().st_size / 1024 / 1024 marker = " ←fp16" if "fp16" in f.name else "" print(f" {f.name}: {size:.1f} MB{marker}") if __name__ == "__main__": main()