sam3-text-onnx / quantize_sam3_fp16.py
danilobukvic's picture
Upload quantize_sam3_fp16.py with huggingface_hub
7f9881d verified
Raw History Blame Contribute Delete
3.8 kB
"""Convert SAM3 ONNX components from fp32 to fp16.
Halves file size with effectively zero accuracy loss on most modern hardware
(GPU, CPU with AVX-512, all modern phones, WebGPU). This is the safest
quantization step — no calibration data required, no risk of correctness drift.
Produces per-component fp16 files alongside the originals:
sam3-onnx-test/vision_encoder.onnx (fp32)
sam3-onnx-test/vision_encoder_fp16.onnx (fp16, ~half size)
... etc.
Run validate_sam3_e2e.py with --variant fp16 afterward to confirm correctness.
"""
from pathlib import Path
import time
import onnx
from onnxconverter_common import float16
OUTPUT_DIR = Path("sam3-onnx-test")
# (input_file, output_file, output_data_file_name_if_external)
COMPONENTS = [
("vision_encoder.onnx", "vision_encoder_fp16.onnx", "vision_encoder_fp16.onnx.data"),
("text_encoder.onnx", "text_encoder_fp16.onnx", "text_encoder_fp16.onnx.data"),
("decoder.onnx", "decoder_fp16.onnx", "decoder_fp16.onnx.data"),
]
def convert(in_name: str, out_name: str, data_name: str) -> None:
in_path = OUTPUT_DIR / in_name
out_path = OUTPUT_DIR / out_name
print(f"\n[{in_name}] → [{out_name}]")
print(f" Loading {in_path} ...")
t0 = time.time()
model = onnx.load(str(in_path))
in_size = in_path.stat().st_size
in_data = OUTPUT_DIR / f"{in_name}.data"
if in_data.exists():
in_size += in_data.stat().st_size
print(f" Source size: {in_size / 1024 / 1024:.1f} MB total")
print(f" Converting to fp16 (keep_io_types=True so inputs/outputs stay fp32) ...")
# keep_io_types=True means the model accepts fp32 inputs and returns fp32 outputs
# but uses fp16 internally. This avoids needing to convert tensors before feeding.
model_fp16 = float16.convert_float_to_float16(
model,
keep_io_types=True,
disable_shape_infer=False,
)
print(f" Saving to {out_path} ...")
# If the model is large, ONNX will spill weights into an external .data file
# automatically. Force external data for the big components to keep .onnx
# files under the 2 GB protobuf limit and to mirror the original structure.
is_big = in_size > 100 * 1024 * 1024
if is_big:
# Remove existing external data files to ensure clean save
for p in [out_path, OUTPUT_DIR / data_name]:
if p.exists():
p.unlink()
onnx.save(
model_fp16,
str(out_path),
save_as_external_data=True,
all_tensors_to_one_file=True,
location=data_name,
size_threshold=1024, # spill any tensor > 1 KB
)
else:
onnx.save(model_fp16, str(out_path))
out_size = out_path.stat().st_size
out_data = OUTPUT_DIR / data_name
if out_data.exists():
out_size += out_data.stat().st_size
elapsed = time.time() - t0
print(f" Output size: {out_size / 1024 / 1024:.1f} MB total")
print(f" Reduction: {in_size / out_size:.2f}x ({elapsed:.1f}s)")
def main() -> None:
print("SAM3 fp32 → fp16 quantization\n" + "=" * 40)
for in_name, out_name, data_name in COMPONENTS:
if not (OUTPUT_DIR / in_name).exists():
print(f"\n[SKIP] {in_name} not found, did you run the export scripts?")
continue
try:
convert(in_name, out_name, data_name)
except Exception as e:
print(f" ❌ FAILED: {type(e).__name__}: {e}")
print("\n" + "=" * 40)
print("Done. Files in output dir:")
for f in sorted(OUTPUT_DIR.iterdir()):
size = f.stat().st_size / 1024 / 1024
marker = " ←fp16" if "fp16" in f.name else ""
print(f" {f.name}: {size:.1f} MB{marker}")
if __name__ == "__main__":
main()