INNER CODE UNIT · Python
quantize
Ki6an/fastT5 · fastT5/onnx_exporter.py:261
def quantize(models_name_or_path):
"""
Quantize the weights of the model from float32 to in8 to allow very efficient inference on modern CPU
Uses unsigned ints for activation values, signed ints for weights, per
https://onnxruntime.ai/docs/performance/quantization.html#data-type-selection
it is faster on most CPU architectures
Args:
onnx_model_path: Path to location the exported ONNX model is stored
Returns: The Path generated for the quantized
"""
from onnxruntime.quantization import quantize_dynamic, QuantType
bar = Bar("Quantizing...", max=3)
quant_model_paths = []
for model in models_name_or_path:
model_name = model.as_posix()