final / scripts /04_quantize.sh
logeshsenthil13's picture
Upload folder using huggingface_hub
da9358b verified
Raw
History Blame Contribute Delete
3.66 kB
#!/usr/bin/env bash
# Step 4: ModelOpt ONNX post-training quantization of the plugin-free vision encoder (no PyTorch involved).
#
# usage: 04_quantize.sh <tag> [extra modelopt args...] (INT8 W8A8, per-channel weights / per-tensor activations)
# env: CALIB_METHOD=max|entropy (default max)
# EXCLUDE="regex1 regex2" extra --nodes_to_exclude patterns (re.match against node names)
# IN_ONNX=<path> input model (default onnx/fp16_noplugin, use the 03b smoothed model for INT8)
#
# Recipe (Edge-LLM's vision recipe applied to the ONNX with ModelOpt, in INT8):
# * only Gemm (linear) layers get Q/DQ -> attention MatMul/Softmax stay FP16, patch-embed Conv untouched
# * /blocks.31/mlp/down_proj/Gemm is excluded: the exporter deliberately runs it in FP32 because its
# activations overflow FP16 (see tensorrt_edgellm/visual_models/qwen2_5_vl_model.py); those activations
# are exactly the ones that must not be squeezed through a per-tensor 8-bit scale.
# * --high_precision_dtype fp32 : keep the exported FP16/FP32 mix exactly as is. With the default (fp16)
# ModelOpt runs autocast and converts the FP32 overflow-workaround sub-graph to FP16 -> overflow -> garbage.
set -euo pipefail
QTAG=$1; shift 1
MODE=int8
source "$(dirname "$0")/env.sh"
# make the pip-installed CUDA libs (needed by onnxruntime-gpu's CUDA EP inside ModelOpt) visible
export LD_LIBRARY_PATH=$(ls -d $SP/nvidia/*/lib 2>/dev/null | tr '\n' ':')${LD_LIBRARY_PATH:-}
SHAPES=$(cat $ROOT/data/calib_${TAG}.shapes)
IN=${IN_ONNX:-$ROOT/onnx/fp16_noplugin/model.onnx} # e.g. IN_ONNX=$ROOT/onnx/fp16_smooth_a0.5/model.onnx
OUT_DIR=$ROOT/onnx/${QTAG}
OUT=$OUT_DIR/model.onnx
mkdir -p $OUT_DIR
LOG=$ROOT/logs/04_quantize_${QTAG}.log
EXCL=( '/blocks\.31/mlp/down_proj/Gemm' )
for p in ${EXCLUDE:-}; do EXCL+=( "$p" ); done
echo "in=$IN mode=$MODE calib=${CALIB_METHOD:-max} exclude=${EXCL[*]} out=$OUT" | tee $LOG
python -m modelopt.onnx.quantization \
--onnx_path $IN \
--quantize_mode $MODE \
--calibration_method ${CALIB_METHOD:-max} \
--calibration_data_path $ROOT/data/calib_${TAG}.npz \
--calibration_shapes "$SHAPES" \
--calibration_eps cuda:0 cpu \
--op_types_to_quantize Gemm \
--nodes_to_exclude "${EXCL[@]}" \
--high_precision_dtype fp32 \
--disable_mha_qdq \
--use_external_data_format \
--output_path $OUT \
--log_level INFO "$@" 2>&1 | tee -a $LOG
# sanity: what got quantized, and is the FP32 workaround still intact?
python - "$OUT" <<'PY' 2>&1 | tee -a $LOG
import sys, onnx
from collections import Counter
m = onnx.load(sys.argv[1], load_external_data=False)
c = Counter(n.op_type for n in m.graph.node)
print("opsets:", [(o.domain, o.version) for o in m.opset_import])
print("QuantizeLinear:", c["QuantizeLinear"], "DequantizeLinear:", c["DequantizeLinear"], "Gemm:", c["Gemm"], "MatMul:", c["MatMul"], "Conv:", c["Conv"])
prod = {o: n for n in m.graph.node for o in n.output}
q_gemms = [n.name for n in m.graph.node if n.op_type == "Gemm" and prod.get(n.input[0]) is not None and prod[n.input[0]].op_type == "DequantizeLinear"]
u_gemms = [n.name for n in m.graph.node if n.op_type == "Gemm" and n.name not in q_gemms]
print(f"Gemms with quantized input: {len(q_gemms)} ; NOT quantized ({len(u_gemms)}): {u_gemms}")
inits = {i.name: i for i in m.graph.initializer}
d = [n for n in m.graph.node if n.name == "/blocks.31/mlp/down_proj/Gemm"]
if d:
w = d[0].input[1]; dt = inits[w].data_type if w in inits else -1
print("blocks.31 down_proj weight dtype:", onnx.TensorProto.DataType.Name(dt) if dt > 0 else "non-initializer", "(expected FLOAT)")
PY
echo "done -> $OUT (log: $LOG)"