rodrigoramosrs commited on
Commit
252883c
·
verified ·
1 Parent(s): 907eff1

Upload scripts/quantize_veriloop.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. scripts/quantize_veriloop.py +19 -10
scripts/quantize_veriloop.py CHANGED
@@ -1,8 +1,10 @@
1
  """NVFP4 W4A4 quantization for VeriLoop-E2 (Qwen3.8-27B based).
2
 
3
- Uses NVIDIA Model Optimizer's canonical recipe verbatim:
4
- NVFP4_W4A4_WEIGHT_LOCAL_HESSIAN_CFG
5
- (local Hessian + fp8 scale sweep, static weight scales + dynamic input scales).
 
 
6
 
7
  Calibration: nvidia/Nemotron-Competitive-Programming-v1 (streaming), defaults
8
  512 samples x 512 tokens (262144 tokens total).
@@ -93,13 +95,20 @@ def build_device_map():
93
 
94
 
95
  def build_quant_cfg():
96
- # NVIDIA canonical recipe, VERBATIM (no overrides):
97
- # local_hessian + fp8_scale_sweep, global *weight_quantizer (static) +
98
- # *input_quantizer (dynamic), with the right exclusions
99
- # (mtp/visual/lm_head/embeddings). Extra overrides only duplicate
100
- # calibration state; a forced layerwise mode grows an unbounded per-layer
101
- # capture cache. The base global mode does not accumulate.
102
- return copy.deepcopy(mtq.NVFP4_W4A4_WEIGHT_LOCAL_HESSIAN_CFG)
 
 
 
 
 
 
 
103
 
104
 
105
  def messages_to_text(messages):
 
1
  """NVFP4 W4A4 quantization for VeriLoop-E2 (Qwen3.8-27B based).
2
 
3
+ Uses NVIDIA Model Optimizer's canonical recipe
4
+ (NVFP4_W4A4_WEIGHT_LOCAL_HESSIAN_CFG: local Hessian + fp8 scale sweep,
5
+ static weight scales + dynamic input scales) with linear_attn blocks and
6
+ self-attention projections kept in BF16, matching validated NVFP4
7
+ releases for this architecture family.
8
 
9
  Calibration: nvidia/Nemotron-Competitive-Programming-v1 (streaming), defaults
10
  512 samples x 512 tokens (262144 tokens total).
 
95
 
96
 
97
  def build_quant_cfg():
98
+ # NVIDIA canonical recipe + granularity adjustments: linear_attn (GDN)
99
+ # fully BF16 plus BF16 self-attention, matching validated NVFP4 releases
100
+ # for this architecture family (MLP-only NVFP4). NVFP4 attention/GDN
101
+ # weights produce degenerate output on some stacks; MLP-only is the
102
+ # widely-deployed pattern (conv1d/in_proj_a/in_proj_b already disabled
103
+ # in the base recipe). Appended last: entries apply in list order,
104
+ # later overrides earlier.
105
+ cfg = copy.deepcopy(mtq.NVFP4_W4A4_WEIGHT_LOCAL_HESSIAN_CFG)
106
+ for name in ["*linear_attn.in_proj_qkv*", "*linear_attn.in_proj_z*",
107
+ "*linear_attn.out_proj*",
108
+ "*self_attn.q_proj*", "*self_attn.k_proj*",
109
+ "*self_attn.v_proj*", "*self_attn.o_proj*"]:
110
+ cfg["quant_cfg"].append({"quantizer_name": name, "enable": False})
111
+ return cfg
112
 
113
 
114
  def messages_to_text(messages):