Text Generation
Transformers
Safetensors
English
Chinese
qwen3_5_text
veriloop
veriloop-coder
code
coding-agent
software-engineering
mathematical-reasoning
nvfp4
modelopt
tensorrt-llm
vllm
sglang
code-optimized
quantization
open-source
apache-2.0
qwen3_5
self-harness
harness-engineering
surface-host-adapter
evidence-binding
rollback
uncertainty-calibration
long-context
vertical-code-model
recursive-improvement
conversational
Instructions to use rodrigoramosrs/veriloop-coder-e2-nvfp4 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use rodrigoramosrs/veriloop-coder-e2-nvfp4 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="rodrigoramosrs/veriloop-coder-e2-nvfp4") messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("rodrigoramosrs/veriloop-coder-e2-nvfp4") model = AutoModelForCausalLM.from_pretrained("rodrigoramosrs/veriloop-coder-e2-nvfp4", device_map="auto") messages = [ {"role": "user", "content": "Who are you?"}, ] inputs = tokenizer.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=40) print(tokenizer.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use rodrigoramosrs/veriloop-coder-e2-nvfp4 with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "rodrigoramosrs/veriloop-coder-e2-nvfp4" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "rodrigoramosrs/veriloop-coder-e2-nvfp4", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/rodrigoramosrs/veriloop-coder-e2-nvfp4
- SGLang
How to use rodrigoramosrs/veriloop-coder-e2-nvfp4 with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "rodrigoramosrs/veriloop-coder-e2-nvfp4" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "rodrigoramosrs/veriloop-coder-e2-nvfp4", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "rodrigoramosrs/veriloop-coder-e2-nvfp4" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "rodrigoramosrs/veriloop-coder-e2-nvfp4", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use rodrigoramosrs/veriloop-coder-e2-nvfp4 with Docker Model Runner:
docker model run hf.co/rodrigoramosrs/veriloop-coder-e2-nvfp4
| """NVFP4 W4A4 quantization for VeriLoop-E2 (Qwen3.8-27B based). | |
| Uses NVIDIA Model Optimizer's canonical recipe | |
| (NVFP4_W4A4_WEIGHT_LOCAL_HESSIAN_CFG: local Hessian + fp8 scale sweep, | |
| static weight scales + dynamic input scales) with linear_attn blocks and | |
| self-attention projections kept in BF16, matching validated NVFP4 | |
| releases for this architecture family. | |
| Calibration: nvidia/Nemotron-Competitive-Programming-v1 (streaming), defaults | |
| 512 samples x 512 tokens (262144 tokens total). | |
| Requirements: | |
| - 3+ CUDA GPUs holding ~55 GB total for a 27B BF16 model | |
| (tuned on a 17/34/17 GB layout; adjust --layers-split otherwise). | |
| - A C compiler for the Triton JIT (on Windows: run inside a VS | |
| Native Tools prompt, i.e. vcvars64, with CC pointing at cl.exe). | |
| - Enough RAM to hold the model for the CPU-side export (~80 GB for 27B). | |
| Usage: | |
| python quantize_veriloop.py --model ./model-bf16 --output ./model-nvfp4 | |
| """ | |
| import argparse | |
| import os | |
| import sys | |
| def parse_args(): | |
| p = argparse.ArgumentParser(description="VeriLoop-E2 -> NVFP4 quantization") | |
| p.add_argument("--model", default="./model-bf16", | |
| help="Source BF16 HuggingFace model dir (default: ./model-bf16)") | |
| p.add_argument("--output", default="./model-nvfp4", | |
| help="Output dir for the NVFP4 checkpoint (default: ./model-nvfp4)") | |
| p.add_argument("--calib-size", type=int, default=512, | |
| help="Calibration samples (default: 512)") | |
| p.add_argument("--calib-seq-len", type=int, default=512, | |
| help="Calibration sequence length (default: 512)") | |
| p.add_argument("--layers-split", default="16,32,16", | |
| help="Comma-separated layer counts per visible GPU, must sum to 64 " | |
| "(default tuned for a 17/34/17 GB VRAM layout: 16,32,16)") | |
| p.add_argument("--gpu-order", default=None, | |
| help="Optional CUDA_VISIBLE_DEVICES value, e.g. '2,0,1' to make a " | |
| "specific physical GPU cuda:0 (default: natural order)") | |
| return p.parse_args() | |
| ARGS = parse_args() | |
| if ARGS.gpu_order: | |
| os.environ["CUDA_VISIBLE_DEVICES"] = ARGS.gpu_order | |
| if os.name == "nt": | |
| # Triton JIT needs a C compiler; run from a VS Native Tools prompt. | |
| os.environ.setdefault("CC", "cl") | |
| os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True") | |
| import time | |
| import copy | |
| import torch | |
| from pathlib import Path | |
| from transformers import AutoModelForCausalLM, AutoTokenizer | |
| import modelopt.torch.quantization as mtq | |
| MODEL_PATH = ARGS.model | |
| OUTPUT_PATH = ARGS.output | |
| CALIB_SPLIT = "competitive_coding_python_part00" | |
| CALIB_SIZE = ARGS.calib_size | |
| CALIB_SEQ_LEN = ARGS.calib_seq_len | |
| _split = [int(x) for x in ARGS.layers_split.split(",")] | |
| assert sum(_split) == 64, "--layers-split must sum to 64" | |
| assert len(_split) == torch.cuda.device_count(), \ | |
| "--layers-split must have one entry per visible GPU" | |
| def build_device_map(): | |
| # Explicit map: all 64 transformer layers (~48.7 GB) must live on CUDA, | |
| # because the dynamic block quantizer requires CUDA amax tensors. | |
| # Modules with disabled quantizers (embed/head/norm) stay on CPU. | |
| bounds, acc = [], 0 | |
| for count in _split: | |
| bounds.append((acc, acc + count)) | |
| acc += count | |
| dm = { | |
| "model.embed_tokens": "cpu", | |
| "model.norm": "cpu", | |
| "model.rotary_emb": "cpu", | |
| "lm_head": "cpu", | |
| } | |
| for i in range(64): | |
| for dev, (lo, hi) in enumerate(bounds): | |
| if lo <= i < hi: | |
| dm[f"model.layers.{i}"] = dev | |
| break | |
| return dm | |
| def build_quant_cfg(): | |
| # NVIDIA canonical recipe + granularity adjustments: linear_attn (GDN) | |
| # fully BF16 plus BF16 self-attention, matching validated NVFP4 releases | |
| # for this architecture family (MLP-only NVFP4). NVFP4 attention/GDN | |
| # weights produce degenerate output on some stacks; MLP-only is the | |
| # widely-deployed pattern (conv1d/in_proj_a/in_proj_b already disabled | |
| # in the base recipe). Appended last: entries apply in list order, | |
| # later overrides earlier. | |
| cfg = copy.deepcopy(mtq.NVFP4_W4A4_WEIGHT_LOCAL_HESSIAN_CFG) | |
| for name in ["*linear_attn.in_proj_qkv*", "*linear_attn.in_proj_z*", | |
| "*linear_attn.out_proj*", | |
| "*self_attn.q_proj*", "*self_attn.k_proj*", | |
| "*self_attn.v_proj*", "*self_attn.o_proj*"]: | |
| cfg["quant_cfg"].append({"quantizer_name": name, "enable": False}) | |
| return cfg | |
| def messages_to_text(messages): | |
| parts = [] | |
| for msg in messages: | |
| role = msg.get("role", "") | |
| content = msg.get("content", "") | |
| if content: | |
| parts.append(f"{role}: {content}") | |
| return "\n".join(parts) | |
| os.makedirs(OUTPUT_PATH, exist_ok=True) | |
| print("=" * 60) | |
| print("NVFP4 W4A4 Quantization") | |
| print("Model: VeriLoop-E2 (Qwen3.8-27B)") | |
| for i in range(torch.cuda.device_count()): | |
| p = torch.cuda.get_device_properties(i) | |
| free, _ = torch.cuda.mem_get_info(i) | |
| print(f" cuda:{i}: {p.name}, total {p.total_memory/1e9:.1f} GB, free {free/1e9:.1f} GB") | |
| print(f"Calibration: {CALIB_SIZE}x{CALIB_SEQ_LEN} from Nemotron-Competitive-Programming-v1") | |
| print("=" * 60) | |
| # Step 1: Load model across GPUs | |
| print("\n[1/5] Loading model (multi-GPU)...") | |
| t0 = time.time() | |
| tokenizer = AutoTokenizer.from_pretrained(MODEL_PATH, trust_remote_code=True) | |
| model = AutoModelForCausalLM.from_pretrained( | |
| MODEL_PATH, | |
| dtype=torch.bfloat16, | |
| device_map=build_device_map(), | |
| trust_remote_code=True, | |
| ) | |
| print(f" Model loaded in {time.time()-t0:.1f}s") | |
| model.config.use_cache = False # saves activation memory during calibration | |
| n_cpu = sum(1 for v in getattr(model, "hf_device_map", {}).values() if v == "cpu") | |
| print(f" Modules on CPU: {n_cpu} (expected: lm_head/embed/norm with quantizer OFF)") | |
| # Hook fix: "cpu" modules may arrive with execution_device=cuda:N, which makes | |
| # modelopt's writeback (pre_forward) materialize multi-GB weights (embed/head) | |
| # on the GPU. With exec=cpu the weight materializes on CPU (where it already | |
| # lives in the weights map) and forward stays correct (the next layer's hook | |
| # moves activations to CUDA). | |
| from accelerate.hooks import AlignDevicesHook | |
| _fixed = 0 | |
| for mod_name, dev in (getattr(model, "hf_device_map", {}) or {}).items(): | |
| if dev != "cpu": | |
| continue | |
| m = model | |
| for p in mod_name.split("."): | |
| m = getattr(m, p) | |
| for sub in m.modules(): | |
| hook = getattr(sub, "_hf_hook", None) | |
| if isinstance(hook, AlignDevicesHook) and hook.execution_device != torch.device("cpu"): | |
| hook.execution_device = torch.device("cpu") | |
| _fixed += 1 | |
| print(f" Hooks redirected to CPU: {_fixed}") | |
| try: | |
| print(f" HF device map: {model.hf_device_map}") | |
| except Exception: | |
| pass | |
| for i in range(torch.cuda.device_count()): | |
| print(f" cuda:{i} allocated={torch.cuda.memory_allocated(i)/1e9:.1f} GB") | |
| # Step 2: Calibration data | |
| print(f"\n[2/5] Loading calibration data ({CALIB_SIZE} samples)...") | |
| t0 = time.time() | |
| from datasets import load_dataset | |
| calib_data = [] | |
| ds = load_dataset("nvidia/Nemotron-Competitive-Programming-v1", split=CALIB_SPLIT, streaming=True) | |
| for item in ds: | |
| if len(calib_data) >= CALIB_SIZE: | |
| break | |
| text = messages_to_text(item.get("messages", [])) | |
| if not text or len(text.strip()) < 100: | |
| continue | |
| encoded = tokenizer.encode(text, truncation=True, max_length=CALIB_SEQ_LEN) | |
| if len(encoded) < 32: | |
| continue | |
| if len(encoded) < CALIB_SEQ_LEN: | |
| encoded = encoded + [0] * (CALIB_SEQ_LEN - len(encoded)) | |
| calib_data.append(torch.tensor(encoded[:CALIB_SEQ_LEN], dtype=torch.long)) | |
| if len(calib_data) % 16 == 0: | |
| print(f" Collected {len(calib_data)}/{CALIB_SIZE}...") | |
| print(f" Calibration data: {len(calib_data)} seqs in {time.time()-t0:.1f}s") | |
| # Step 3: Forward loop | |
| print("\n[3/5] Running calibration...") | |
| def forward_loop(m): | |
| m.eval() | |
| # cpu/meta params first in line -> use the first CUDA param's device | |
| dev = next(p.device for p in m.parameters() if p.device.type == "cuda") | |
| torch.cuda.empty_cache() # fights fragmentation (no expandable_segments on Windows) | |
| print(f" forward_loop device: {dev}") | |
| for i, ids in enumerate(calib_data): | |
| try: | |
| m(input_ids=ids.unsqueeze(0).to(dev), labels=ids.unsqueeze(0).to(dev)) | |
| except Exception as e: | |
| print(f" Warning sample {i}: {type(e).__name__}: {e}") | |
| continue | |
| if (i + 1) % 16 == 0: | |
| print(f" Calibrated {i+1}/{len(calib_data)}...") | |
| t0 = time.time() | |
| forward_loop(model) | |
| print(f" Calibration took {time.time()-t0:.1f}s") | |
| # Step 4: Quantize | |
| print("\n[4/5] Applying NVFP4 W4A4 quantization...") | |
| t0 = time.time() | |
| model = mtq.quantize(model, build_quant_cfg(), forward_loop) | |
| print(f" Quantization took {time.time()-t0:.1f}s") | |
| mtq.print_quant_summary(model) | |
| # Materialize meta weights of CPU modules (embed/head/norm): the export's dummy | |
| # forward builds its fake input from next(model.parameters()).device, and meta | |
| # tensors break everything ("Cannot copy out of meta tensor"). pre_forward with | |
| # exec=cpu (fix above) brings the weight to CPU without touching VRAM; | |
| # offload=False pins it there. | |
| print("\n Materializing meta weights on CPU...") | |
| _n_meta = 0 | |
| for mod_name, dev in (getattr(model, "hf_device_map", {}) or {}).items(): | |
| if dev != "cpu": | |
| continue | |
| m = model | |
| for p in mod_name.split("."): | |
| m = getattr(m, p) | |
| hook = getattr(m, "_hf_hook", None) | |
| if hook is None: | |
| continue | |
| if any(p.device.type == "meta" for p in m.parameters()): | |
| hook.pre_forward(m) | |
| _n_meta += 1 | |
| hook.offload = False | |
| print(f" Materialized modules: {_n_meta}") | |
| _n_meta_left = sum(1 for p in model.parameters() if p.device.type == "meta") | |
| print(f" Remaining meta params: {_n_meta_left}") | |
| # Export on CPU: move the 64 layers (bf16 weights + NVFP4 scales) to RAM. | |
| # Export quantizes one linear at a time and needs transient workspace on top of | |
| # the resident base, which overflows smaller GPUs. RAM needs roughly: | |
| # ~48.7 GB weights + ~9 GB scales + ~15 GB quantized output + temps. | |
| print("\n Moving layers to CPU for export...") | |
| from accelerate.hooks import AlignDevicesHook as _ADH | |
| for i in range(64): | |
| layer = model.model.layers[i] | |
| layer.to("cpu") | |
| for sub in layer.modules(): | |
| hook = getattr(sub, "_hf_hook", None) | |
| if isinstance(hook, _ADH): | |
| hook.io_device = torch.device("cpu") | |
| hook.execution_device = torch.device("cpu") | |
| model.hf_device_map[f"model.layers.{i}"] = "cpu" | |
| import gc as _gc | |
| _gc.collect() | |
| for _d in range(torch.cuda.device_count()): | |
| with torch.cuda.device(_d): | |
| torch.cuda.empty_cache() | |
| print(" Layers on CPU. VRAM now:", | |
| " / ".join(f"cuda:{d}={torch.cuda.memory_allocated(d)/1e9:.1f}GB" | |
| for d in range(torch.cuda.device_count()))) | |
| # Step 5: Export (workaround for a modelopt multimodal export bug where | |
| # config.architectures ends up None and is_multimodal_model crashes on it) | |
| print("\n[5/5] Exporting...") | |
| try: | |
| import modelopt.torch.export.model_utils as mu | |
| _orig = mu.is_multimodal_model | |
| def _safe_is_mm(m): | |
| try: | |
| return _orig(m) | |
| except TypeError: | |
| archs = getattr(getattr(m, "config", None), "architectures", None) | |
| print(f" is_multimodal_model fallback, architectures={archs} -> False") | |
| return False | |
| mu.is_multimodal_model = _safe_is_mm | |
| print(" Patched is_multimodal_model (None-safe)") | |
| except Exception as e: | |
| print(f" Patch skipped: {e}") | |
| if getattr(model.config, "architectures", None) is None: | |
| model.config.architectures = ["Qwen3_5ForConditionalGeneration"] | |
| print(f" Fixed config.architectures={model.config.architectures}") | |
| from modelopt.torch.export import export_hf_checkpoint | |
| # Free VRAM before export; calib buffers are no longer needed. | |
| import gc | |
| del calib_data | |
| gc.collect() | |
| for _d in range(torch.cuda.device_count()): | |
| with torch.cuda.device(_d): | |
| torch.cuda.empty_cache() | |
| torch.cuda.synchronize() | |
| print(" CUDA cache freed:", | |
| " / ".join(f"cuda:{d}={torch.cuda.memory_allocated(d)/1e9:.1f}GB" | |
| for d in range(torch.cuda.device_count()))) | |
| t0 = time.time() | |
| with torch.inference_mode(): | |
| export_hf_checkpoint(model, export_dir=OUTPUT_PATH, max_shard_size="4GB") | |
| print(f" Export took {time.time()-t0:.1f}s") | |
| print("\n Copying tokenizer/config files...") | |
| import shutil | |
| for fname in [ | |
| "tokenizer.json", "tokenizer_config.json", "special_tokens_map.json", | |
| "config.json", "generation_config.json", "model.safetensors.index.json", | |
| "chat_template.jinja", "merges.txt", "vocab.json", | |
| "preprocessor_config.json", "configuration.json", | |
| ]: | |
| src = os.path.join(MODEL_PATH, fname) | |
| dst = os.path.join(OUTPUT_PATH, fname) | |
| if os.path.exists(src) and not os.path.exists(dst): | |
| shutil.copy2(src, dst) | |
| files = list(Path(OUTPUT_PATH).rglob("*.safetensors")) | |
| total_w = sum(f.stat().st_size for f in files) | |
| total_all = sum(f.stat().st_size for f in Path(OUTPUT_PATH).rglob("*") if f.is_file()) | |
| print(f"\n{'=' * 60}") | |
| print("Quantization complete!") | |
| print(f"Output: {OUTPUT_PATH}") | |
| print(f"Shards: {len(files)}, weights {total_w/1e9:.2f} GB, total {total_all/1e9:.2f} GB") | |
| print(f"{'=' * 60}") | |