| { |
| "model": "Zyphra/ZAYA1-8B", |
| "quantization": { |
| "method": "compressed-tensors", |
| "format": "float-quantized", |
| "scheme": "NVFP4", |
| "scheme_variant": "w4a4", |
| "num_bits_weight": 4, |
| "num_bits_activation": 4, |
| "type": "float", |
| "strategy": "tensor_group", |
| "group_size": 16, |
| "target_modules": [ |
| "Linear" |
| ], |
| "ignore": [ |
| "lm_head", |
| "re:.*router.*", |
| "re:.*norm.*", |
| "re:.*qkv.*", |
| "re:.*cca.*" |
| ], |
| "weight_packed_dtype": "uint8", |
| "weight_scale_dtype": "float8_e4m3fn", |
| "weight_global_scale_dtype": "float32", |
| "input_global_scale_dtype": "float32", |
| "zero_point": "none (symmetric)", |
| "compressor": "NVFP4PackedCompressor" |
| }, |
| "calibration": { |
| "data_path": "data/calibration/arcmix/calibration_data.pt", |
| "num_samples_used": 977, |
| "batch_size": 1, |
| "linears_calibrated": 936 |
| }, |
| "mixed_precision": { |
| "threshold": 500.0, |
| "outlier_layers": [ |
| 1, |
| 19, |
| 31, |
| 37, |
| 39, |
| 65, |
| 69, |
| 71, |
| 73, |
| 75, |
| 77, |
| 79 |
| ], |
| "bf16_exempted_modules": 384, |
| "w4a4_compressed_modules": 936, |
| "description": "12 MoE layers where max_abs > 500 kept at BF16 (384 Linears). FusedMoE requires uniform quantization per layer." |
| }, |
| "hardware": { |
| "gpu": "NVIDIA GeForce RTX 5070 Ti", |
| "vram": "16 GB", |
| "cuda": "13.0" |
| }, |
| "modules_compressed": 936, |
| "output_size_bytes": 9492784787, |
| "dry_run": false |
| } |