batmac commited on
Commit
c0f8664
·
verified ·
1 Parent(s): 3834faf

Keep classifier head in fp32 so the checkpoint loads on Linux CPU

Browse files
Files changed (5) hide show
  1. README.md +11 -5
  2. config.json +3 -1
  3. model.safetensors +2 -2
  4. scripts/README.md +4 -1
  5. scripts/quantize.py +29 -2
README.md CHANGED
@@ -36,11 +36,17 @@ training data and evaluation.
36
  | Method | bitsandbytes NF4, weight-only |
37
  | Compute dtype | float32 |
38
  | Double quantization | off |
39
- | Quantized modules | all `torch.nn.Linear` (attention, dense, pooler, classifier) |
 
40
  | Checkpoint size | ~699 MB (original fp32: ~1.74 GB) |
41
  | Resident memory | ~648 MB (original fp32: ~1,660 MB) |
42
 
43
- Embeddings, LayerNorm, and the sigmoid head remain in full precision.
 
 
 
 
 
44
 
45
  ## Usage
46
 
@@ -86,8 +92,8 @@ measurements below. It is not needed to use the model; see
86
  Measured against the fp32 weights of the original model on 16 held-in prompts
87
  spanning clearly human to clearly AI text:
88
 
89
- - mean absolute change in P(AI): 0.021
90
- - maximum absolute change in P(AI): 0.080
91
  - verdict flips at a 0.5 decision threshold: 0
92
 
93
  Quantized scores are close but not identical to the fp32 model, and the largest
@@ -126,4 +132,4 @@ additional error on top of that. MIT license, inherited from the original model.
126
  journal={Working Notes of CLEF},
127
  year={2026}
128
  }
129
- ```
 
36
  | Method | bitsandbytes NF4, weight-only |
37
  | Compute dtype | float32 |
38
  | Double quantization | off |
39
+ | Quantized modules | 145 `torch.nn.Linear` layers (attention, dense, pooler) |
40
+ | Kept in fp32 | classifier head, embeddings, LayerNorm |
41
  | Checkpoint size | ~699 MB (original fp32: ~1.74 GB) |
42
  | Resident memory | ~648 MB (original fp32: ~1,660 MB) |
43
 
44
+ The classifier head is deliberately left in fp32. It is a `[1, 1024]` matrix, and
45
+ bitsandbytes' packed CPU kernel asserts that each quantized layer's output
46
+ dimension is divisible by its block size, which 1 is not. Quantizing the head
47
+ works on Apple Silicon MPS but raises
48
+ `AssertionError: N must be divisible by block_n` on Linux CPU, so the head is kept
49
+ exact to make one checkpoint that loads everywhere.
50
 
51
  ## Usage
52
 
 
92
  Measured against the fp32 weights of the original model on 16 held-in prompts
93
  spanning clearly human to clearly AI text:
94
 
95
+ - mean absolute change in P(AI): 0.020
96
+ - maximum absolute change in P(AI): 0.076
97
  - verdict flips at a 0.5 decision threshold: 0
98
 
99
  Quantized scores are close but not identical to the fp32 model, and the largest
 
132
  journal={Working Notes of CLEF},
133
  year={2026}
134
  }
135
+ ```
config.json CHANGED
@@ -44,7 +44,9 @@
44
  "bnb_4bit_use_double_quant": false,
45
  "llm_int8_enable_fp32_cpu_offload": false,
46
  "llm_int8_has_fp16_weight": false,
47
- "llm_int8_skip_modules": null,
 
 
48
  "llm_int8_threshold": 6.0,
49
  "load_in_4bit": true,
50
  "load_in_8bit": false,
 
44
  "bnb_4bit_use_double_quant": false,
45
  "llm_int8_enable_fp32_cpu_offload": false,
46
  "llm_int8_has_fp16_weight": false,
47
+ "llm_int8_skip_modules": [
48
+ "classifier"
49
+ ],
50
  "llm_int8_threshold": 6.0,
51
  "load_in_4bit": true,
52
  "load_in_8bit": false,
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ae345b48c7f26cde6edeaf860a107a3fc2f54fed0b055bfe9a50ffbf9b7c97f9
3
- size 698682667
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b6d34cd0e7f8744d9954f9594b320213ac847a399516182f79bf73f1b75504fe
3
+ size 698685757
scripts/README.md CHANGED
@@ -21,7 +21,10 @@ uploaded directly. Optional argument: output directory, defaulting to
21
  python scripts/quantize.py
22
  ```
23
 
24
- It finishes by reloading the saved checkpoint and printing a P(AI) sanity value.
 
 
 
25
 
26
  ## `bench_quant.py`
27
 
 
21
  python scripts/quantize.py
22
  ```
23
 
24
+ It keeps the classifier head in fp32, because bitsandbytes' packed CPU kernel
25
+ requires each quantized layer's output dimension to divide evenly by its block
26
+ size and the head is `[1, 1024]`. It then asserts that no quantized layer would
27
+ break that kernel, and prints a reload sanity value.
28
 
29
  ## `bench_quant.py`
30
 
scripts/quantize.py CHANGED
@@ -1,7 +1,7 @@
1
  """Save a 4-bit NF4 checkpoint of the detector for fast repeated loading.
2
 
3
- Writes to models/gradient-ai-text-detector-4bit, which app.py picks up
4
- automatically. The quantized checkpoint is roughly 650 MB instead of 1.7 GB.
5
 
6
  Usage:
7
  .venv/bin/python scripts/quantize.py [output_dir]
@@ -22,6 +22,30 @@ DEFAULT_OUT = Path(__file__).parent.parent / "models" / "gradient-ai-text-detect
22
 
23
  SAMPLE = "In today's rapidly evolving digital landscape, organizations must unlock value."
24
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
25
 
26
  def main(out_dir):
27
  out_dir = Path(out_dir)
@@ -33,10 +57,13 @@ def main(out_dir):
33
  load_in_4bit=True,
34
  bnb_4bit_quant_type="nf4",
35
  bnb_4bit_compute_dtype=torch.float32,
 
36
  ),
37
  )
38
  model.eval()
39
  model.to(torch.device("cpu"))
 
 
40
  model.save_pretrained(out_dir, safe_serialization=True)
41
  AutoTokenizer.from_pretrained(MODEL_ID).save_pretrained(out_dir)
42
  AutoTokenizer.from_pretrained(MODEL_ID, use_fast=False).save_pretrained(out_dir)
 
1
  """Save a 4-bit NF4 checkpoint of the detector for fast repeated loading.
2
 
3
+ Writes to models/gradient-ai-text-detector-4bit by default. The quantized
4
+ checkpoint is roughly 650 MB instead of 1.7 GB.
5
 
6
  Usage:
7
  .venv/bin/python scripts/quantize.py [output_dir]
 
22
 
23
  SAMPLE = "In today's rapidly evolving digital landscape, organizations must unlock value."
24
 
25
+ # bitsandbytes' packed CPU kernel asserts that each quantized weight's output
26
+ # dimension divides evenly by its block size. The classifier head is [1, 1024],
27
+ # so it must stay in fp32 or the checkpoint cannot run on Linux CPU (Spaces).
28
+ SKIP_MODULES = ["classifier"]
29
+ BLOCK = 64
30
+
31
+
32
+ def check_cpu_compatible(model):
33
+ """Fail loudly rather than ship a checkpoint only macOS can load."""
34
+ quantized = [
35
+ module
36
+ for module in model.modules()
37
+ if type(module).__name__ == "Linear4bit"
38
+ ]
39
+ offenders = sorted(
40
+ {module.out_features for module in quantized if module.out_features % BLOCK}
41
+ )
42
+ if offenders:
43
+ raise SystemExit(
44
+ f"quantized layers with out_features {offenders} are not divisible by "
45
+ f"{BLOCK}; add them to SKIP_MODULES"
46
+ )
47
+ return len(quantized)
48
+
49
 
50
  def main(out_dir):
51
  out_dir = Path(out_dir)
 
57
  load_in_4bit=True,
58
  bnb_4bit_quant_type="nf4",
59
  bnb_4bit_compute_dtype=torch.float32,
60
+ llm_int8_skip_modules=SKIP_MODULES,
61
  ),
62
  )
63
  model.eval()
64
  model.to(torch.device("cpu"))
65
+ print(f"quantized {check_cpu_compatible(model)} linear layers")
66
+ print(f"kept in fp32: {SKIP_MODULES} ({model.classifier.weight.dtype})")
67
  model.save_pretrained(out_dir, safe_serialization=True)
68
  AutoTokenizer.from_pretrained(MODEL_ID).save_pretrained(out_dir)
69
  AutoTokenizer.from_pretrained(MODEL_ID, use_fast=False).save_pretrained(out_dir)