Nawah-RuleCheck-v2 / LADDER.json
oddadmix's picture
ship LADDER.json for reproducibility
3f050ad verified
Raw
History Blame Contribute Delete
19.2 kB
{
"title": "Emhotob Arabic rule-checking ladder — 500K / 1M / 5M / 50M",
"task": "binary rule checking: given an Arabic text and a rule in natural Arabic, predict مطابق (valid) / مخالف (invalid)",
"generated": "2026-09-01",
"eval_protocol": {
"harness": "/opt/projects/math-gsk8/rules/ladder/eval_all.py",
"note": "all four rungs re-scored by this one script in one session; nothing carried over from a model card or an earlier training log",
"splits": {
"eval_seen_wording": {
"n": 6624,
"what": "held-out texts (task_id split, zero text overlap), rule phrasings the model trained on — text generalisation",
"majority_baseline": 0.6401
},
"eval_unseen_wording": {
"n": 6624,
"what": "same held-out texts, rule phrasings held out of training entirely (from the same 631-paraphrase pool) — semantic generalisation",
"majority_baseline": 0.6401
},
"counterfactual_minimal_pairs": {
"n": 1080,
"what": "9 rules x real corpus texts with a surgical edit that flips the checker verdict; canonical rule wording. Tests sensitivity to the text.",
"majority_baseline": 0.5
},
"ood_handwritten_wording": {
"n": 1200,
"what": "10 rules x 120 held-out texts, rule stated in a hand-written terse/colloquial wording outside the paraphrase pool — the hard split",
"majority_baseline": 0.5925
}
},
"latency": "single example, batch 1, float32, median of 200 runs after 30 warmups. GPU = RTX 5090. CPU = 2 threads. Same input for every model."
},
"caveats": [
"Best checkpoint is selected on eval_unseen_wording macro-F1 (the training script's metric_for_best_model). That split is therefore mildly optimistic for every rung. It is the same protocol for all four, so the comparison holds; the counterfactual and hand-written-wording splits are untouched by that selection.",
"The 50M rung was retuned during this work. It originally shipped at 3e-4 / 3 epochs (carried over from the Emhotob SFT ladder, never swept), scoring 0.9799 unseen and 0.8667 hand-written. The published weights are now the swept 1e-4 / 10 epoch run: 0.9980 and 0.8883. All four rungs in this table are therefore independently LR-tuned on the same data and the same eval.",
"The 500K rung scores 0.5558 on hand-written rule wordings, BELOW the 0.5925 majority-class baseline: on wording it has not seen, always answering the majority class beats it.",
"'Smaller needs a higher learning rate' holds only below 5M. Best LR by rung: 500K 6e-3, 1M 1e-3, 5M 1e-4, 51.8M 1e-4 - a 60x climb from 5M down to 500K, then flat at the top: the 5M and the 51.8M share the same optimum. Neither recipe transfers: the 51.8M's LR on the 500K base scores 0.9179 unseen (vs 0.9780 tuned), and the 500K's LR on the 5M base scores 0.77."
],
"models": [
{
"model": "oddadmix/Nawah-RuleCheck-500K",
"url": "https://huggingface.co/oddadmix/Nawah-RuleCheck-500K",
"rung": "500K",
"base_model": "oddadmix/Emhotob-500K-v2",
"arch": {
"hidden_size": 16,
"layers": 2,
"heads": 2,
"kv_heads": 1,
"intermediate": 48,
"vocab": 32000,
"max_pos": 2048,
"tied_embeddings": true
},
"params_total": 518256,
"params_non_embedding": 6256,
"weights_bytes": 2075200,
"weights_mb": 2.08,
"train": {
"lr": "6e-3",
"epochs": 15,
"warmup": 500,
"batch_size": 64,
"train_pairs": 159240,
"max_length": 224,
"precision": "bf16",
"trained_by": "previously released; re-scored here, not retrained",
"lr_sweep": [
{
"lr": "3e-4",
"epochs": 3,
"seen": 0.9583,
"unseen": 0.9179
},
{
"lr": "1e-3",
"epochs": 10,
"seen": 0.9787,
"unseen": 0.946
},
{
"lr": "3e-3",
"epochs": 10,
"seen": 0.9863,
"unseen": 0.9695
},
{
"lr": "6e-3",
"epochs": 10,
"seen": 0.9866,
"unseen": 0.9675
},
{
"lr": "1e-2",
"epochs": 10,
"seen": 0.9792,
"unseen": 0.9635
},
{
"lr": "6e-3",
"epochs": 15,
"seen": 0.9867,
"unseen": 0.978,
"chosen": true
},
{
"lr": "1e-3",
"epochs": 20,
"seen": 0.9841,
"unseen": 0.9653
},
{
"lr": "3e-3",
"epochs": 20,
"seen": 0.9902,
"unseen": 0.9648
}
],
"lr_rationale": "Sweep run in the earlier session that released this model."
},
"accuracy": {
"eval_seen_wording": 0.9866,
"eval_unseen_wording": 0.9778,
"counterfactual_minimal_pairs": 0.9926,
"ood_handwritten_wording": 0.5558
},
"macro_f1": {
"eval_seen_wording": 0.9854,
"eval_unseen_wording": 0.9759,
"ood_handwritten_wording": 0.5552
},
"counterfactual_mean_flip_rate": 0.985,
"latency_ms_bs1": {
"gpu_rtx5090_median": 0.96,
"cpu_2threads_median": 0.326,
"input_tokens": 34
},
"per_rule_unseen": {
"ends_question": 0.9984,
"has_city": 0.966,
"has_date": 0.9804,
"has_number": 0.9804,
"has_phone": 0.9618,
"has_price": 0.9981,
"max_words_25": 0.7,
"max_words_30": 0.8043,
"max_words_40": 0.7391,
"max_words_50": 0.8837,
"min_words_15": 0.9286,
"min_words_20": 0.8857,
"min_words_25": 0.975,
"min_words_30": 0.9487,
"no_email": 1.0,
"no_excess_punct": 0.9861,
"no_latin": 0.9906,
"no_phone": 0.9384,
"no_url": 0.9961
},
"per_rule_ood": {
"no_phone": 0.125,
"has_phone": 0.7667,
"no_latin": 0.4,
"no_url": 0.3917,
"no_email": 0.5417,
"has_price": 1.0,
"has_number": 0.7333,
"no_excess_punct": 0.2,
"ends_question": 0.4917,
"has_date": 0.9083
},
"per_rule_counterfactual": {
"no_phone": 1.0,
"has_phone": 0.975,
"no_latin": 0.9917,
"no_url": 1.0,
"no_email": 1.0,
"has_price": 0.975,
"has_date": 0.9917,
"no_excess_punct": 1.0,
"ends_question": 1.0
}
},
{
"model": "oddadmix/Nawah-RuleCheck-1M",
"url": "https://huggingface.co/oddadmix/Nawah-RuleCheck-1M",
"rung": "1M",
"base_model": "oddadmix/Emhotob-1M-v2",
"arch": {
"hidden_size": 32,
"layers": 4,
"heads": 4,
"kv_heads": 2,
"intermediate": 96,
"vocab": 32000,
"max_pos": 2048,
"tied_embeddings": true
},
"params_total": 1073504,
"params_non_embedding": 49504,
"weights_bytes": 4298088,
"weights_mb": 4.3,
"train": {
"lr": "1e-3",
"epochs": 10,
"warmup": 500,
"batch_size": 64,
"train_pairs": 159240,
"max_length": 224,
"precision": "bf16",
"trained_by": "this session",
"lr_sweep": [
{
"lr": "3e-4",
"epochs": 3,
"seen": 0.9897,
"unseen": 0.9518
},
{
"lr": "1e-4",
"epochs": 10,
"seen": 0.9929,
"unseen": 0.9564
},
{
"lr": "3e-4",
"epochs": 10,
"seen": 0.9964,
"unseen": 0.9849
},
{
"lr": "1e-3",
"epochs": 10,
"seen": 0.9952,
"unseen": 0.9866,
"chosen": true
},
{
"lr": "3e-3",
"epochs": 10,
"seen": 0.9914,
"unseen": 0.9707
},
{
"lr": "6e-3",
"epochs": 10,
"seen": 0.9902,
"unseen": 0.9778
},
{
"lr": "3e-4",
"epochs": 15,
"seen": 0.9946,
"unseen": 0.9697
},
{
"lr": "1e-3",
"epochs": 15,
"seen": 0.9944,
"unseen": 0.9843
},
{
"lr": "1e-3",
"epochs": 20,
"seen": 0.9955,
"unseen": 0.9814
}
],
"lr_rationale": "1e-3 is a genuine interior optimum: unseen-wording accuracy falls off on both sides (1e-4 -> 0.9564, 3e-3 -> 0.9707). The 51.8M model's published recipe (3e-4, 3 epochs) scores 0.9518 on this base - 3.5 points of the apparent capacity gap at 1M is undertraining, not size. Longer budgets at 1e-3 do not help: 15 and 20 epochs both regress on unseen wording while seen accuracy keeps climbing, which is memorisation of training phrasings."
},
"accuracy": {
"eval_seen_wording": 0.9952,
"eval_unseen_wording": 0.9864,
"counterfactual_minimal_pairs": 0.9991,
"ood_handwritten_wording": 0.7708
},
"macro_f1": {
"eval_seen_wording": 0.9948,
"eval_unseen_wording": 0.9853,
"ood_handwritten_wording": 0.7512
},
"counterfactual_mean_flip_rate": 0.9981,
"latency_ms_bs1": {
"gpu_rtx5090_median": 1.532,
"cpu_2threads_median": 0.546,
"input_tokens": 34
},
"per_rule_unseen": {
"ends_question": 1.0,
"has_city": 0.9907,
"has_date": 0.9982,
"has_number": 1.0,
"has_phone": 0.9943,
"has_price": 0.9981,
"max_words_25": 0.9,
"max_words_30": 0.9783,
"max_words_40": 0.8261,
"max_words_50": 0.8837,
"min_words_15": 0.9286,
"min_words_20": 0.8857,
"min_words_25": 0.875,
"min_words_30": 0.8205,
"no_email": 1.0,
"no_excess_punct": 0.9304,
"no_latin": 0.9984,
"no_phone": 0.9958,
"no_url": 1.0
},
"per_rule_ood": {
"no_phone": 0.5333,
"has_phone": 0.2417,
"no_latin": 0.6667,
"no_url": 1.0,
"no_email": 0.875,
"has_price": 1.0,
"has_number": 0.8583,
"no_excess_punct": 0.9917,
"ends_question": 0.5917,
"has_date": 0.95
},
"per_rule_counterfactual": {
"no_phone": 1.0,
"has_phone": 1.0,
"no_latin": 1.0,
"no_url": 1.0,
"no_email": 1.0,
"has_price": 0.9917,
"has_date": 1.0,
"no_excess_punct": 1.0,
"ends_question": 1.0
}
},
{
"model": "oddadmix/Nawah-RuleCheck-5M",
"url": "https://huggingface.co/oddadmix/Nawah-RuleCheck-5M",
"rung": "5M",
"base_model": "oddadmix/Emhotob-5M-v2",
"arch": {
"hidden_size": 128,
"layers": 5,
"heads": 4,
"kv_heads": 2,
"intermediate": 384,
"vocab": 32000,
"max_pos": 2048,
"tied_embeddings": true
},
"params_total": 5080704,
"params_non_embedding": 984704,
"weights_bytes": 20328000,
"weights_mb": 20.33,
"train": {
"lr": "1e-4",
"epochs": 10,
"warmup": 500,
"batch_size": 64,
"train_pairs": 159240,
"max_length": 224,
"precision": "bf16",
"trained_by": "this session",
"lr_sweep": [
{
"lr": "3e-4",
"epochs": 3,
"seen": 0.9973,
"unseen": 0.9834
},
{
"lr": "3e-5",
"epochs": 10,
"seen": 0.9977,
"unseen": 0.9893
},
{
"lr": "1e-4",
"epochs": 10,
"seen": 0.9979,
"unseen": 0.9947,
"chosen": true
},
{
"lr": "3e-4",
"epochs": 10,
"seen": 0.9971,
"unseen": 0.9846
},
{
"lr": "1e-3",
"epochs": 10,
"seen": 0.9925,
"unseen": 0.9786
},
{
"lr": "3e-3",
"epochs": 10,
"seen": 0.902,
"unseen": 0.88
},
{
"lr": "6e-3",
"epochs": 10,
"seen": 0.7847,
"unseen": 0.7714
},
{
"lr": "1e-4",
"epochs": 15,
"seen": 0.997,
"unseen": 0.9896
},
{
"lr": "1e-4",
"epochs": 20,
"seen": 0.9964,
"unseen": 0.9921
},
{
"lr": "3e-4",
"epochs": 15,
"seen": 0.9967,
"unseen": 0.9931
},
{
"lr": "3e-4",
"epochs": 20,
"seen": 0.9955,
"unseen": 0.9918
}
],
"lr_rationale": "The expected direction reverses here. At 500K the best LR is 6e-3; at 5M it is 1e-4 - *lower* than the 51.8M model's 3e-4, and 60x lower than the 500K's. Above 1e-3 this base degrades fast (3e-3 -> 0.88, 6e-3 -> 0.77 unseen), so copying the tiny model's recipe up the ladder would have been as wrong as copying the big model's recipe down it. 1e-4 is bracketed on both sides (3e-5 -> 0.9893, 3e-4 -> 0.9846)."
},
"accuracy": {
"eval_seen_wording": 0.9979,
"eval_unseen_wording": 0.9949,
"counterfactual_minimal_pairs": 0.9926,
"ood_handwritten_wording": 0.8642
},
"macro_f1": {
"eval_seen_wording": 0.9977,
"eval_unseen_wording": 0.9944,
"ood_handwritten_wording": 0.8559
},
"counterfactual_mean_flip_rate": 0.9852,
"latency_ms_bs1": {
"gpu_rtx5090_median": 1.796,
"cpu_2threads_median": 1.058,
"input_tokens": 34
},
"per_rule_unseen": {
"ends_question": 1.0,
"has_city": 0.9938,
"has_date": 0.9982,
"has_number": 1.0,
"has_phone": 1.0,
"has_price": 1.0,
"max_words_25": 0.95,
"max_words_30": 0.9783,
"max_words_40": 0.913,
"max_words_50": 1.0,
"min_words_15": 0.9762,
"min_words_20": 0.9429,
"min_words_25": 1.0,
"min_words_30": 0.9487,
"no_email": 1.0,
"no_excess_punct": 0.9985,
"no_latin": 0.9969,
"no_phone": 0.9639,
"no_url": 1.0
},
"per_rule_ood": {
"no_phone": 0.65,
"has_phone": 0.7,
"no_latin": 0.975,
"no_url": 1.0,
"no_email": 0.95,
"has_price": 0.9917,
"has_number": 0.975,
"no_excess_punct": 1.0,
"ends_question": 0.4,
"has_date": 1.0
},
"per_rule_counterfactual": {
"no_phone": 1.0,
"has_phone": 1.0,
"no_latin": 0.95,
"no_url": 1.0,
"no_email": 1.0,
"has_price": 0.9833,
"has_date": 1.0,
"no_excess_punct": 1.0,
"ends_question": 1.0
}
},
{
"model": "oddadmix/Nawah-RuleCheck-v2",
"url": "https://huggingface.co/oddadmix/Nawah-RuleCheck-v2",
"rung": "50M",
"base_model": "oddadmix/50M-2048-Emhotob",
"arch": {
"hidden_size": 512,
"layers": 12,
"heads": 8,
"kv_heads": 4,
"intermediate": 1408,
"vocab": 32000,
"max_pos": 2048,
"tied_embeddings": true
},
"params_total": 51787264,
"params_non_embedding": 35403264,
"weights_bytes": 207161320,
"weights_mb": 207.16,
"train": {
"lr": "1e-4",
"epochs": 10,
"warmup": 500,
"batch_size": 64,
"train_pairs": 159240,
"max_length": 224,
"precision": "bf16",
"trained_by": "this session (replaced the earlier untuned 3e-4 / 3-epoch release)",
"lr_sweep": [
{
"lr": "3e-4",
"epochs": 3,
"seen": 0.9929,
"unseen": 0.9799,
"chosen": false
},
{
"lr": "3e-5",
"epochs": 10,
"seen": 0.9973,
"unseen": 0.9964,
"chosen": false
},
{
"lr": "1e-4",
"epochs": 10,
"seen": 0.998,
"unseen": 0.998,
"chosen": true
},
{
"lr": "3e-4",
"epochs": 10,
"seen": 0.9962,
"unseen": 0.9852,
"chosen": false
},
{
"lr": "1e-4",
"epochs": 15,
"seen": 0.9985,
"unseen": 0.9959,
"chosen": false
}
],
"lr_rationale": "This rung originally shipped at 3e-4 / 3 epochs, carried over from the Emhotob SFT ladder without a sweep, and it was leaving real accuracy on the table: 0.9799 unseen against 0.9980 here, and 0.8667 against 0.8883 on hand-written wording. 1e-4 is bracketed on both sides (3e-5 -> 0.9964, 3e-4 -> 0.9852) and a longer budget does not help (15 epochs -> 0.9959). These weights are the retuned ones."
},
"accuracy": {
"eval_seen_wording": 0.998,
"eval_unseen_wording": 0.998,
"counterfactual_minimal_pairs": 0.9991,
"ood_handwritten_wording": 0.8883
},
"macro_f1": {
"eval_seen_wording": 0.9979,
"eval_unseen_wording": 0.9979,
"ood_handwritten_wording": 0.8866
},
"counterfactual_mean_flip_rate": 0.9981,
"latency_ms_bs1": {
"gpu_rtx5090_median": 3.744,
"cpu_2threads_median": 16.576,
"input_tokens": 34
},
"per_rule_unseen": {
"ends_question": 1.0,
"has_city": 0.9985,
"has_date": 1.0,
"has_number": 1.0,
"has_phone": 1.0,
"has_price": 1.0,
"max_words_25": 0.95,
"max_words_30": 1.0,
"max_words_40": 0.8696,
"max_words_50": 1.0,
"min_words_15": 0.9524,
"min_words_20": 0.9429,
"min_words_25": 0.975,
"min_words_30": 1.0,
"no_email": 1.0,
"no_excess_punct": 0.9985,
"no_latin": 1.0,
"no_phone": 0.9958,
"no_url": 1.0
},
"per_rule_ood": {
"no_phone": 0.9917,
"has_phone": 0.525,
"no_latin": 1.0,
"no_url": 1.0,
"no_email": 0.4417,
"has_price": 1.0,
"has_number": 0.9333,
"no_excess_punct": 1.0,
"ends_question": 1.0,
"has_date": 0.9917
},
"per_rule_counterfactual": {
"no_phone": 1.0,
"has_phone": 1.0,
"no_latin": 1.0,
"no_url": 1.0,
"no_email": 1.0,
"has_price": 0.9917,
"has_date": 1.0,
"no_excess_punct": 1.0,
"ends_question": 1.0
}
}
],
"diagnostics": []
}