Text Classification
Transformers
Safetensors
Arabic
llama
arabic
rule-checking
compliance
moderation
tiny-model
on-device
text-embeddings-inference
Instructions to use oddadmix/Nawah-RuleCheck-v2 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use oddadmix/Nawah-RuleCheck-v2 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-classification", model="oddadmix/Nawah-RuleCheck-v2")# Load model directly from transformers import AutoTokenizer, AutoModelForSequenceClassification tokenizer = AutoTokenizer.from_pretrained("oddadmix/Nawah-RuleCheck-v2") model = AutoModelForSequenceClassification.from_pretrained("oddadmix/Nawah-RuleCheck-v2", device_map="auto") - Notebooks
- Google Colab
- Kaggle
| """ | |
| Turn the rule-checking cache into a binary classification set, with each rule stated in many | |
| different ways. | |
| One (text, rule) pair is one example. The label is the *computed* verdict from rules_common, so | |
| it does not depend on how the rule is worded - which is what makes paraphrasing free supervision. | |
| Two independent splits, because there are two ways this model can fail to generalise: | |
| * by text - split on task_id, so no text appears in both train and eval; | |
| * by wording - each rule's paraphrase pool is split too, and a fifth of the phrasings are held | |
| out of training entirely. | |
| Eval is emitted twice per pair: once with a phrasing the model trained on (measuring text | |
| generalisation) and once with a phrasing it has never seen (measuring semantic generalisation). | |
| The first trained model scored 0.996 on its single training wording and 0.57-0.76 on rephrasings; | |
| without the second eval that gap is invisible. | |
| """ | |
| import json | |
| import os | |
| import random | |
| import re | |
| import sys | |
| from collections import Counter | |
| from pathlib import Path | |
| sys.path.insert(0, "/opt/projects/math-gsk8/rules") | |
| import rules_common as rc | |
| CACHE = Path(os.environ.get("CACHE", "/opt/projects/math-gsk8/rules/out_rules/generations.jsonl")) | |
| OUT = Path(os.environ.get("OUT_DIR", "data")); OUT.mkdir(exist_ok=True) | |
| PARA = Path(os.environ.get("PARA", "rule_paraphrases.json")) | |
| EVAL_FRAC = float(os.environ.get("EVAL_FRAC", 0.04)) | |
| PARA_EVAL_FRAC = float(os.environ.get("PARA_EVAL_FRAC", 0.2)) | |
| SEED = 42 | |
| def family(rid): | |
| m = re.fullmatch(r"(min_words|max_words)_(\d+)", rid) | |
| return (m.group(1), m.group(2)) if m else (rid, None) | |
| def base_statement(rid, region): | |
| fam, n = family(rid) | |
| if n: | |
| return ("ูุฌุจ ุฃูุง ููู ุงููุต ุนู {n} ููู ุฉ" if fam == "min_words" | |
| else "ูุฌุจ ุฃูุง ูุฒูุฏ ุงููุต ุนู {n} ููู ุฉ").replace("{n}", n) | |
| s = rc.STATEMENTS.get(rid, rid) | |
| if rid == "has_city": | |
| s += " ู ู ูุฐู ุงูู ุฏู: " + "ุ ".join(rc.REGION_CITIES.get(region, rc.CITIES[:5])) | |
| return s | |
| def statement(rid, region): # kept for callers that want the canonical form | |
| return base_statement(rid, region) | |
| def build_pools(): | |
| """-> {family: (train_phrasings, eval_phrasings)} as templates, {n} still unfilled.""" | |
| d = json.load(open(PARA, encoding="utf-8")) | |
| rng = random.Random(SEED) | |
| pools = {} | |
| for fam, paras in d["paraphrases"].items(): | |
| items = sorted(set(paras)) | |
| rng.shuffle(items) | |
| k = max(1, int(len(items) * PARA_EVAL_FRAC)) | |
| # The canonical statement stays on the train side - it is the wording the corpus itself | |
| # was generated with, and holding it out would test something nobody will ever type. | |
| pools[fam] = (items[k:] + [d["originals"][fam]], items[:k]) | |
| return pools | |
| def render(fam_stmt, rid, region): | |
| _, n = family(rid) | |
| s = fam_stmt.replace("{n}", n) if n else fam_stmt | |
| if rid == "has_city" and "ู ู ูุฐู ุงูู ุฏู" not in s: | |
| s += " ู ู ูุฐู ุงูู ุฏู: " + "ุ ".join(rc.REGION_CITIES.get(region, rc.CITIES[:5])) | |
| return s | |
| def main(): | |
| pools = build_pools() | |
| pairs, tasks = [], set() | |
| for line in open(CACHE, encoding="utf-8"): | |
| rec = json.loads(line) | |
| for item in rc.parse_items(rec["raw"]): | |
| if not rc.validate(item)[0]: | |
| continue | |
| tasks.add(rec["task_id"]) | |
| for rid, truth in rc.truth_verdicts(item).items(): | |
| pairs.append({"task_id": rec["task_id"], "rule": rid, | |
| "text": item["text"], "label": int(bool(truth)), | |
| "region": rec["axes"]["region"]}) | |
| rng = random.Random(SEED) | |
| tl = sorted(tasks); rng.shuffle(tl) | |
| eval_tasks = set(tl[: max(1, int(len(tl) * EVAL_FRAC))]) | |
| train_raw = [p for p in pairs if p["task_id"] not in eval_tasks] | |
| eval_raw = [p for p in pairs if p["task_id"] in eval_tasks] | |
| train_texts = {p["text"] for p in train_raw} | |
| eval_raw = [p for p in eval_raw if p["text"] not in train_texts] | |
| train, ev = [], [] | |
| for p in train_raw: | |
| fam, _ = family(p["rule"]) | |
| pool = pools.get(fam, ([base_statement(p["rule"], p["region"])], []))[0] | |
| r = dict(p); r["statement"] = render(rng.choice(pool), p["rule"], p["region"]) | |
| r["para_split"] = "train" | |
| train.append(r) | |
| for p in eval_raw: | |
| fam, _ = family(p["rule"]) | |
| seen, unseen = pools.get(fam, ([base_statement(p["rule"], p["region"])], [])) | |
| for tag, pool in (("seen", seen), ("unseen", unseen or seen)): | |
| r = dict(p); r["statement"] = render(rng.choice(pool), p["rule"], p["region"]) | |
| r["para_split"] = tag | |
| ev.append(r) | |
| rng.shuffle(train) | |
| for name, rows in (("train", train), ("eval", ev)): | |
| with open(OUT / f"{name}.jsonl", "w", encoding="utf-8") as fh: | |
| for r in rows: | |
| fh.write(json.dumps(r, ensure_ascii=False) + "\n") | |
| n_seen = sum(1 for r in ev if r["para_split"] == "seen") | |
| print(f"tasks {len(tl):,} eval tasks {len(eval_tasks):,}") | |
| print(f"train pairs {len(train):,} {Counter(r['label'] for r in train)}") | |
| print(f"eval pairs {len(ev):,} ({n_seen:,} seen-wording + {len(ev)-n_seen:,} unseen-wording)") | |
| print(f"text overlap: {len({r['text'] for r in train} & {r['text'] for r in ev})}") | |
| print(f"distinct phrasings in train: {len({r['statement'] for r in train}):,}") | |
| print(f"wording overlap train/eval-unseen: " | |
| f"{len({r['statement'] for r in train} & {r['statement'] for r in ev if r['para_split']=='unseen'})}") | |
| if __name__ == "__main__": | |
| main() | |