Nawah-RuleCheck-1M / rules_common.py
oddadmix's picture
Arabic rule-checking classifier โ€” size-ladder rung
667e19a verified
Raw
History Blame Contribute Delete
18.4 kB
"""
Shared pieces for the Arabic rule-checking corpus.
Same discipline as synth_common, applied to a task whose ground truth is a set of boolean
verdicts instead of a number:
* RULES โ€” a library of rules that are *programmatically decidable*. Each carries an Arabic
statement for the prompt and a checker over the raw text.
* build_task โ€” task_id -> (axes draw, prompt). Deterministic, so the run stays resumable.
* parse_items โ€” split a raw completion into {text, analysis, verdicts, rule_ids}
* validate โ€” recompute every verdict with the checkers and reject rows the model got wrong
Why the checkers matter: if the model both writes the text and judges it, the labels are only as
good as the model, and its errors correlate with the text it produced. Here the text is generated
and the labels are *computed*, so a row is either consistent with ground truth or it is dropped.
That is the same move audit() makes for arithmetic.
Balance comes from the prompt asking for a specific satisfy/violate pattern per item. The model
often misses the target - that is fine and expected. The checkers record what the text actually
does, so a missed target costs balance, never correctness.
"""
import os
import random
import re
ITEMS_PER_TASK = 3
Q, T, A, E = "### ู†ุต", "### ุชุญู„ูŠู„", "### ุงู„ุญูƒู…", "### ู†ู‡ุงูŠุฉ"
# ---------------------------------------------------------------- checkers
AR_DIGITS = str.maketrans("ู ูกูขูฃูคูฅูฆูงูจูฉ", "0123456789")
CURRENCY = "ุฑูŠุงู„|ุฑูŠุงู„ุง|ุฑูŠุงู„ู‹ุง|ุฌู†ูŠู‡|ุฌู†ูŠู‡ุง|ุฌู†ูŠู‡ู‹ุง|ุฏุฑู‡ู…|ุฏุฑู‡ู…ุง|ุฏุฑู‡ู…ู‹ุง|ุฏูŠู†ุงุฑ|ุฏูŠู†ุงุฑุง|ุฏูŠู†ุงุฑู‹ุง|ู„ูŠุฑุฉ|ุฏูˆู„ุงุฑ"
# A short list is a correctness bug, not a shortcut: any city missing from it silently
# mislabels a row, and the label is what the whole corpus is for. The smoke test caught
# ุงู„ุฏู…ุงู… missing. Kept deliberately broad across every region in REGIONS.
CITIES = [
"ุงู„ู‚ุงู‡ุฑุฉ", "ุงู„ุฅุณูƒู†ุฏุฑูŠุฉ", "ุงู„ุฌูŠุฒุฉ", "ุงู„ู…ู†ุตูˆุฑุฉ", "ุทู†ุทุง", "ุงู„ุฒู‚ุงุฒูŠู‚", "ุฃุณูŠูˆุท", "ุงู„ู…ู†ูŠุง",
"ุจูˆุฑุณุนูŠุฏ", "ุงู„ุณูˆูŠุณ", "ุงู„ุฅุณู…ุงุนูŠู„ูŠุฉ", "ุฏู…ูŠุงุท", "ุงู„ุฃู‚ุตุฑ", "ุฃุณูˆุงู†", "ุงู„ููŠูˆู…", "ุจู†ูŠ ุณูˆูŠู",
"ุงู„ุฑูŠุงุถ", "ุฌุฏุฉ", "ู…ูƒุฉ", "ุงู„ู…ุฏูŠู†ุฉ", "ุงู„ุฏู…ุงู…", "ุงู„ุฎุจุฑ", "ุงู„ุธู‡ุฑุงู†", "ุงู„ุทุงุฆู", "ุชุจูˆูƒ",
"ุฃุจู‡ุง", "ุฎู…ูŠุณ ู…ุดูŠุท", "ุจุฑูŠุฏุฉ", "ุนู†ูŠุฒุฉ", "ุญุงุฆู„", "ู†ุฌุฑุงู†", "ุฌุงุฒุงู†", "ุงู„ุฃุญุณุงุก", "ุงู„ู‡ููˆู",
"ูŠู†ุจุน", "ุงู„ุฌุจูŠู„", "ุงู„ู‚ุทูŠู",
"ุฏุจูŠ", "ุฃุจูˆุธุจูŠ", "ุงู„ุดุงุฑู‚ุฉ", "ุนุฌู…ุงู†", "ุฑุฃุณ ุงู„ุฎูŠู…ุฉ", "ุงู„ูุฌูŠุฑุฉ", "ุฃู… ุงู„ู‚ูŠูˆูŠู†", "ุงู„ุนูŠู†",
"ุงู„ุฏูˆุญุฉ", "ุงู„ุฑูŠุงู†", "ุงู„ูˆูƒุฑุฉ", "ุงู„ุฎูˆุฑ",
"ุงู„ูƒูˆูŠุช", "ุญูˆู„ูŠ", "ุงู„ูุฑูˆุงู†ูŠุฉ", "ุงู„ุฌู‡ุฑุงุก", "ุงู„ุฃุญู…ุฏูŠ",
"ุงู„ู…ู†ุงู…ุฉ", "ุงู„ู…ุญุฑู‚", "ุงู„ุฑูุงุน", "ู…ุฏูŠู†ุฉ ุนูŠุณู‰",
"ู…ุณู‚ุท", "ุตู„ุงู„ุฉ", "ุตุญุงุฑ", "ู†ุฒูˆู‰", "ุตูˆุฑ",
"ุจุบุฏุงุฏ", "ุงู„ุจุตุฑุฉ", "ุงู„ู…ูˆุตู„", "ุฃุฑุจูŠู„", "ุงู„ู†ุฌู", "ูƒุฑุจู„ุงุก", "ุงู„ุณู„ูŠู…ุงู†ูŠุฉ", "ูƒุฑูƒูˆูƒ",
"ุงู„ู†ุงุตุฑูŠุฉ", "ุงู„ุฑู…ุงุฏูŠ", "ุฏู‡ูˆูƒ", "ุจุนู‚ูˆุจุฉ",
"ุนู…ู‘ุงู†", "ุนู…ุงู†", "ุฅุฑุจุฏ", "ุงุฑุจุฏ", "ุงู„ุฒุฑู‚ุงุก", "ุงู„ุนู‚ุจุฉ", "ุงู„ุณู„ุท", "ู…ุงุฏุจุง", "ุงู„ูƒุฑูƒ",
"ุชูˆู†ุณ", "ุตูุงู‚ุณ", "ุณูˆุณุฉ", "ุงู„ู‚ูŠุฑูˆุงู†", "ุจู†ุฒุฑุช", "ู‚ุงุจุณ", "ู†ุงุจู„", "ุงู„ู…ู†ุณุชูŠุฑ",
"ุงู„ุฏุงุฑ ุงู„ุจูŠุถุงุก", "ุงู„ุฑุจุงุท", "ู…ุฑุงูƒุด", "ูุงุณ", "ุทู†ุฌุฉ", "ุฃูƒุงุฏูŠุฑ", "ู…ูƒู†ุงุณ", "ูˆุฌุฏุฉ",
"ุชุทูˆุงู†", "ุงู„ู‚ู†ูŠุทุฑุฉ", "ุณู„ุง",
"ุงู„ุฎุฑุทูˆู…", "ุจูŠุฑูˆุช", "ุฏู…ุดู‚", "ุญู„ุจ", "ุทุฑุงุจู„ุณ", "ุจู†ุบุงุฒูŠ", "ุตู†ุนุงุก", "ุนุฏู†", "ู†ูˆุงูƒุดูˆุท",
]
DAYS = ["ุงู„ุณุจุช", "ุงู„ุฃุญุฏ", "ุงู„ุงุซู†ูŠู†", "ุงู„ุฅุซู†ูŠู†", "ุงู„ุซู„ุงุซุงุก", "ุงู„ุฃุฑุจุนุงุก", "ุงู„ุฎู…ูŠุณ", "ุงู„ุฌู…ุนุฉ"]
MONTHS = ["ูŠู†ุงูŠุฑ", "ูุจุฑุงูŠุฑ", "ู…ุงุฑุณ", "ุฃุจุฑูŠู„", "ู…ุงูŠูˆ", "ูŠูˆู†ูŠูˆ", "ูŠูˆู„ูŠูˆ", "ุฃุบุณุทุณ",
"ุณุจุชู…ุจุฑ", "ุฃูƒุชูˆุจุฑ", "ู†ูˆูู…ุจุฑ", "ุฏูŠุณู…ุจุฑ"]
def _n(text):
return text.translate(AR_DIGITS)
def _words(text):
return [w for w in re.split(r"\s+", text.strip()) if w]
def _has_phone(text):
"""A run of 8-15 digits that is not a price.
8 is the floor, not 9: Kuwait, Qatar, Bahrain, Oman and Tunisia all use 8-digit numbers, and
the smoke test surfaced 98765432 and 55443322 as phone numbers the old 9-digit threshold
silently missed. A run immediately followed by a currency word is a price, not a phone.
"""
t = _n(text)
for m in re.finditer(r"(?:\+|00)?\d[\d\s\-]{6,}\d", t):
digits = re.sub(r"\D", "", m.group(0))
if not (8 <= len(digits) <= 15):
continue
if re.match(r"\s*(?:" + CURRENCY + r")", t[m.end():m.end() + 14]):
continue # "12000000 ุฏูŠู†ุงุฑ" is a price
return True
return False
CHECKS = {
"has_price": lambda t: bool(re.search(r"\d+\s*(?:" + CURRENCY + r")", _n(t))),
"has_phone": _has_phone,
"no_phone": lambda t: not _has_phone(t),
"no_latin": lambda t: not re.search(r"[A-Za-z]", t),
"has_city": lambda t: any(c in t for c in CITIES),
"no_url": lambda t: not re.search(r"https?://|www\.|\.com|\.net|\.org", t, re.I),
"no_email": lambda t: not re.search(r"[^\s@]+@[^\s@]+\.[^\s@]+", t),
"has_number": lambda t: bool(re.search(r"\d", _n(t))),
"no_excess_punct": lambda t: not re.search(r"[!ุŸ]\s*[!ุŸ]", t),
"ends_question": lambda t: t.strip().rstrip("โ€โ€Ž").endswith(("ุŸ", "?")),
"has_date": lambda t: any(d in t for d in DAYS + MONTHS)
or bool(re.search(r"\d{1,2}\s*/\s*\d{1,2}", _n(t))),
}
# Arabic statement shown in the prompt, keyed the same way.
STATEMENTS = {
"has_price": "ูŠุฌุจ ุฃู† ูŠุฐูƒุฑ ุงู„ู†ุต ุณุนุฑู‹ุง ู…ู‚ุชุฑู†ู‹ุง ุจุนู…ู„ุฉ",
"has_phone": "ูŠุฌุจ ุฃู† ูŠุฐูƒุฑ ุงู„ู†ุต ุฑู‚ู… ู‡ุงุชู ู„ู„ุชูˆุงุตู„ (ู…ู† 8 ุฅู„ู‰ 15 ุฎุงู†ุฉ)",
"no_phone": "ูŠุฌุจ ุฃู„ุง ูŠุญุชูˆูŠ ุงู„ู†ุต ุนู„ู‰ ุฃูŠ ุฑู‚ู… ู‡ุงุชู (ู…ู† 8 ุฅู„ู‰ 15 ุฎุงู†ุฉ)",
"no_latin": "ูŠุฌุจ ุฃู„ุง ูŠุญุชูˆูŠ ุงู„ู†ุต ุนู„ู‰ ุฃูŠ ุญุฑูˆู ู„ุงุชูŠู†ูŠุฉ",
"has_city": "ูŠุฌุจ ุฃู† ูŠุฐูƒุฑ ุงู„ู†ุต ุงุณู… ู…ุฏูŠู†ุฉ",
"no_url": "ูŠุฌุจ ุฃู„ุง ูŠุญุชูˆูŠ ุงู„ู†ุต ุนู„ู‰ ุฑุงุจุท ุฃูˆ ู…ูˆู‚ุน ุฅู„ูƒุชุฑูˆู†ูŠ",
"no_email": "ูŠุฌุจ ุฃู„ุง ูŠุญุชูˆูŠ ุงู„ู†ุต ุนู„ู‰ ุจุฑูŠุฏ ุฅู„ูƒุชุฑูˆู†ูŠ",
"has_number": "ูŠุฌุจ ุฃู† ูŠุญุชูˆูŠ ุงู„ู†ุต ุนู„ู‰ ุฑู‚ู… ูˆุงุญุฏ ุนู„ู‰ ุงู„ุฃู‚ู„",
"no_excess_punct": "ูŠุฌุจ ุฃู„ุง ูŠุญุชูˆูŠ ุงู„ู†ุต ุนู„ู‰ ุนู„ุงู…ุงุช ุชุนุฌุจ ุฃูˆ ุงุณุชูู‡ุงู… ู…ุชุชุงู„ูŠุฉ",
"ends_question": "ูŠุฌุจ ุฃู† ูŠู†ุชู‡ูŠ ุงู„ู†ุต ุจุนู„ุงู…ุฉ ุงุณุชูู‡ุงู…",
"has_date": "ูŠุฌุจ ุฃู† ูŠุฐูƒุฑ ุงู„ู†ุต ุชุงุฑูŠุฎู‹ุง ุฃูˆ ูŠูˆู…ู‹ุง ุฃูˆ ุดู‡ุฑู‹ุง",
}
# Parameterised length rules are built per task, so their id carries the bound.
def _length_rule(kind, n):
rid = f"{kind}_{n}"
if kind == "min_words":
return rid, f"ูŠุฌุจ ุฃู„ุง ูŠู‚ู„ ุงู„ู†ุต ุนู† {n} ูƒู„ู…ุฉ", (lambda t, n=n: len(_words(t)) >= n)
return rid, f"ูŠุฌุจ ุฃู„ุง ูŠุฒูŠุฏ ุงู„ู†ุต ุนู† {n} ูƒู„ู…ุฉ", (lambda t, n=n: len(_words(t)) <= n)
def rule_check(rid, text):
"""-> bool. Works for both library rules and the parameterised length ones."""
if rid in CHECKS:
return CHECKS[rid](text)
m = re.fullmatch(r"(min_words|max_words)_(\d+)", rid or "")
if m:
n = int(m.group(2))
return len(_words(text)) >= n if m.group(1) == "min_words" else len(_words(text)) <= n
return None # unknown id - caller rejects the row
# ---------------------------------------------------------------- variation grid
DOC_TYPES = [
("ุฅุนู„ุงู† ู…ุจูˆุจ ู„ุจูŠุน ุณู„ุนุฉ ู…ุณุชุนู…ู„ุฉ", "ุจุงุฆุน ูุฑุฏ ูŠู†ุดุฑ ุฅุนู„ุงู†ู‹ุง"),
("ุชุฐูƒุฑุฉ ุฏุนู… ูู†ูŠ ู…ู† ุนู…ูŠู„", "ุนู…ูŠู„ ูŠุดุฑุญ ู…ุดูƒู„ุฉ ููŠ ุฎุฏู…ุฉ"),
("ุฅุนู„ุงู† ูˆุธูŠูุฉ ุดุงุบุฑุฉ", "ุดุฑูƒุฉ ุชุนู„ู† ุนู† ูˆุธูŠูุฉ"),
("ูˆุตู ู…ู†ุชุฌ ููŠ ู…ุชุฌุฑ ุฅู„ูƒุชุฑูˆู†ูŠ", "ู…ุชุฌุฑ ูŠุตู ู…ู†ุชุฌู‹ุง ู„ู„ุจูŠุน"),
("ุดูƒูˆู‰ ุนู…ูŠู„ ุนู„ู‰ ุฎุฏู…ุฉ", "ุนู…ูŠู„ ุบูŠุฑ ุฑุงุถู ูŠูƒุชุจ ุดูƒูˆู‰"),
("ุฑุณุงู„ุฉ ุชุณูˆูŠู‚ูŠุฉ ู‚ุตูŠุฑุฉ", "ู…ุชุฌุฑ ูŠุฑุณู„ ุนุฑุถู‹ุง ู„ุนู…ู„ุงุฆู‡"),
("ุฅุนู„ุงู† ุนู† ุนู‚ุงุฑ ู„ู„ุฅูŠุฌุงุฑ", "ู…ุงู„ูƒ ูŠุนุฑุถ ุดู‚ุฉ ุฃูˆ ู…ุญู„ู‹ุง"),
("ู…ู†ุดูˆุฑ ููŠ ู…ุฌู…ูˆุนุฉ ุจูŠุน ูˆุดุฑุงุก", "ุดุฎุต ูŠู†ุดุฑ ููŠ ู…ุฌู…ูˆุนุฉ"),
("ุทู„ุจ ุนุฑุถ ุณุนุฑ ู…ู† ู…ูˆุฑู‘ุฏ", "ู…ูˆุธู ู…ุดุชุฑูŠุงุช ูŠุทู„ุจ ุนุฑุถู‹ุง"),
("ุฅุนู„ุงู† ุนู† ุฏูˆุฑุฉ ุชุฏุฑูŠุจูŠุฉ", "ู…ุฑูƒุฒ ุชุฏุฑูŠุจ ูŠุนู„ู† ุนู† ุฏูˆุฑุฉ"),
("ุจู„ุงุบ ุนู† ุนุทู„ ููŠ ู…ุฑูู‚", "ุณุงูƒู† ูŠุจู„ุบ ุนู† ุนุทู„"),
("ุนุฑุถ ุฎุฏู…ุฉ ุชูˆุตูŠู„", "ู…ู†ุฏูˆุจ ูŠุนุฑุถ ุฎุฏู…ุชู‡"),
]
REGIONS = ["ู…ุตุฑ", "ุงู„ุณุนูˆุฏูŠุฉ", "ุงู„ุฅู…ุงุฑุงุช", "ู‚ุทุฑ", "ุงู„ูƒูˆูŠุช", "ุงู„ุนุฑุงู‚", "ุงู„ุฃุฑุฏู†", "ุชูˆู†ุณ", "ุงู„ู…ุบุฑุจ"]
# "Is this a city?" has no exact answer - district, town and neighbourhood shade into each other,
# and the smoke test produced ุงู„ุณุงู„ู…ูŠุฉ and ุงู„ุตู„ูŠุจุฎุงุช (real Kuwaiti districts) alongside ุงู„ุฃุฑุฏู†
# (a country the model called a city). A rule library that claims decidable labels cannot hold a
# predicate that needs world knowledge, so the rule now *names* acceptable cities per region and
# the checker's gazetteer is a superset of every list shown.
REGION_CITIES = {
"ู…ุตุฑ": ["ุงู„ู‚ุงู‡ุฑุฉ", "ุงู„ุฅุณูƒู†ุฏุฑูŠุฉ", "ุงู„ุฌูŠุฒุฉ", "ุงู„ู…ู†ุตูˆุฑุฉ", "ุฃุณูŠูˆุท"],
"ุงู„ุณุนูˆุฏูŠุฉ": ["ุงู„ุฑูŠุงุถ", "ุฌุฏุฉ", "ุงู„ุฏู…ุงู…", "ุงู„ุฎุจุฑ", "ุงู„ุทุงุฆู"],
"ุงู„ุฅู…ุงุฑุงุช": ["ุฏุจูŠ", "ุฃุจูˆุธุจูŠ", "ุงู„ุดุงุฑู‚ุฉ", "ุนุฌู…ุงู†", "ุงู„ุนูŠู†"],
"ู‚ุทุฑ": ["ุงู„ุฏูˆุญุฉ", "ุงู„ุฑูŠุงู†", "ุงู„ูˆูƒุฑุฉ", "ุงู„ุฎูˆุฑ"],
"ุงู„ูƒูˆูŠุช": ["ุงู„ูƒูˆูŠุช", "ุญูˆู„ูŠ", "ุงู„ูุฑูˆุงู†ูŠุฉ", "ุงู„ุฌู‡ุฑุงุก", "ุงู„ุฃุญู…ุฏูŠ"],
"ุงู„ุนุฑุงู‚": ["ุจุบุฏุงุฏ", "ุงู„ุจุตุฑุฉ", "ุงู„ู…ูˆุตู„", "ุฃุฑุจูŠู„", "ุงู„ู†ุฌู"],
"ุงู„ุฃุฑุฏู†": ["ุนู…ู‘ุงู†", "ุฅุฑุจุฏ", "ุงู„ุฒุฑู‚ุงุก", "ุงู„ุนู‚ุจุฉ", "ุงู„ุณู„ุท"],
"ุชูˆู†ุณ": ["ุชูˆู†ุณ", "ุตูุงู‚ุณ", "ุณูˆุณุฉ", "ุงู„ู‚ูŠุฑูˆุงู†", "ุจู†ุฒุฑุช"],
"ุงู„ู…ุบุฑุจ": ["ุงู„ุฏุงุฑ ุงู„ุจูŠุถุงุก", "ุงู„ุฑุจุงุท", "ู…ุฑุงูƒุด", "ูุงุณ", "ุทู†ุฌุฉ"],
}
# Rules that co-determine each other give a free label: a model can score them without reading
# the text. no_phone is the exact negation of has_phone, and both has_phone and has_price force
# has_number to pass because a phone number and a price both contain digits. Never draw a
# conflicting pair into the same task.
CONFLICTS = [
("has_phone", "no_phone"),
("has_phone", "has_number"),
("has_price", "has_number"),
]
def _conflicts(rid, chosen_ids):
return any((rid == a and b in chosen_ids) or (rid == b and a in chosen_ids)
for a, b in CONFLICTS)
PROMPT = """ุฃู†ุช ุชูƒุชุจ ุจูŠุงู†ุงุช ุชุฏุฑูŠุจูŠุฉ ู„ู†ู…ูˆุฐุฌ ูŠุชุญู‚ู‚ ู…ู† ู…ุทุงุจู‚ุฉ ุงู„ู†ุตูˆุต ู„ู‚ูˆุงุนุฏ ู…ุญุฏุฏุฉ.
ุงูƒุชุจ {k} ุฃู…ุซู„ุฉ ู…ุณุชู‚ู„ุฉ. ูƒู„ ู…ุซุงู„ ุนุจุงุฑุฉ ุนู† ู†ุต ุนุฑุจูŠ ูˆุงู‚ุนูŠ ู…ู† ู†ูˆุน: {doc_type} ({doc_hint}) ููŠ {region}.
ุงู„ู‚ูˆุงุนุฏ ุงู„ู…ุทู„ูˆุจ ูุญุตู‡ุง:
{rules_block}
ู‡ุฏู ุงู„ูƒุชุงุจุฉ (ู„ู„ุชู†ูˆูŠุน ูู‚ุทุŒ ูˆู„ูŠุณ ู‡ูˆ ุงู„ุญูƒู…):
{pattern_block}
โš ๏ธ ุงู„ุฃู‡ู… ููŠ ู‡ุฐู‡ ุงู„ู…ู‡ู…ุฉ:
ุจุนุฏ ุฃู† ุชูƒุชุจ ุงู„ู†ุตุŒ ุงูุญุตู‡ ู…ู† ุฌุฏูŠุฏ ูƒุฃู†ูƒ ุชุฑุงู‡ ู„ุฃูˆู„ ู…ุฑุฉ ูˆุงุญูƒู… ุนู„ู‰ ู…ุง ูŠุญุชูˆูŠู‡ **ูุนู„ู‹ุง**ุŒ ูˆู„ูŠุณ ุนู„ู‰ ู…ุง
ูƒู†ุช ุชู†ูˆูŠ ูƒุชุงุจุชู‡. ุฅู† ุฎุงู„ู ุงู„ู†ุต ู‡ุฏู ุงู„ูƒุชุงุจุฉ ุฃุนู„ุงู‡ุŒ ูุงูƒุชุจ ุงู„ุญู‚ูŠู‚ุฉ ูƒู…ุง ู‡ูŠ ููŠ ู‚ุณู… ุงู„ุญูƒู…. ุงู„ุญูƒู… ูˆุตู
ู„ู…ุง ููŠ ุงู„ู†ุตุŒ ูˆู„ูŠุณ ุชูƒุฑุงุฑู‹ุง ู„ู„ุชุนู„ูŠู…ุงุช.
ุงูƒุชุจ ูƒู„ ู…ุซุงู„ ุจู‡ุฐุง ุงู„ุดูƒู„ ุจุงู„ุถุจุท:
{Q}
(ุงู„ู†ุต ุงู„ุนุฑุจูŠ ู‡ู†ุงุŒ ู…ู† ุณุทุฑ ุฅู„ู‰ ุซู„ุงุซุฉ ุฃุณุทุฑ)
{T}
(ุณุทุฑ ูˆุงุญุฏ ู„ูƒู„ ู‚ุงุนุฏุฉ: ุงู‚ุชุจุณ ุงู„ุฏู„ูŠู„ ุงู„ุญุฑููŠ ู…ู† ุงู„ู†ุต ุจูŠู† ุนู„ุงู…ุชูŠ ุชู†ุตูŠุตุŒ ุซู… ุงุฐูƒุฑ ุงู„ู†ุชูŠุฌุฉ)
{A}
{verdict_template}
{E}
ู‚ูˆุงุนุฏ ู…ู‡ู…ุฉ:
- ููŠ ุณุทุฑ ุงู„ุญูƒู… ุงูƒุชุจ ุงู„ู…ุนุฑู‘ู ุงู„ุฅู†ุฌู„ูŠุฒูŠ ู„ู„ู‚ุงุนุฏุฉ ูƒู…ุง ู‡ูˆุŒ ุซู… ูƒู„ู…ุฉ ูˆุงุญุฏุฉ: ู…ุทุงุจู‚ ุฃูˆ ู…ุฎุงู„ู.
- "ู…ุทุงุจู‚" ุชุนู†ูŠ ุฃู† ุงู„ู†ุต ูŠุญู‚ู‚ ู…ุง ุชุทู„ุจู‡ ุงู„ู‚ุงุนุฏุฉ. "ู…ุฎุงู„ู" ุชุนู†ูŠ ุฃู†ู‡ ู„ุง ูŠุญู‚ู‚ู‡. ู„ุง ุชุฎู„ุท ุจูŠู†ู‡ุง ูˆุจูŠู†
ู‡ุฏู ุงู„ูƒุชุงุจุฉ.
- ุงู„ู†ุต ู†ูุณู‡ ูŠุฌุจ ุฃู† ูŠูƒูˆู† ุทุจูŠุนูŠู‹ุง ูˆูˆุงู‚ุนูŠู‹ุงุŒ ูˆู„ูŠุณ ู…ุตู†ูˆุนู‹ุง ู„ูŠุจุฏูˆ ูƒุงุฎุชุจุงุฑ.
- ู†ูˆู‘ุน ุงู„ุฃุณู…ุงุก ูˆุงู„ุฃุฑู‚ุงู… ูˆุงู„ุชูุงุตูŠู„ ุจูŠู† ุงู„ุฃู…ุซู„ุฉ.
- ู„ุง ุชูƒุชุจ ุฃูŠ ุดูŠุก ุฎุงุฑุฌ ุงู„ูˆุณูˆู…."""
def build_task(task_id: int, seed: int = 1234):
"""task_id -> (axes, prompt). Deterministic, so a resumed run redraws identical prompts."""
rng = random.Random(seed * 1_000_003 + task_id)
doc_type, doc_hint = rng.choice(DOC_TYPES)
region = rng.choice(REGIONS)
n_rules = rng.choice([3, 3, 4])
pool = [(rid, STATEMENTS[rid], None) for rid in CHECKS]
if rng.random() < 0.55: # roughly half the tasks carry a length bound
kind = rng.choice(["min_words", "max_words"])
n = rng.choice([15, 20, 25, 30]) if kind == "min_words" else rng.choice([25, 30, 40, 50])
pool.append(_length_rule(kind, n))
# Draw one at a time so a conflicting rule can be skipped rather than poisoning the task.
chosen = []
for cand in rng.sample(pool, len(pool)):
if len(chosen) >= n_rules:
break
if not _conflicts(cand[0], {c[0] for c in chosen}):
chosen.append(cand)
n_rules = len(chosen)
examples = "ุŒ ".join(REGION_CITIES.get(region, CITIES[:5]))
chosen = [(rid, (stmt + f" ู…ู† ู‡ุฐู‡ ุงู„ู…ุฏู†: {examples}") if rid == "has_city" else stmt, chk)
for rid, stmt, chk in chosen]
# Force at least one satisfied and one violated so the labels do not collapse to all-pass.
pattern = [True] * n_rules
n_violate = rng.choice([1, 1, 2])
for i in rng.sample(range(n_rules), min(n_violate, n_rules - 1)):
pattern[i] = False
rules_block = "\n".join(f"- {rid}: {stmt}" for rid, stmt, _ in chosen)
# Deliberately different vocabulary from the verdict words (ู…ุทุงุจู‚/ู…ุฎุงู„ู): when the writing
# goal and the verdict share wording, the model echoes the instruction instead of inspecting
# the text it wrote. The smoke test showed exactly that - "ู…ุฎุงู„ู because the text DOES
# contain a phone number" on a rule requiring one.
want = [rid for (rid, _, _), ok in zip(chosen, pattern) if ok]
avoid = [rid for (rid, _, _), ok in zip(chosen, pattern) if not ok]
parts = []
if want:
parts.append("- ุงุฌุนู„ ุงู„ู†ุต ูŠุณุชูˆููŠ: " + "ุŒ ".join(want))
if avoid:
parts.append("- ูˆุงุฌุนู„ู‡ ู„ุง ูŠุณุชูˆููŠ: " + "ุŒ ".join(avoid))
pattern_block = "\n".join(parts)
# Listed in a different order than the writing goal, so position carries no hint.
order = sorted((rid for rid, _, _ in chosen))
verdict_template = "\n".join(f"{rid}: ..." for rid in order)
axes = {"doc_type": doc_type, "region": region,
"rule_ids": [rid for rid, _, _ in chosen],
"target_pattern": pattern}
prompt = PROMPT.format(k=ITEMS_PER_TASK, doc_type=doc_type, doc_hint=doc_hint, region=region,
rules_block=rules_block, pattern_block=pattern_block,
verdict_template=verdict_template, Q=Q, T=T, A=A, E=E)
return axes, prompt
# ---------------------------------------------------------------- parsing
BLOCK_RE = re.compile(
re.escape(Q) + r"(?P<text>.*?)" + re.escape(T) + r"(?P<analysis>.*?)"
+ re.escape(A) + r"(?P<verdicts>.*?)" + re.escape(E), re.S)
VERDICT_RE = re.compile(r"^\s*([A-Za-z_][A-Za-z0-9_]*)\s*[:๏ผš]\s*(ู…ุทุงุจู‚|ู…ุฎุงู„ู)\s*$", re.M)
def parse_items(raw: str):
"""-> [{text, analysis, verdicts: {rule_id: bool}, rule_ids: [...]}]"""
items = []
for m in BLOCK_RE.finditer(raw or ""):
verdicts, order = {}, []
for v in VERDICT_RE.finditer(m.group("verdicts")):
rid = v.group(1)
if rid not in verdicts:
order.append(rid)
verdicts[rid] = (v.group(2) == "ู…ุทุงุจู‚")
items.append({"text": m.group("text").strip(),
"analysis": m.group("analysis").strip(),
"verdicts": verdicts, "rule_ids": order})
return items
# ---------------------------------------------------------------- validation
def validate(item, min_text=20, max_text=900, min_analysis=20, max_analysis=1200):
"""
Full accept/reject for one parsed item -> (ok, reason).
The decisive check is the last one: every verdict the model stated is recomputed from the
text by the rule's own checker. A row whose analysis reads fluently but whose verdicts do not
match what the text actually does is exactly the failure mode a fluency check cannot catch.
"""
text, analysis, verdicts = item["text"], item["analysis"], item["verdicts"]
if not (min_text <= len(text) <= max_text):
return False, "text_length"
if not (min_analysis <= len(analysis) <= max_analysis):
return False, "analysis_length"
if not verdicts:
return False, "no_verdicts"
if len(verdicts) < 2:
return False, "too_few_rules"
if any(tag in text for tag in (Q, T, A, E)):
return False, "tag_leak"
for rid, stated in verdicts.items():
truth = rule_check(rid, text)
if truth is None:
return False, f"unknown_rule:{rid}"
if truth != stated:
return False, f"verdict_mismatch:{rid}"
return True, "ok"
def truth_verdicts(item):
"""Ground truth recomputed from the text - what a reward function should compare against."""
return {rid: rule_check(rid, item["text"]) for rid in item["rule_ids"]}
def dedup_key(text: str) -> str:
"""Numbers masked out, so the same template with different values collapses to one key."""
return re.sub(r"\d+", "#", re.sub(r"\W+", "", text))