Spaces:
Running
Running
Upload 26 files
Browse files- .gitattributes +1 -0
- README.md +12 -5
- README_AR.md +5 -0
- _headers +7 -0
- alignment.py +180 -0
- app.py +181 -0
- benchmark_format.py +115 -0
- camelbert_adapter.py +103 -0
- config.json +7 -0
- detector.py +178 -0
- examples.json +227 -0
- hadith.idx.gz +3 -0
- hadith.json +3 -0
- idgham.py +71 -0
- index.html +516 -17
- index_builder.py +163 -0
- islamic_unified_dataset.jsonl +0 -0
- islamiceval_dev_subset.jsonl +0 -0
- llm_client.py +96 -0
- normalization.py +117 -0
- quran.idx.gz +3 -0
- quran.json +0 -0
- retrieval.py +187 -0
- scanner.py +349 -0
- similarity.py +121 -0
- ui.py +534 -0
- verifier.py +571 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
hadith.json filter=lfs diff=lfs merge=lfs -text
|
README.md
CHANGED
|
@@ -1,10 +1,17 @@
|
|
| 1 |
---
|
| 2 |
-
title:
|
| 3 |
-
emoji:
|
| 4 |
-
colorFrom:
|
| 5 |
-
colorTo:
|
| 6 |
sdk: static
|
|
|
|
| 7 |
pinned: false
|
|
|
|
|
|
|
| 8 |
---
|
| 9 |
|
| 10 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
+
title: Quran & Hadith Verification
|
| 3 |
+
emoji: 🕌
|
| 4 |
+
colorFrom: green
|
| 5 |
+
colorTo: yellow
|
| 6 |
sdk: static
|
| 7 |
+
app_file: index.html
|
| 8 |
pinned: false
|
| 9 |
+
license: mit
|
| 10 |
+
short_description: Verify Quran and Hadith quotations
|
| 11 |
---
|
| 12 |
|
| 13 |
+
# Quran & Hadith Verification
|
| 14 |
+
|
| 15 |
+
Finds Quran verses and Hadith in text and verifies them against their sources.
|
| 16 |
+
|
| 17 |
+
Arabic version: [README_AR.md](README_AR.md)
|
README_AR.md
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# التحقق من هلوسة القرآن والحديث وتصحيحها
|
| 2 |
+
|
| 3 |
+
يكتشف الآيات والأحاديث في أي نص، ويعرض الدليل، ويقترح التصحيح من نص المصدر.
|
| 4 |
+
يعمل الكود كاملًا داخل متصفحك (بايثون عبر Pyodide)، ولا يُرسل النص الملصوق إلى أي خادم.
|
| 5 |
+
النسخة الإنجليزية: [README.md](README.md).
|
_headers
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/*
|
| 2 |
+
X-Content-Type-Options: nosniff
|
| 3 |
+
Referrer-Policy: no-referrer
|
| 4 |
+
/index/*
|
| 5 |
+
Cache-Control: public, max-age=86400
|
| 6 |
+
/config.json
|
| 7 |
+
Cache-Control: no-store
|
alignment.py
ADDED
|
@@ -0,0 +1,180 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Word-level sequence alignment between a quotation and a source text.
|
| 2 |
+
|
| 3 |
+
* Phonetic skeletons (``normalization.aligned_words``) make Uthmani / modern spelling and diacritics irrelevant for
|
| 4 |
+
matching, while every substituted, missing, extra or re-ordered word still shows up as an edit operation.
|
| 5 |
+
* A dynamic sliding window locates the best-matching region inside long sources (Hadith with chains of narrators).
|
| 6 |
+
* The gap threshold scales with quotation length: a short quotation tolerates no stray words, a long one a few.
|
| 7 |
+
* Diacritic conflicts are reported separately as notes, because spelling of harakat differs legitimately between prints.
|
| 8 |
+
"""
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
from difflib import SequenceMatcher
|
| 12 |
+
from math import ceil
|
| 13 |
+
from typing import Container, Dict, List, Optional, Sequence, Tuple
|
| 14 |
+
|
| 15 |
+
from idgham import apply_idgham
|
| 16 |
+
from normalization import aligned_words, vowel_signature
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
def lcs_length(a: Sequence[str], b: Sequence[str]) -> int:
|
| 20 |
+
m, n = len(a), len(b)
|
| 21 |
+
if m == 0 or n == 0:
|
| 22 |
+
return 0
|
| 23 |
+
if m < n:
|
| 24 |
+
a, b, m, n = b, a, n, m
|
| 25 |
+
prev = [0] * (n + 1)
|
| 26 |
+
for i in range(m):
|
| 27 |
+
curr = [0] * (n + 1)
|
| 28 |
+
for j in range(n):
|
| 29 |
+
curr[j + 1] = prev[j] + 1 if a[i] == b[j] else max(curr[j], prev[j + 1])
|
| 30 |
+
prev = curr
|
| 31 |
+
return prev[n]
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def allowed_gap(n_tokens: int) -> int:
|
| 35 |
+
"""Dynamic gap threshold: how many differing words still count as 'the same passage, altered'."""
|
| 36 |
+
return 0 if n_tokens <= 3 else max(1, ceil(0.2 * n_tokens))
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
def find_window(quote: List[str], source: List[str], gap: int) -> Tuple[int, int]:
|
| 40 |
+
"""Best source region for the quotation: windows of ``len(quote)+gap`` ranked by shared words, refined by LCS."""
|
| 41 |
+
n, window = len(quote), len(quote) + gap
|
| 42 |
+
if len(source) <= window + 2 * gap + 8:
|
| 43 |
+
return 0, len(source)
|
| 44 |
+
quote_set = set(quote)
|
| 45 |
+
prefix = [0]
|
| 46 |
+
for word in source:
|
| 47 |
+
prefix.append(prefix[-1] + (word in quote_set))
|
| 48 |
+
starts = sorted(range(len(source) - window + 1), key=lambda i: prefix[i + window] - prefix[i], reverse=True)[:8]
|
| 49 |
+
best_start = max(starts, key=lambda i: lcs_length(quote, source[i : i + window]))
|
| 50 |
+
return max(0, best_start - gap), min(len(source), best_start + window + gap)
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
def contains_sequence(source: Sequence[str], quote: Sequence[str]) -> bool:
|
| 54 |
+
"""True if ``quote`` appears as a contiguous run of whole words in ``source``."""
|
| 55 |
+
n = len(quote)
|
| 56 |
+
return n > 0 and any(source[i : i + n] == list(quote) for i in range(len(source) - n + 1))
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
def locate_region(q_norm: List[str], s_norm: List[str], gap: int, slack: Optional[int] = None):
|
| 60 |
+
"""Region of the source that corresponds to the quotation (matched part plus slack for edge differences).
|
| 61 |
+
|
| 62 |
+
Returns ``(lo, hi, matcher)`` where the matcher compares the quotation with ``s_norm[lo:hi]``."""
|
| 63 |
+
slack = gap if slack is None else slack
|
| 64 |
+
lo, hi = find_window(q_norm, s_norm, gap)
|
| 65 |
+
matcher = SequenceMatcher(None, q_norm, s_norm[lo:hi], autojunk=False)
|
| 66 |
+
blocks = [b for b in matcher.get_matching_blocks() if b.size]
|
| 67 |
+
if blocks:
|
| 68 |
+
first, last = blocks[0], blocks[-1]
|
| 69 |
+
base = lo
|
| 70 |
+
lo = base + max(0, first.b - first.a - slack)
|
| 71 |
+
hi = min(hi, base + last.b + last.size + (len(q_norm) - last.a - last.size) + slack)
|
| 72 |
+
matcher = SequenceMatcher(None, q_norm, s_norm[lo:hi], autojunk=False)
|
| 73 |
+
return lo, hi, matcher
|
| 74 |
+
|
| 75 |
+
|
| 76 |
+
def best_region(quote: str, source: str) -> str:
|
| 77 |
+
"""The part of ``source`` that a (possibly partial) quotation refers to, as original words.
|
| 78 |
+
|
| 79 |
+
Scoring a snippet against the whole verse / Hadith would punish it for being shorter than its source; comparing it
|
| 80 |
+
with this region keeps the score about what was actually quoted."""
|
| 81 |
+
q_pairs, s_pairs = aligned_words(quote), aligned_words(source)
|
| 82 |
+
if not q_pairs or len(s_pairs) <= len(q_pairs) + allowed_gap(len(q_pairs)) + 2:
|
| 83 |
+
return source
|
| 84 |
+
q_norm, s_norm = [p[1] for p in q_pairs], [p[1] for p in s_pairs]
|
| 85 |
+
lo, hi, _ = locate_region(q_norm, s_norm, allowed_gap(len(q_norm)), slack=0)
|
| 86 |
+
return " ".join(p[0] for p in s_pairs[lo:hi]) if hi > lo else source
|
| 87 |
+
|
| 88 |
+
|
| 89 |
+
def _diacritic_notes(quote_words: List[str], source_words: List[str], idgham_words: List[str]) -> List[dict]:
|
| 90 |
+
"""Words whose written harakat contradict the source (missing harakat are fine; idgham spelling is accepted)."""
|
| 91 |
+
notes = []
|
| 92 |
+
for q_word, s_word, g_word in zip(quote_words, source_words, idgham_words):
|
| 93 |
+
q_sig = vowel_signature(q_word)
|
| 94 |
+
if not any(marks for _, marks in q_sig):
|
| 95 |
+
continue
|
| 96 |
+
conflict = True
|
| 97 |
+
for variant in (s_word, g_word):
|
| 98 |
+
v_sig = vowel_signature(variant)
|
| 99 |
+
if len(v_sig) == len(q_sig) and all(set(qm) <= set(vm) for (_, qm), (_, vm) in zip(q_sig, v_sig)):
|
| 100 |
+
conflict = False
|
| 101 |
+
break
|
| 102 |
+
if conflict:
|
| 103 |
+
notes.append({"word": q_word, "source_word": s_word})
|
| 104 |
+
return notes
|
| 105 |
+
|
| 106 |
+
|
| 107 |
+
def _orthographic_variant(q_tokens: Sequence[str], s_tokens: Sequence[str], vocabulary: Optional[Container[str]]) -> bool:
|
| 108 |
+
"""Same letters written differently: spacing (ياأيها / يا أيها) or a medial alef (إسحق / إسحاق).
|
| 109 |
+
|
| 110 |
+
The alef variant is tolerated only when the quoted spelling is not itself a word of the corpus (so قتل / قاتل stay a
|
| 111 |
+
real difference)."""
|
| 112 |
+
if "".join(q_tokens) == "".join(s_tokens):
|
| 113 |
+
return True
|
| 114 |
+
if vocabulary is not None and len(q_tokens) == 1 and len(s_tokens) == 1:
|
| 115 |
+
q, s = q_tokens[0], s_tokens[0]
|
| 116 |
+
return q.replace("ا", "") == s.replace("ا", "") and q not in vocabulary
|
| 117 |
+
return False
|
| 118 |
+
|
| 119 |
+
|
| 120 |
+
def align(quote: str, source: str, vocabulary: Optional[Container[str]] = None) -> Dict[str, object]:
|
| 121 |
+
"""Align ``quote`` to the best matching region of ``source`` and describe every difference.
|
| 122 |
+
|
| 123 |
+
``vocabulary``: phonetic-skeleton words of the corpus; enables the medial-alef spelling tolerance."""
|
| 124 |
+
q_pairs, s_pairs = aligned_words(quote), aligned_words(source)
|
| 125 |
+
idgham_pairs = aligned_words(apply_idgham(source)) if len(source) < 20000 else s_pairs
|
| 126 |
+
if len(idgham_pairs) != len(s_pairs):
|
| 127 |
+
idgham_pairs = s_pairs
|
| 128 |
+
q_orig, q_norm = [p[0] for p in q_pairs], [p[1] for p in q_pairs]
|
| 129 |
+
s_orig, s_norm = [p[0] for p in s_pairs], [p[1] for p in s_pairs]
|
| 130 |
+
g_orig = [p[0] for p in idgham_pairs]
|
| 131 |
+
|
| 132 |
+
gap = allowed_gap(len(q_norm))
|
| 133 |
+
lo, hi, matcher = locate_region(q_norm, s_norm, gap)
|
| 134 |
+
|
| 135 |
+
raw_ops = [(tag, i1, i2, j1 + lo, j2 + lo) for tag, i1, i2, j1, j2 in matcher.get_opcodes()]
|
| 136 |
+
# Source words before / after a partial quotation are not errors unless the quotation itself contains them elsewhere
|
| 137 |
+
# (a moved word); drop such edge insertions.
|
| 138 |
+
deleted = {w for tag, i1, i2, _, _ in raw_ops if tag in ("delete", "replace") for w in q_norm[i1:i2]}
|
| 139 |
+
for position in (0, -1):
|
| 140 |
+
if len(raw_ops) > 1 and raw_ops[position][0] == "insert" and not (set(s_norm[raw_ops[position][3]:raw_ops[position][4]]) & deleted):
|
| 141 |
+
raw_ops.pop(position)
|
| 142 |
+
|
| 143 |
+
ops, quote_side, source_side, missing, extra, notes = [], [], [], [], [], []
|
| 144 |
+
mismatches = 0
|
| 145 |
+
orthographic = 0
|
| 146 |
+
for position, (tag, i1, i2, j1, j2) in enumerate(raw_ops):
|
| 147 |
+
source_idx = list(range(j1, j2))
|
| 148 |
+
if tag == "replace" and _orthographic_variant(q_norm[i1:i2], s_norm[j1:j2], vocabulary):
|
| 149 |
+
tag, orthographic = "equal", orthographic + 1 # spelling-only difference, not a different word
|
| 150 |
+
if tag == "replace" and position in (0, len(raw_ops) - 1) and len(raw_ops) > 1 and (j2 - j1) > (i2 - i1):
|
| 151 |
+
# at the edges the slack may have pulled in unrelated source words: keep only words the quotation also uses
|
| 152 |
+
keep = [j for j in source_idx if s_norm[j] in set(q_norm[i1:i2]) | deleted]
|
| 153 |
+
source_idx = keep
|
| 154 |
+
tag = "replace" if keep else "delete"
|
| 155 |
+
ops.append({"op": tag, "span": " ".join(q_orig[i1:i2]), "source": " ".join(s_orig[j] for j in source_idx)})
|
| 156 |
+
if tag == "equal":
|
| 157 |
+
notes += _diacritic_notes(q_orig[i1:i2], s_orig[j1:j2], g_orig[j1:j2])
|
| 158 |
+
else:
|
| 159 |
+
mismatches += max(i2 - i1, len(source_idx))
|
| 160 |
+
quote_side += q_norm[i1:i2]
|
| 161 |
+
source_side += [s_norm[j] for j in source_idx]
|
| 162 |
+
extra += q_orig[i1:i2] if tag in ("delete", "replace") else []
|
| 163 |
+
missing += [s_orig[j] for j in source_idx] if tag in ("insert", "replace") else []
|
| 164 |
+
|
| 165 |
+
matched = sum(i2 - i1 for tag, i1, i2, _, _ in raw_ops if tag == "equal")
|
| 166 |
+
exact = bool(q_norm) and mismatches == 0
|
| 167 |
+
return {
|
| 168 |
+
"word_similarity": round(2 * matched / max(len(q_norm) + (raw_ops[-1][4] - raw_ops[0][3] if raw_ops else 0), 1), 3),
|
| 169 |
+
"word_diff": ops,
|
| 170 |
+
"missing_from_span": missing,
|
| 171 |
+
"extra_in_span": extra,
|
| 172 |
+
"source_excerpt": " ".join(op["source"] for op in ops if op["source"]),
|
| 173 |
+
"exact": exact,
|
| 174 |
+
"mismatches": mismatches,
|
| 175 |
+
"allowed_gap": gap,
|
| 176 |
+
"near": (not exact) and matched > 0 and mismatches <= gap,
|
| 177 |
+
"reordered": bool(quote_side) and sorted(quote_side) == sorted(source_side),
|
| 178 |
+
"diacritic_notes": notes,
|
| 179 |
+
"orthographic_variants": orthographic,
|
| 180 |
+
}
|
app.py
ADDED
|
@@ -0,0 +1,181 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""التحقق من هلوسة القرآن والحديث وتصحيحها: Gradio interface.
|
| 2 |
+
|
| 3 |
+
python app.py # http://127.0.0.1:7860
|
| 4 |
+
|
| 5 |
+
Mode A verifies pasted text (rule + corpus detection and, when configured, CAMeLBERT, merged silently). Mode B ("ask then verify") sends the question to a single
|
| 6 |
+
pre-configured OpenAI client and verifies the answer; the key comes from the OPENAI_API_KEY environment variable only
|
| 7 |
+
(never from the form, never from the repository). The in-browser page (build_static_space.py) shares these handlers.
|
| 8 |
+
"""
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
import json
|
| 12 |
+
import logging
|
| 13 |
+
import os
|
| 14 |
+
import threading
|
| 15 |
+
from pathlib import Path
|
| 16 |
+
from typing import List, Optional
|
| 17 |
+
|
| 18 |
+
import ui
|
| 19 |
+
from camelbert_adapter import analyze_hybrid, entities_to_spans, query_hosted_model
|
| 20 |
+
from llm_client import LLMError, generate, sanitize_answer
|
| 21 |
+
from verifier import MAX_INPUT_CHARS, IslamicContentVerifier
|
| 22 |
+
|
| 23 |
+
logger = logging.getLogger(__name__)
|
| 24 |
+
|
| 25 |
+
EXAMPLES_PATH = Path(__file__).resolve().parent / "demo" / "examples.json"
|
| 26 |
+
|
| 27 |
+
_pipeline: Optional[IslamicContentVerifier] = None
|
| 28 |
+
_lock = threading.Lock()
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
def get_pipeline() -> IslamicContentVerifier:
|
| 32 |
+
"""Created once. The Quran index loads immediately; the Hadith index loads lazily (see ``warm_in_background``)."""
|
| 33 |
+
global _pipeline
|
| 34 |
+
with _lock:
|
| 35 |
+
if _pipeline is None:
|
| 36 |
+
_pipeline = IslamicContentVerifier()
|
| 37 |
+
return _pipeline
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
def warm_in_background() -> None:
|
| 41 |
+
threading.Thread(target=lambda: get_pipeline().retriever.warm(), daemon=True).start()
|
| 42 |
+
|
| 43 |
+
|
| 44 |
+
def load_examples(path: Path = EXAMPLES_PATH) -> List[dict]:
|
| 45 |
+
try:
|
| 46 |
+
with open(path, encoding="utf-8") as handle:
|
| 47 |
+
return json.load(handle)
|
| 48 |
+
except (OSError, json.JSONDecodeError):
|
| 49 |
+
logger.exception("Could not load demo examples from %s", path)
|
| 50 |
+
return []
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
def _analyze(text: str, entities_json: str = "") -> dict:
|
| 54 |
+
"""One analysis path: the bundled rule + corpus detector and (when configured) the fine-tuned CAMeLBERT model run
|
| 55 |
+
together and are merged, with no user-facing choice. In the browser the page calls the hosted model and passes its
|
| 56 |
+
entities; locally ``ICV_HF_MODEL`` does the same server-side. Any model failure is silent: the bundled detector alone
|
| 57 |
+
gives the answer."""
|
| 58 |
+
pipeline = get_pipeline()
|
| 59 |
+
model_spans: list = []
|
| 60 |
+
try:
|
| 61 |
+
if entities_json:
|
| 62 |
+
model_spans = entities_to_spans(text, json.loads(entities_json))
|
| 63 |
+
elif os.environ.get("ICV_HF_MODEL", "").strip():
|
| 64 |
+
model_spans = query_hosted_model(text, os.environ["ICV_HF_MODEL"].strip(), os.environ.get("HF_TOKEN", ""))
|
| 65 |
+
except (RuntimeError, ValueError, TypeError):
|
| 66 |
+
logger.warning("Hosted model unavailable; using the bundled detector only")
|
| 67 |
+
model_spans = []
|
| 68 |
+
return analyze_hybrid(pipeline, text, model_spans)
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
def verify_text(text: str, entities_json: str = "") -> str:
|
| 72 |
+
"""Mode A. Never raises: problems become Arabic notices."""
|
| 73 |
+
if not text or not text.strip():
|
| 74 |
+
return ui.render_message("الرجاء إدخال نص للتحقق منه.", "warn")
|
| 75 |
+
try:
|
| 76 |
+
return ui.render_results(_analyze(text, entities_json))
|
| 77 |
+
except ValueError:
|
| 78 |
+
return ui.render_message(f"النص طويل جدًا (الحد الأقصى {MAX_INPUT_CHARS} حرف).", "warn")
|
| 79 |
+
except Exception:
|
| 80 |
+
logger.exception("Verification failed")
|
| 81 |
+
return ui.render_message("حدث خطأ غير متوقع أثناء التحقق.", "bad")
|
| 82 |
+
|
| 83 |
+
|
| 84 |
+
def verify_generated_answer(answer: str, entities_json: str = "") -> str:
|
| 85 |
+
"""Verify a model answer and show it above the report (also used by the in-browser page)."""
|
| 86 |
+
try:
|
| 87 |
+
answer = sanitize_answer(answer)
|
| 88 |
+
return ui.render_results(_analyze(answer, entities_json), generated_answer=answer)
|
| 89 |
+
except ValueError:
|
| 90 |
+
return ui.render_message(f"الإجابة طويلة جدًا (الحد الأقصى {MAX_INPUT_CHARS} حرف).", "warn")
|
| 91 |
+
except Exception:
|
| 92 |
+
logger.exception("Verification of the generated answer failed")
|
| 93 |
+
return ui.render_message("تعذّر التحقق من إجابة النموذج.", "bad")
|
| 94 |
+
|
| 95 |
+
|
| 96 |
+
def ask_then_verify(prompt: str) -> str:
|
| 97 |
+
"""Mode B: ask the pre-configured model, then verify every quotation in its answer."""
|
| 98 |
+
try:
|
| 99 |
+
answer = generate(None, prompt)
|
| 100 |
+
except LLMError as exc:
|
| 101 |
+
return ui.render_message(str(exc), "warn")
|
| 102 |
+
return verify_generated_answer(answer)
|
| 103 |
+
|
| 104 |
+
|
| 105 |
+
def build_interface():
|
| 106 |
+
import gradio as gr
|
| 107 |
+
|
| 108 |
+
examples = load_examples()
|
| 109 |
+
|
| 110 |
+
def next_example(index: int):
|
| 111 |
+
if not examples:
|
| 112 |
+
return "", 0
|
| 113 |
+
return examples[index % len(examples)]["text"], (index + 1) % len(examples)
|
| 114 |
+
|
| 115 |
+
samples = [e for e in examples if e.get("question")]
|
| 116 |
+
|
| 117 |
+
def next_ask_example(index: int):
|
| 118 |
+
if not samples:
|
| 119 |
+
return "", 0
|
| 120 |
+
return samples[index % len(samples)]["question"], (index + 1) % len(samples)
|
| 121 |
+
|
| 122 |
+
def saved_answer_check(question: str):
|
| 123 |
+
"""Verify the saved answer that belongs to the example question (no live model call)."""
|
| 124 |
+
for sample in samples:
|
| 125 |
+
if sample["question"] == question:
|
| 126 |
+
return verify_generated_answer(sample["text"])
|
| 127 |
+
return ui.render_message("اختر مثالًا أولًا.", "warn")
|
| 128 |
+
|
| 129 |
+
theme = gr.themes.Base(primary_hue="emerald", neutral_hue="stone")
|
| 130 |
+
with gr.Blocks(title=ui.APP_TITLE, css=ui.CSS, theme=theme, head=f"<script>{ui.COPY_JS}</script>") as demo:
|
| 131 |
+
gr.HTML(ui.HERO)
|
| 132 |
+
with gr.Tabs():
|
| 133 |
+
with gr.Tab("تحقّق مباشر"):
|
| 134 |
+
example_index = gr.State(0)
|
| 135 |
+
text_input = gr.Textbox(label="النص المراد التحقق منه", lines=9, max_lines=24, placeholder=ui.PLACEHOLDER,
|
| 136 |
+
rtl=True, elem_classes="input-area")
|
| 137 |
+
with gr.Row():
|
| 138 |
+
verify_button = gr.Button("تحقّق من النص", variant="primary", scale=3)
|
| 139 |
+
example_button = gr.Button("جرّب مثالًا", variant="secondary", scale=2)
|
| 140 |
+
results = gr.HTML(elem_classes="results")
|
| 141 |
+
verify_button.click(verify_text, inputs=text_input, outputs=results)
|
| 142 |
+
example_button.click(next_example, inputs=example_index, outputs=[text_input, example_index]).then(
|
| 143 |
+
verify_text, inputs=text_input, outputs=results)
|
| 144 |
+
with gr.Tab("اسأل ثم تحقّق"):
|
| 145 |
+
gr.HTML('<div class="icv"><div class="notice">اكتب سؤالًا، وسيجيب عنه النظام مباشرةً، ثم يفحص كل آية '
|
| 146 |
+
'وحديث ورد في الإجابة ويعرض الأخطاء والتصحيحات.</div></div>')
|
| 147 |
+
prompt = gr.Textbox(label="سؤالك", lines=3, placeholder=ui.PROMPT_PLACEHOLDER, rtl=True, elem_classes="input-area")
|
| 148 |
+
with gr.Row():
|
| 149 |
+
ask_button = gr.Button("اسأل ثم تحقّق", variant="primary", scale=3)
|
| 150 |
+
ask_example_button = gr.Button("جرّب مثالًا", variant="secondary", scale=2)
|
| 151 |
+
answer_results = gr.HTML(elem_classes="results")
|
| 152 |
+
ask_button.click(ask_then_verify, inputs=prompt, outputs=answer_results)
|
| 153 |
+
ask_example_state = gr.State(0)
|
| 154 |
+
ask_example_button.click(next_ask_example, inputs=ask_example_state, outputs=[prompt, ask_example_state]).then(
|
| 155 |
+
saved_answer_check, inputs=prompt, outputs=answer_results)
|
| 156 |
+
gr.HTML(ui.DISCLAIMER)
|
| 157 |
+
return demo
|
| 158 |
+
|
| 159 |
+
|
| 160 |
+
def _load_dotenv(path: Path = Path(__file__).resolve().parent / ".env") -> None:
|
| 161 |
+
"""Minimal ``.env`` reader for local runs (no extra dependency); existing environment variables win."""
|
| 162 |
+
try:
|
| 163 |
+
lines = path.read_text(encoding="utf-8").splitlines()
|
| 164 |
+
except OSError:
|
| 165 |
+
return
|
| 166 |
+
for line in lines:
|
| 167 |
+
name, sep, value = line.strip().partition("=")
|
| 168 |
+
if sep and name and not name.startswith("#") and value.strip():
|
| 169 |
+
os.environ.setdefault(name.strip(), value.strip().strip('"').strip("'"))
|
| 170 |
+
|
| 171 |
+
|
| 172 |
+
def main() -> None:
|
| 173 |
+
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
|
| 174 |
+
_load_dotenv()
|
| 175 |
+
get_pipeline()
|
| 176 |
+
warm_in_background()
|
| 177 |
+
build_interface().queue().launch(share=os.environ.get("ICV_SHARE") == "1")
|
| 178 |
+
|
| 179 |
+
|
| 180 |
+
if __name__ == "__main__":
|
| 181 |
+
main()
|
benchmark_format.py
ADDED
|
@@ -0,0 +1,115 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""IslamicEval 2025 (Subtask 1) input / output formats.
|
| 2 |
+
|
| 3 |
+
The pipeline result is converted to the three official submission files:
|
| 4 |
+
|
| 5 |
+
1A Question_ID <TAB> Span_Start <TAB> Span_End <TAB> Span_Type Span_Type = Ayah | Hadith | No_Spans (end exclusive)
|
| 6 |
+
1B Sequence_ID <TAB> Label Sequence_ID = <QID>_<annotation number>; Correct | Incorrect
|
| 7 |
+
1C Sequence_ID <TAB> Correction Sequence_ID = <QID>_<start>_<end>; only spans judged Incorrect;
|
| 8 |
+
'خطأ' when there is no source / no safe correction
|
| 9 |
+
|
| 10 |
+
The layout follows the competition-format routine of the project's research notebook, which was written against the
|
| 11 |
+
organisers' scoring code. It is **not** re-validated against the live CodaBench scorer here (the organisers' repository
|
| 12 |
+
could not be fetched when this module was written), so compare with the official sample files before a real submission.
|
| 13 |
+
|
| 14 |
+
Command line: python benchmark_format.py responses.xml outputs/ (XML: <Question><ID/><Response/></Question> ...)
|
| 15 |
+
"""
|
| 16 |
+
from __future__ import annotations
|
| 17 |
+
|
| 18 |
+
import csv
|
| 19 |
+
import io
|
| 20 |
+
import json
|
| 21 |
+
import re
|
| 22 |
+
import sys
|
| 23 |
+
from pathlib import Path
|
| 24 |
+
from typing import Dict, Iterable, List, Sequence, Tuple
|
| 25 |
+
|
| 26 |
+
NO_SOURCE = "خطأ"
|
| 27 |
+
FILE_NAMES = ("task1A_predictions.tsv", "task1B_predictions.tsv", "task1C_predictions.tsv")
|
| 28 |
+
Row = Tuple
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
def _clean(value) -> str:
|
| 32 |
+
return str(value).replace("\t", " ").replace("\r", " ").replace("\n", " ").strip()
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
def competition_rows(result: dict, qid: str) -> Tuple[List[Row], List[Row], List[Row]]:
|
| 36 |
+
"""``(rows_1A, rows_1B, rows_1C)`` for one analysed response."""
|
| 37 |
+
a, b, c = [], [], []
|
| 38 |
+
spans = result["spans"]
|
| 39 |
+
if not spans:
|
| 40 |
+
a.append((qid, 0, 0, "No_Spans"))
|
| 41 |
+
for k, span in enumerate(spans, 1):
|
| 42 |
+
a.append((qid, span["start"], span["end"], span["type"]))
|
| 43 |
+
verdict = span["verification"]["verdict"]
|
| 44 |
+
b.append((f"{qid}_{k}", verdict))
|
| 45 |
+
if verdict == "Incorrect":
|
| 46 |
+
correction = span["correction"]["text"] if (span["status"] == "CORRECTED" and span["correction"]) else NO_SOURCE
|
| 47 |
+
c.append((f"{qid}_{span['start']}_{span['end']}", _clean(correction)))
|
| 48 |
+
return a, b, c
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
def to_tsv(rows: Iterable[Row]) -> str:
|
| 52 |
+
return "".join("\t".join(str(x) for x in row) + "\n" for row in rows)
|
| 53 |
+
|
| 54 |
+
|
| 55 |
+
def official_payload(result: dict, qid: str = "Q1") -> Dict[str, str]:
|
| 56 |
+
"""The three files as TSV strings (used by the browser page for downloads)."""
|
| 57 |
+
return {name: to_tsv(rows) for name, rows in zip(FILE_NAMES, competition_rows(result, qid))}
|
| 58 |
+
|
| 59 |
+
|
| 60 |
+
def write_competition_files(all_rows: Sequence[Sequence[Row]], out_dir: str = "outputs") -> List[Path]:
|
| 61 |
+
out = Path(out_dir)
|
| 62 |
+
out.mkdir(parents=True, exist_ok=True)
|
| 63 |
+
paths = []
|
| 64 |
+
for name, rows in zip(FILE_NAMES, all_rows):
|
| 65 |
+
path = out / name
|
| 66 |
+
path.write_text(to_tsv(rows), encoding="utf-8", newline="")
|
| 67 |
+
paths.append(path)
|
| 68 |
+
return paths
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
def load_responses(xml_path: str) -> Dict[str, str]:
|
| 72 |
+
"""``{question id: response}`` from the competition XML."""
|
| 73 |
+
data = Path(xml_path).read_text(encoding="utf-8")
|
| 74 |
+
out = {}
|
| 75 |
+
for block in re.findall(r"<Question>(.*?)</Question>", data, re.S):
|
| 76 |
+
ident, response = re.search(r"<ID>(.*?)</ID>", block, re.S), re.search(r"<Response>(.*?)</Response>", block, re.S)
|
| 77 |
+
if ident and response:
|
| 78 |
+
out[ident.group(1).strip()] = response.group(1)
|
| 79 |
+
return out
|
| 80 |
+
|
| 81 |
+
|
| 82 |
+
def read_tsv(path: str) -> List[dict]:
|
| 83 |
+
"""Rows of a TSV with a header line (gold files)."""
|
| 84 |
+
with open(path, encoding="utf-8", newline="") as handle:
|
| 85 |
+
rows = list(csv.reader(handle, delimiter="\t"))
|
| 86 |
+
return [dict(zip(rows[0], row)) for row in rows[1:]] if rows else []
|
| 87 |
+
|
| 88 |
+
|
| 89 |
+
def run_on_texts(pipeline, texts: Dict[str, str], out_dir: str = "outputs") -> Dict[str, dict]:
|
| 90 |
+
"""Analyse ``{qid: text}``, write the three official TSVs plus ``full_report.json`` into ``out_dir``."""
|
| 91 |
+
acc: Tuple[List[Row], List[Row], List[Row]] = ([], [], [])
|
| 92 |
+
results = {}
|
| 93 |
+
for qid, text in texts.items():
|
| 94 |
+
result = pipeline.analyze(text)
|
| 95 |
+
results[qid] = result
|
| 96 |
+
for target, part in zip(acc, competition_rows(result, qid)):
|
| 97 |
+
target.extend(part)
|
| 98 |
+
write_competition_files(acc, out_dir)
|
| 99 |
+
(Path(out_dir) / "full_report.json").write_text(json.dumps(results, ensure_ascii=False, indent=2), encoding="utf-8")
|
| 100 |
+
return results
|
| 101 |
+
|
| 102 |
+
|
| 103 |
+
def main(argv: Sequence[str]) -> int:
|
| 104 |
+
if len(argv) < 2:
|
| 105 |
+
print(__doc__)
|
| 106 |
+
return 2
|
| 107 |
+
from verifier import IslamicContentVerifier
|
| 108 |
+
|
| 109 |
+
results = run_on_texts(IslamicContentVerifier(), load_responses(argv[1]), argv[2] if len(argv) > 2 else "outputs")
|
| 110 |
+
print(f"{len(results)} responses written to {argv[2] if len(argv) > 2 else 'outputs'}/")
|
| 111 |
+
return 0
|
| 112 |
+
|
| 113 |
+
|
| 114 |
+
if __name__ == "__main__":
|
| 115 |
+
sys.exit(main(sys.argv))
|
camelbert_adapter.py
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Adapter for a fine-tuned CAMeLBERT-MSA span detector (Subtask 1A) hosted on Hugging Face, plus a simulation mode.
|
| 2 |
+
|
| 3 |
+
Two ways to get detection spans into the verification pipeline:
|
| 4 |
+
|
| 5 |
+
* ``entities_to_spans`` turns the output of a Hugging Face *token-classification* call (aggregated entities, or raw
|
| 6 |
+
``B-Ayah`` / ``I-Ayah`` / ``B-Hadith`` / ``I-Hadith`` tokens) into ``[{label, start, end, score}]``.
|
| 7 |
+
The browser page calls the hosted model and hands the entities to ``verify_with_entities``.
|
| 8 |
+
* ``simulate_spans`` a stand-in used when no hosted model is configured or reachable: the bundled hybrid detector
|
| 9 |
+
produces the spans, and they are labelled as simulated so nobody mistakes them for model output.
|
| 10 |
+
|
| 11 |
+
No model weights live in this repository. Train with ``research/train_detector.py`` (``--model CAMeL-Lab/bert-base-arabic-camelbert-msa``),
|
| 12 |
+
push the result to the Hub and set its id in the page configuration (see docs/DEPLOYMENT.md).
|
| 13 |
+
"""
|
| 14 |
+
from __future__ import annotations
|
| 15 |
+
|
| 16 |
+
import json
|
| 17 |
+
import re
|
| 18 |
+
import urllib.error
|
| 19 |
+
import urllib.request
|
| 20 |
+
from typing import Dict, Iterable, List, Optional
|
| 21 |
+
|
| 22 |
+
HF_ENDPOINT = "https://router.huggingface.co/hf-inference/models/{model}" # configurable; not verified against the live service
|
| 23 |
+
|
| 24 |
+
LABEL_ALIASES = {"ayah": "Ayah", "quran": "Ayah", "label_1": "Ayah", "label_2": "Ayah",
|
| 25 |
+
"hadith": "Hadith", "label_3": "Hadith", "label_4": "Hadith"}
|
| 26 |
+
DEFAULT_MIN_SCORE = 0.5
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
def _label_of(raw: str) -> Optional[str]:
|
| 30 |
+
name = re.sub(r"^[BI]-", "", str(raw or ""), flags=re.I).strip().lower()
|
| 31 |
+
return LABEL_ALIASES.get(name)
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def entities_to_spans(text: str, entities: Iterable[dict], min_score: float = DEFAULT_MIN_SCORE,
|
| 35 |
+
max_gap: int = 1) -> List[dict]:
|
| 36 |
+
"""Merge Hugging Face token-classification output into character spans.
|
| 37 |
+
|
| 38 |
+
Works with ``aggregation_strategy="simple"`` output (``entity_group``) and with raw output (``entity`` = ``B-Ayah`` ...).
|
| 39 |
+
Pieces of the same label separated by at most ``max_gap`` characters are joined; low-score pieces are dropped."""
|
| 40 |
+
pieces = []
|
| 41 |
+
for item in entities or []:
|
| 42 |
+
label = _label_of(item.get("entity_group") or item.get("entity"))
|
| 43 |
+
start, end = item.get("start"), item.get("end")
|
| 44 |
+
if label is None or start is None or end is None or end <= start:
|
| 45 |
+
continue
|
| 46 |
+
pieces.append({"label": label, "start": int(start), "end": int(end), "score": float(item.get("score", 1.0)),
|
| 47 |
+
"begin": str(item.get("entity", "")).upper().startswith("B-")})
|
| 48 |
+
pieces.sort(key=lambda p: (p["start"], p["end"]))
|
| 49 |
+
merged: List[dict] = []
|
| 50 |
+
for piece in pieces:
|
| 51 |
+
last = merged[-1] if merged else None
|
| 52 |
+
if last and last["label"] == piece["label"] and not piece["begin"] and piece["start"] - last["end"] <= max_gap:
|
| 53 |
+
last["end"] = max(last["end"], piece["end"])
|
| 54 |
+
last["scores"].append(piece["score"])
|
| 55 |
+
else:
|
| 56 |
+
merged.append({"label": piece["label"], "start": piece["start"], "end": piece["end"], "scores": [piece["score"]]})
|
| 57 |
+
spans = []
|
| 58 |
+
for item in merged:
|
| 59 |
+
start, end = item["start"], item["end"]
|
| 60 |
+
while start < end and text[start].isspace():
|
| 61 |
+
start += 1
|
| 62 |
+
while end > start and text[end - 1].isspace():
|
| 63 |
+
end -= 1
|
| 64 |
+
score = sum(item["scores"]) / len(item["scores"])
|
| 65 |
+
if end > start and score >= min_score:
|
| 66 |
+
spans.append({"label": item["label"], "start": start, "end": end, "score": round(score, 4)})
|
| 67 |
+
return spans
|
| 68 |
+
|
| 69 |
+
|
| 70 |
+
def simulate_spans(pipeline, text: str) -> List[dict]:
|
| 71 |
+
"""Spans from the bundled hybrid detector, flagged as simulated (``score`` is ``None``)."""
|
| 72 |
+
return [{"label": s.label, "start": s.start, "end": s.end, "score": None} for s in pipeline.detect(text)]
|
| 73 |
+
|
| 74 |
+
|
| 75 |
+
def analyze_with_spans(pipeline, text: str, spans: List[dict], engine: str = "camelbert") -> dict:
|
| 76 |
+
"""Run verification + correction on externally detected spans and record which engine produced them."""
|
| 77 |
+
result = pipeline.analyze_spans(text, [{"label": s["label"], "start": s["start"], "end": s["end"]} for s in spans])
|
| 78 |
+
scores: Dict[tuple, Optional[float]] = {(s["start"], s["end"]): s.get("score") for s in spans}
|
| 79 |
+
for report in result["spans"]:
|
| 80 |
+
score = scores.get((report["start"], report["end"]))
|
| 81 |
+
report["detection"] = {"backend": engine, "confidence": None if score is None else round(score, 4)}
|
| 82 |
+
result["detector"] = engine
|
| 83 |
+
return result
|
| 84 |
+
|
| 85 |
+
|
| 86 |
+
def query_hosted_model(text: str, model: str, token: str = "", endpoint: str = HF_ENDPOINT, timeout: float = 30.0) -> List[dict]:
|
| 87 |
+
"""Call a hosted token-classification model (server-side use: the local Gradio app and tests with a mock server).
|
| 88 |
+
|
| 89 |
+
Returns ``entities_to_spans`` output. Raises ``RuntimeError`` with a short message on any failure so the caller can fall
|
| 90 |
+
back to the simulation."""
|
| 91 |
+
headers = {"Content-Type": "application/json"}
|
| 92 |
+
if token:
|
| 93 |
+
headers["Authorization"] = f"Bearer {token}"
|
| 94 |
+
body = json.dumps({"inputs": text, "parameters": {"aggregation_strategy": "simple"}}).encode("utf-8")
|
| 95 |
+
request = urllib.request.Request(endpoint.format(model=model), data=body, headers=headers, method="POST")
|
| 96 |
+
try:
|
| 97 |
+
with urllib.request.urlopen(request, timeout=timeout) as response:
|
| 98 |
+
payload = json.loads(response.read().decode("utf-8"))
|
| 99 |
+
except (urllib.error.URLError, TimeoutError, json.JSONDecodeError) as exc:
|
| 100 |
+
raise RuntimeError("hosted model unavailable") from exc
|
| 101 |
+
if not isinstance(payload, list):
|
| 102 |
+
raise RuntimeError("unexpected response from the hosted model")
|
| 103 |
+
return entities_to_spans(text, payload)
|
config.json
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"askEndpoint": "https://icv-ask-proxy.ghada-islamic-verifier-2026.workers.dev",
|
| 3 |
+
"hf": {
|
| 4 |
+
"model": "",
|
| 5 |
+
"endpoint": "https://router.huggingface.co/hf-inference/models/{model}"
|
| 6 |
+
}
|
| 7 |
+
}
|
detector.py
ADDED
|
@@ -0,0 +1,178 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Quotation detection (Subtask 1A): finds Quran / Hadith quotations in a text and their boundaries.
|
| 2 |
+
|
| 3 |
+
The detector is rule-based: quotation marks and brackets mark candidate segments. The *type* of a segment (Ayah /
|
| 4 |
+
Hadith) is decided by the reference corpora themselves (word coverage against the Quran and the six Hadith books).
|
| 5 |
+
Introductory phrases such as "قال الله تعالى" or "قال رسول الله ﷺ" are only a tie-breaker / fallback hint, never a
|
| 6 |
+
requirement: a quotation is found and typed even when no such phrase precedes it. No model weights are loaded."""
|
| 7 |
+
from __future__ import annotations
|
| 8 |
+
|
| 9 |
+
import re
|
| 10 |
+
from dataclasses import dataclass
|
| 11 |
+
from typing import List, Optional
|
| 12 |
+
|
| 13 |
+
from normalization import normalize_for_matching, normalize_lenient, normalize_strict, tokenize
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 17 |
+
@dataclass
|
| 18 |
+
class DetectedSpan:
|
| 19 |
+
start: int
|
| 20 |
+
end: int # exclusive
|
| 21 |
+
label: str # 'Ayah' | 'Hadith'
|
| 22 |
+
confidence: Optional[float] # None for the rule backend
|
| 23 |
+
source: str # 'rules' | 'given'
|
| 24 |
+
text: str = ""
|
| 25 |
+
hint: Optional[str] = None # type suggested by an introductory phrase, if any (may disagree with ``label``)
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
QUOTE_CHARS = " \t\r\n\"“”«»﴿﴾{}()[]"
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
def trim_span(text: str, start: int, end: int):
|
| 32 |
+
"""Drop whitespace and quotation marks at both edges (gold spans exclude the quote marks)."""
|
| 33 |
+
while start < end and text[start] in QUOTE_CHARS:
|
| 34 |
+
start += 1
|
| 35 |
+
while end > start and text[end - 1] in QUOTE_CHARS + ".،,؛:":
|
| 36 |
+
end -= 1
|
| 37 |
+
return start, end
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
AYAH_TRIGGERS = [
|
| 41 |
+
"قال الله", "قوله تعالى", "قال تعالى", "يقول الله", "يقول تعالى", "قال سبحانه", "قوله سبحانه", "في كتابه",
|
| 42 |
+
"سورة", "الآية", "الاية", "الآيات", "القرآن", "القران", "كتاب الله", "عز وجل", "جل جلاله", "فقال تعالى",
|
| 43 |
+
"ذكر الله", "﴿",
|
| 44 |
+
]
|
| 45 |
+
HADITH_TRIGGERS = [
|
| 46 |
+
"رسول الله", "النبي", "صلى الله عليه وسلم", "ﷺ", "عليه الصلاة والسلام", "حديث", "رواه", "روى", "متفق عليه",
|
| 47 |
+
"الحديث", "فقال", "قال ص", "صلى الله عليه", "وسلم",
|
| 48 |
+
]
|
| 49 |
+
_FORMULA_WORDS = {
|
| 50 |
+
normalize_for_matching(w)
|
| 51 |
+
for w in "قال قالت رسول الله صلى عليه وسلم النبي تعالى سبحانه عز وجل فقال يقول الكريم الشريف الحديث الآية روى رواه عن أن أنه البخاري ومسلم".split()
|
| 52 |
+
}
|
| 53 |
+
_BRACKET_PAIRS = [("“", "”"), ("«", "»"), ("﴿", "﴾"), ("{", "}"), ("(", ")"), ("[", "]")]
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
class RuleDetector:
|
| 57 |
+
"""Quotation-mark and trigger-phrase detector with an optional corpus lookup (no training, no GPU)."""
|
| 58 |
+
|
| 59 |
+
def __init__(self, retriever=None, min_words: int = 3, context_chars: int = 110,
|
| 60 |
+
min_corpus_cov: float = 0.6, decouple_triggers: bool = True, quoted_min_cov: float = 0.5,
|
| 61 |
+
quoted_min_cov_hadith: float = 0.75) -> None:
|
| 62 |
+
"""``decouple_triggers``: type a delimited segment from the corpora first and use introductory phrases only as a
|
| 63 |
+
hint (needs a retriever). ``quoted_min_cov``: lower coverage bar for text the author explicitly delimited."""
|
| 64 |
+
self.kb, self.min_words, self.context_chars, self.min_corpus_cov = retriever, min_words, context_chars, min_corpus_cov
|
| 65 |
+
self.decouple_triggers = decouple_triggers and retriever is not None
|
| 66 |
+
self.quoted_min_cov, self.quoted_min_cov_hadith = quoted_min_cov, quoted_min_cov_hadith
|
| 67 |
+
|
| 68 |
+
@staticmethod
|
| 69 |
+
def _segments(text: str):
|
| 70 |
+
segments = []
|
| 71 |
+
quote_positions = [m.start() for m in re.finditer('"', text)]
|
| 72 |
+
if len(quote_positions) % 2 == 0:
|
| 73 |
+
pairs = zip(quote_positions[0::2], quote_positions[1::2]) # opening/closing pairs
|
| 74 |
+
else: # a stray quote: fall back to every consecutive pair
|
| 75 |
+
pairs = zip(quote_positions, quote_positions[1:])
|
| 76 |
+
for a, b in pairs:
|
| 77 |
+
segments.append((a + 1, b))
|
| 78 |
+
for opener, closer in _BRACKET_PAIRS:
|
| 79 |
+
for m in re.finditer(re.escape(opener) + r"(.*?)" + re.escape(closer), text, re.S):
|
| 80 |
+
segments.append((m.start(1), m.end(1)))
|
| 81 |
+
return segments
|
| 82 |
+
|
| 83 |
+
@staticmethod
|
| 84 |
+
def _trigger_type(context: str) -> Optional[str]:
|
| 85 |
+
best_end, best_label = -1, None
|
| 86 |
+
for label, triggers in (("Ayah", AYAH_TRIGGERS), ("Hadith", HADITH_TRIGGERS)):
|
| 87 |
+
for trigger in triggers:
|
| 88 |
+
pos = context.rfind(trigger)
|
| 89 |
+
if pos >= 0 and pos + len(trigger) > best_end:
|
| 90 |
+
best_end, best_label = pos + len(trigger), label
|
| 91 |
+
return best_label
|
| 92 |
+
|
| 93 |
+
def _corpus_coverage(self, span: str):
|
| 94 |
+
"""Highest word coverage of the span by any top Quran ayah / Hadith candidate."""
|
| 95 |
+
if self.kb is None:
|
| 96 |
+
return 0.0, 0.0
|
| 97 |
+
quran_words = set(tokenize(normalize_strict(span)))
|
| 98 |
+
if not quran_words:
|
| 99 |
+
return 0.0, 0.0
|
| 100 |
+
quran_cov = self._quran_window_coverage(quran_words, span)
|
| 101 |
+
hadith_words = set(tokenize(normalize_lenient(span)))
|
| 102 |
+
hadith_cov = max(
|
| 103 |
+
(len(hadith_words & set(tokenize(normalize_lenient(c["text"])))) / len(hadith_words)
|
| 104 |
+
for c in self.kb.search_hadith(span, top_k=5)),
|
| 105 |
+
default=0.0,
|
| 106 |
+
) if hadith_words else 0.0
|
| 107 |
+
return quran_cov, hadith_cov
|
| 108 |
+
|
| 109 |
+
def _quran_window_coverage(self, quran_words: set, span: str) -> float:
|
| 110 |
+
"""Best word coverage of the span by a single ayah or by 2-3 consecutive ayahs around a top candidate (a quotation
|
| 111 |
+
often runs across an ayah boundary, and no single ayah then covers it)."""
|
| 112 |
+
kb, best = self.kb, 0.0
|
| 113 |
+
for cand in kb.search_quran_ayahs(span, top_k=5):
|
| 114 |
+
surah = kb.quran_by_surah.get(cand["surah_id"], {})
|
| 115 |
+
for first in range(cand["ayah_id"] - 2, cand["ayah_id"] + 1):
|
| 116 |
+
words: set = set()
|
| 117 |
+
for length in (1, 2, 3):
|
| 118 |
+
idx = surah.get(first + length - 1)
|
| 119 |
+
if idx is None:
|
| 120 |
+
break
|
| 121 |
+
ayah_words = set(tokenize(normalize_strict(kb.quran[idx]["text"])))
|
| 122 |
+
if length > 1 and not any(len(w) >= 4 for w in quran_words & ayah_words):
|
| 123 |
+
break # every ayah of a multi-ayah window must contribute a real (not particle-like) word of the quotation
|
| 124 |
+
words |= ayah_words
|
| 125 |
+
if first <= cand["ayah_id"] <= first + length - 1:
|
| 126 |
+
best = max(best, len(quran_words & words) / len(quran_words))
|
| 127 |
+
return best
|
| 128 |
+
|
| 129 |
+
def _label_from_corpus(self, n_words: int, trigger: Optional[str], quran_cov: float, hadith_cov: float) -> Optional[str]:
|
| 130 |
+
"""Corpus-first typing. The introductory phrase only breaks near-ties or types an altered quotation whose
|
| 131 |
+
coverage is too low for the corpus to decide on its own."""
|
| 132 |
+
quran_ok = quran_cov >= self.quoted_min_cov
|
| 133 |
+
hadith_ok = hadith_cov >= self.quoted_min_cov_hadith # Hadith records are long: stricter, ordinary prose overlaps them
|
| 134 |
+
if (quran_ok or hadith_ok) and n_words >= 4:
|
| 135 |
+
if trigger and ((quran_ok if trigger == "Ayah" else hadith_ok)) and abs(quran_cov - hadith_cov) < 0.25:
|
| 136 |
+
return trigger
|
| 137 |
+
if quran_ok and hadith_ok:
|
| 138 |
+
return "Ayah" if quran_cov >= hadith_cov else "Hadith"
|
| 139 |
+
return "Ayah" if quran_ok else "Hadith"
|
| 140 |
+
return trigger # may be None: a delimited segment that matches nothing and has no hint is not reported
|
| 141 |
+
|
| 142 |
+
def detect(self, text: str) -> List[DetectedSpan]:
|
| 143 |
+
candidates = []
|
| 144 |
+
for start, end in self._segments(text):
|
| 145 |
+
start, end = trim_span(text, start, end)
|
| 146 |
+
if end <= start:
|
| 147 |
+
continue
|
| 148 |
+
inner = text[start:end]
|
| 149 |
+
words = [w for w in normalize_for_matching(inner).split() if w]
|
| 150 |
+
if len(words) < 2 or len(inner) > 3000:
|
| 151 |
+
continue
|
| 152 |
+
trigger = self._trigger_type(text[max(0, start - self.context_chars):start])
|
| 153 |
+
if len(words) < self.min_words and not trigger: # a very short quote is only taken after an introductory phrase
|
| 154 |
+
continue
|
| 155 |
+
if sum(w in _FORMULA_WORDS for w in words) / len(words) >= 0.6:
|
| 156 |
+
continue
|
| 157 |
+
quran_cov, hadith_cov = self._corpus_coverage(inner)
|
| 158 |
+
label = None
|
| 159 |
+
if self.decouple_triggers:
|
| 160 |
+
label = self._label_from_corpus(len(words), trigger, quran_cov, hadith_cov)
|
| 161 |
+
elif trigger:
|
| 162 |
+
label = trigger
|
| 163 |
+
other, mine = (hadith_cov, quran_cov) if trigger == "Ayah" else (quran_cov, hadith_cov)
|
| 164 |
+
if other >= 0.8 and mine < 0.5:
|
| 165 |
+
label = "Hadith" if trigger == "Ayah" else "Ayah"
|
| 166 |
+
elif max(quran_cov, hadith_cov) >= self.min_corpus_cov and len(words) >= 4:
|
| 167 |
+
label = "Ayah" if quran_cov >= hadith_cov else "Hadith"
|
| 168 |
+
if label is None:
|
| 169 |
+
continue
|
| 170 |
+
candidates.append((bool(trigger), max(quran_cov, hadith_cov), end - start, start, end, label, trigger))
|
| 171 |
+
|
| 172 |
+
candidates.sort(key=lambda c: (c[0], c[1], c[2]), reverse=True) # trigger first, then corpus match, then length
|
| 173 |
+
taken = []
|
| 174 |
+
for _, _, _, start, end, label, hint in candidates:
|
| 175 |
+
if all(end <= t_start or start >= t_end for t_start, t_end, _, _ in taken):
|
| 176 |
+
taken.append((start, end, label, hint))
|
| 177 |
+
taken.sort()
|
| 178 |
+
return [DetectedSpan(s, e, label, None, "rules", text[s:e], hint) for s, e, label, hint in taken]
|
examples.json
ADDED
|
@@ -0,0 +1,227 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"id": "scenario_all_correct_ayahs",
|
| 4 |
+
"title": "كل الآيات صحيحة",
|
| 5 |
+
"question": "ما فضل الصبر واليسر بعد العسر وغاية خلق الإنسان في القرآن؟",
|
| 6 |
+
"scenario": "سيناريو للتجربة: كل الآيات صحيحة",
|
| 7 |
+
"text": "ذكر القرآن حكمة خلق الإنسان وبشّر بعد الشدة بالفرج. قال الله تعالى: \"وَمَا خَلَقْتُ الْجِنَّ وَالْإِنسَ إِلَّا لِيَعْبُدُونِ\" [الذاريات: 56]. وقال سبحانه: \"فَإِنَّ مَعَ الْعُسْرِ يُسْرًا إِنَّ مَعَ الْعُسْرِ يُسْرًا\" [الشرح: 5-6]. وقال عز وجل: \"وَاسْتَعِينُوا بِالصَّبْرِ وَالصَّلَاةِ ۚ وَإِنَّهَا لَكَبِيرَةٌ إِلَّا عَلَى الْخَاشِعِينَ\" [البقرة: 45].",
|
| 8 |
+
"built_from": "كُتب من نصوص القرآن والحديث المضمّنة؛ الأخطاء مصطنعة لغرض التجربة",
|
| 9 |
+
"observed_statuses": [
|
| 10 |
+
"VERIFIED",
|
| 11 |
+
"VERIFIED",
|
| 12 |
+
"VERIFIED"
|
| 13 |
+
]
|
| 14 |
+
},
|
| 15 |
+
{
|
| 16 |
+
"id": "scenario_all_wrong_ayahs",
|
| 17 |
+
"title": "كل الآيات خاطئة",
|
| 18 |
+
"question": "ما الآيات الواردة في العبادة والفرج والصبر؟",
|
| 19 |
+
"scenario": "سيناريو للتجربة: كل الآيات خاطئة",
|
| 20 |
+
"text": "وردت في القرآن آيات في هذا الباب. قال الله تعالى: \"وَمَا خَلَقْتُ الْجِنَّ وَالْإِنسَ إِلَّا لِيَأْكُلُونِ\". وقال تعالى: \"فَإِنَّ مَعَ الْعُسْرِ يُسْرًا إِنَّ بَعْدَ الْعُسْرِ سُرُورًا\". وقال سبحانه: \"وَاسْتَعِينُوا بِالصَّبْرِ وَالصَّلَاةِ ۚ وَإِنَّهَا لَسَهْلَةٌ عَلَى الْخَاشِعِينَ\".",
|
| 21 |
+
"built_from": "كُتب من نصوص القرآن والحديث المضمّنة؛ الأخطاء مصطنعة لغرض التجربة",
|
| 22 |
+
"observed_statuses": [
|
| 23 |
+
"CORRECTED",
|
| 24 |
+
"CORRECTED",
|
| 25 |
+
"CORRECTED"
|
| 26 |
+
]
|
| 27 |
+
},
|
| 28 |
+
{
|
| 29 |
+
"id": "scenario_all_correct_hadiths",
|
| 30 |
+
"title": "كل الأحاديث صحيحة",
|
| 31 |
+
"question": "ما الأحاديث الصحيحة في النية والطهارة وحفظ اللسان؟",
|
| 32 |
+
"scenario": "سيناريو للتجربة: كل الأحاديث صحيحة",
|
| 33 |
+
"text": "من الأحاديث الصحيحة في هذا الباب: قال رسول الله ﷺ: \"إِنَّمَا الْأَعْمَالُ بِالنِّيَّاتِ ، وَإِنَّمَا لِكُلِّ امْرِئٍ مَا نَوَى\" رواه البخاري. وقال ﷺ: \"الطُّهُورُ شَطْرُ الْإِيمَانِ\" رواه مسلم. وقال ﷺ: \"مَنْ كَانَ يُؤْمِنُ بِاللَّهِ وَالْيَوْمِ الْآخِرِ فَلْيَقُلْ خَيْرًا أَوْ لِيَصْمُتْ\" متفق عليه.",
|
| 34 |
+
"built_from": "كُتب من نصوص القرآن والحديث المضمّنة؛ الأخطاء مصطنعة لغرض التجربة",
|
| 35 |
+
"observed_statuses": [
|
| 36 |
+
"VERIFIED",
|
| 37 |
+
"VERIFIED",
|
| 38 |
+
"VERIFIED"
|
| 39 |
+
]
|
| 40 |
+
},
|
| 41 |
+
{
|
| 42 |
+
"id": "scenario_all_wrong_hadiths",
|
| 43 |
+
"title": "كل الأحاديث خاطئة",
|
| 44 |
+
"question": "هل ورد حديث في فضل الذكر والصدقة وبر الوالدين؟",
|
| 45 |
+
"scenario": "سيناريو للتجربة: كل الأحاديث خاطئة",
|
| 46 |
+
"text": "يُروى في هذا الباب: قال رسول الله ﷺ: \"مَنْ لَبِسَ الْأَخْضَرَ يَوْمَ الْجُمُعَةِ زَادَ اللَّهُ فِي رِزْقِهِ عَشْرَةَ أَضْعَافٍ\". وقال ﷺ: \"مَنْ أَكَلَ الْعَسَلَ مَعَ الثُّومِ سَبْعَةَ أَيَّامٍ شُفِيَ مِنْ كُلِّ دَاءٍ بِإِذْنِ اللَّهِ\". وقال ﷺ: \"مَنْ دَخَلَ بَيْتَهُ بِرِجْلِهِ الْيُسْرَى انْتَقَصَ عُمُرُهُ سَنَةً كَامِلَةً\".",
|
| 47 |
+
"built_from": "كُتب من نصوص القرآن والحديث المضمّنة؛ الأخطاء مصطنعة لغرض التجربة",
|
| 48 |
+
"observed_statuses": [
|
| 49 |
+
"UNSUPPORTED",
|
| 50 |
+
"UNSUPPORTED",
|
| 51 |
+
"UNSUPPORTED"
|
| 52 |
+
]
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"id": "scenario_wrong_ayahs_correct_hadiths",
|
| 56 |
+
"title": "آيات خاطئة وأحاديث صحيحة",
|
| 57 |
+
"question": "ما الدليل على فضل الطهارة والنية من القرآن والسنة؟",
|
| 58 |
+
"scenario": "سيناريو للتجربة: آيات خاطئة وأحاديث صحيحة",
|
| 59 |
+
"text": "قال الله تعالى: \"إِنَّ اللَّهَ يُحِبُّ التَّوَّابِينَ وَيُحِبُّ الْمُسْرِفِينَ\". وقال تعالى: \"وَثِيَابَكَ فَطَهِّرْ وَالرُّجْزَ فَاعْبُدْ\". وأما السنة فقال رسول الله ﷺ: \"الطُّهُورُ شَطْرُ الْإِيمَانِ\" رواه مسلم. وقال ﷺ: \"إِنَّمَا الْأَعْمَالُ بِالنِّيَّاتِ ، وَإِنَّمَا لِكُلِّ امْرِئٍ مَا نَوَى\" رواه البخاري.",
|
| 60 |
+
"built_from": "كُتب من نصوص القرآن والحديث المضمّنة؛ الأخطاء مصطنعة لغرض التجربة",
|
| 61 |
+
"observed_statuses": [
|
| 62 |
+
"CORRECTED",
|
| 63 |
+
"CORRECTED",
|
| 64 |
+
"VERIFIED",
|
| 65 |
+
"VERIFIED"
|
| 66 |
+
]
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"id": "scenario_correct_ayahs_wrong_hadiths",
|
| 70 |
+
"title": "آيات صحيحة وأحاديث خاطئة",
|
| 71 |
+
"question": "ما الدليل على فضل الصدق وبر الوالدين والصلاة؟",
|
| 72 |
+
"scenario": "سيناريو للتجربة: آيات صحيحة وأحاديث خاطئة",
|
| 73 |
+
"text": "قال الله تعالى: \"يَا أَيُّهَا الَّذِينَ آمَنُوا اتَّقُوا اللَّهَ وَكُونُوا مَعَ الصَّادِقِينَ\" [التوبة: 119]. وقال تعالى: \"وَقَضَىٰ رَبُّكَ أَلَّا تَعْبُدُوا إِلَّا إِيَّاهُ وَبِالْوَالِدَيْنِ إِحْسَانًا\" [الإسراء: 23]. وقال رسول الله ﷺ: \"مَنْ سَافَرَ يَوْمَ الْأَرْبِعَاءِ لَمْ يَرْجِعْ إِلَّا بِالْخَسَارَةِ وَالْهَمِّ\". وقال ﷺ: \"مَنْ دَخَلَ بَيْتَهُ بِرِجْلِهِ الْيُسْرَى انْتَقَصَ عُمُرُهُ سَنَةً كَامِلَةً\".",
|
| 74 |
+
"built_from": "كُتب من نصوص القرآن والحديث المضمّنة؛ الأخطاء مصطنعة لغرض التجربة",
|
| 75 |
+
"observed_statuses": [
|
| 76 |
+
"VERIFIED",
|
| 77 |
+
"VERIFIED",
|
| 78 |
+
"UNSUPPORTED",
|
| 79 |
+
"UNSUPPORTED"
|
| 80 |
+
]
|
| 81 |
+
},
|
| 82 |
+
{
|
| 83 |
+
"id": "correct_ayah",
|
| 84 |
+
"title": "آية صحيحة",
|
| 85 |
+
"scenario": "correct",
|
| 86 |
+
"text": "قال الله تعالى: \"يَا أَيُّهَا الَّذِينَ آمَنُوا اسْتَعِينُوا بِالصَّبْرِ وَالصَّلَاةِ ۚ إِنَّ اللَّهَ مَعَ الصَّابِرِينَ\". وفي الآية بشارة لمن صبر.",
|
| 87 |
+
"built_from": [
|
| 88 |
+
"سورة البقرة 153"
|
| 89 |
+
],
|
| 90 |
+
"observed_statuses": [
|
| 91 |
+
"VERIFIED"
|
| 92 |
+
]
|
| 93 |
+
},
|
| 94 |
+
{
|
| 95 |
+
"id": "correct_hadith",
|
| 96 |
+
"title": "حديث صحيح",
|
| 97 |
+
"scenario": "correct",
|
| 98 |
+
"text": "وقال رسول الله ﷺ: \"إِنَّمَا الْأَعْمَالُ بِالنِّيَّاتِ ، وَإِنَّمَا لِكُلِّ امْرِئٍ مَا نَوَى ، فَمَنْ كَانَتْ هِجْرَتُهُ إِلَى دُنْيَا يُصِيبُهَا ، أَوْ إِلَى امْرَأَةٍ يَنْكِحُهَا ، فَهِجْرَتُهُ إِلَى مَا هَاجَرَ إِلَيْهِ\". ويستفاد منه أهمية النية.",
|
| 99 |
+
"built_from": [
|
| 100 |
+
"حديث رقم 5"
|
| 101 |
+
],
|
| 102 |
+
"observed_statuses": [
|
| 103 |
+
"VERIFIED"
|
| 104 |
+
]
|
| 105 |
+
},
|
| 106 |
+
{
|
| 107 |
+
"id": "substituted_word",
|
| 108 |
+
"title": "كلمة مستبدلة: «ثم استقم» بدل «فاستقم»",
|
| 109 |
+
"scenario": "altered",
|
| 110 |
+
"text": "قال الله تعالى: \"ثُمَّ اسْتَقِمْ كَمَا أُمِرْتَ وَمَنْ تَابَ مَعَكَ\".",
|
| 111 |
+
"built_from": [
|
| 112 |
+
"سورة هود 112 (فَاسْتَقِمْ ← ثُمَّ اسْتَقِمْ)"
|
| 113 |
+
],
|
| 114 |
+
"observed_statuses": [
|
| 115 |
+
"CORRECTED"
|
| 116 |
+
]
|
| 117 |
+
},
|
| 118 |
+
{
|
| 119 |
+
"id": "swapped_order",
|
| 120 |
+
"title": "تبديل ترتيب الكلمات",
|
| 121 |
+
"scenario": "altered",
|
| 122 |
+
"text": "قال الله تعالى: \"كَمَا أُمِرْتَ فَاسْتَقِمْ وَمَنْ تَابَ مَعَكَ\".",
|
| 123 |
+
"built_from": [
|
| 124 |
+
"سورة هود 112 (تقديم وتأخير)"
|
| 125 |
+
],
|
| 126 |
+
"observed_statuses": [
|
| 127 |
+
"CORRECTED"
|
| 128 |
+
]
|
| 129 |
+
},
|
| 130 |
+
{
|
| 131 |
+
"id": "wrong_diacritic",
|
| 132 |
+
"title": "تشكيل خاطئ لكلمة",
|
| 133 |
+
"scenario": "altered",
|
| 134 |
+
"text": "قال الله تعالى: \"فَاسْتَقَمْ كَمَا أُمِرْتَ وَمَنْ تَابَ مَعَكَ\".",
|
| 135 |
+
"built_from": [
|
| 136 |
+
"سورة هود 112 (فَاسْتَقِمْ ← فَاسْتَقَمْ)"
|
| 137 |
+
],
|
| 138 |
+
"observed_statuses": [
|
| 139 |
+
"VERIFIED"
|
| 140 |
+
]
|
| 141 |
+
},
|
| 142 |
+
{
|
| 143 |
+
"id": "missing_diacritics",
|
| 144 |
+
"title": "بدون تشكيل (مقبول)",
|
| 145 |
+
"scenario": "correct",
|
| 146 |
+
"text": "قال الله تعالى: \"فاستقم كما أمرت ومن تاب معك\".",
|
| 147 |
+
"built_from": [
|
| 148 |
+
"سورة هود 112 بلا تشكيل"
|
| 149 |
+
],
|
| 150 |
+
"observed_statuses": [
|
| 151 |
+
"VERIFIED"
|
| 152 |
+
]
|
| 153 |
+
},
|
| 154 |
+
{
|
| 155 |
+
"id": "partial_citation",
|
| 156 |
+
"title": "اقتباس جزئي من آية",
|
| 157 |
+
"scenario": "partial",
|
| 158 |
+
"text": "قال الله تعالى: \"۞ وَقَضَىٰ رَبُّكَ أَلَّا تَعْبُدُوا إِلَّا\".",
|
| 159 |
+
"built_from": [
|
| 160 |
+
"سورة الإسراء 23 (أول ست كلمات)"
|
| 161 |
+
],
|
| 162 |
+
"observed_statuses": [
|
| 163 |
+
"VERIFIED"
|
| 164 |
+
]
|
| 165 |
+
},
|
| 166 |
+
{
|
| 167 |
+
"id": "missing_word",
|
| 168 |
+
"title": "كلمة ناقصة",
|
| 169 |
+
"scenario": "altered",
|
| 170 |
+
"text": "قال الله تعالى: \"يَا أَيُّهَا آمَنُوا اسْتَعِينُوا بِالصَّبْرِ وَالصَّلَاةِ ۚ إِنَّ ال��َّهَ مَعَ الصَّابِرِينَ\".",
|
| 171 |
+
"built_from": [
|
| 172 |
+
"سورة البقرة 153 (حُذفت الكلمة الثالثة)"
|
| 173 |
+
],
|
| 174 |
+
"observed_statuses": [
|
| 175 |
+
"CORRECTED"
|
| 176 |
+
]
|
| 177 |
+
},
|
| 178 |
+
{
|
| 179 |
+
"id": "unannounced_quote",
|
| 180 |
+
"title": "اقتباس غير معلَن بلا علامات تنصيص",
|
| 181 |
+
"scenario": "unannounced",
|
| 182 |
+
"text": "وهنا نتذكر أن العبد عليه أن يثبت وألا يتراجع، فاستقم كما أمرت ومن تاب معك ولا تطغوا، وهذا من أعظم ما يعين على الطاعة.",
|
| 183 |
+
"built_from": [
|
| 184 |
+
"سورة هود 112 داخل الجملة"
|
| 185 |
+
],
|
| 186 |
+
"observed_statuses": [
|
| 187 |
+
"VERIFIED"
|
| 188 |
+
]
|
| 189 |
+
},
|
| 190 |
+
{
|
| 191 |
+
"id": "uncertain_hadith",
|
| 192 |
+
"title": "حديث مركّب من حديثين",
|
| 193 |
+
"scenario": "uncertain",
|
| 194 |
+
"text": "روى البخاري أن النبي ﷺ قال: \"إِنَّمَا الْأَعْمَالُ بِالنِّيَّاتِ ، وَإِنَّمَا لِكُلِّ امْرِئٍ مَا نَوَى ، فَمَنْ كَانَتْ هِجْرَتُهُ أَهْلَ الْأَرْضِ يَرْحَمْكُمْ مَنْ فِي السَّمَاءِ\".",
|
| 195 |
+
"built_from": [
|
| 196 |
+
"حديث رقم 5 (النصف الأول)",
|
| 197 |
+
"حديث رقم 95766 (النصف الثاني)"
|
| 198 |
+
],
|
| 199 |
+
"observed_statuses": [
|
| 200 |
+
"HUMAN_REVIEW"
|
| 201 |
+
]
|
| 202 |
+
},
|
| 203 |
+
{
|
| 204 |
+
"id": "no_quotation",
|
| 205 |
+
"title": "نص بلا اقتباسات",
|
| 206 |
+
"scenario": "none",
|
| 207 |
+
"text": "الصبر من أهم الصفات التي ينبغي أن يتحلى بها الإنسان في حياته، ويساعده على مواجهة المصاعب بثبات وهدوء.",
|
| 208 |
+
"built_from": [],
|
| 209 |
+
"observed_statuses": []
|
| 210 |
+
},
|
| 211 |
+
{
|
| 212 |
+
"id": "mixed_response",
|
| 213 |
+
"title": "رد مختلط",
|
| 214 |
+
"scenario": "mixed",
|
| 215 |
+
"text": "بخصوص بر الوالدين، قال الله تعالى: \"۞ وَقَضَىٰ رَبُّكَ أَلَّا تَعْبُدُوا إِلَّا إِيَّاهُ وَبِالْوَالِدَيْنِ إِحْسَانًا ۚ إِمَّا يَبْلُغَنَّ عِنْدَكَ الْكِبَرَ أَحَدُهُمَا أَوْ كِلَاهُمَا فَلَا تَقُلْ لَهُمَا أُفٍّ وَلَا تَنْهَرْهُمَا وَقُلْ لَهُمَا قَوْلًا كَرِيمًا\". وقال رسول الله ﷺ: \"الرَّاحِمُونَ يَرْحَمُهُمُ الرَّحْمَنُ ، ارْحَمُوا أَهْلَ الْأَرْضِ يَرْحَمْكُمْ مَنْ فِي السَّمَاءِ\". وقال سبحانه: \"يَا أَيُّهَا آمَنُوا اسْتَعِينُوا بِالصَّبْرِ وَالصَّلَاةِ ۚ إِنَّ اللَّهَ مَعَ الصَّابِرِينَ\". وهذا من مكارم الأخلاق.",
|
| 216 |
+
"built_from": [
|
| 217 |
+
"سورة الإسراء 23",
|
| 218 |
+
"حديث رقم 95766",
|
| 219 |
+
"سورة البقرة 153 (كلمة ناقصة)"
|
| 220 |
+
],
|
| 221 |
+
"observed_statuses": [
|
| 222 |
+
"VERIFIED",
|
| 223 |
+
"VERIFIED",
|
| 224 |
+
"CORRECTED"
|
| 225 |
+
]
|
| 226 |
+
}
|
| 227 |
+
]
|
hadith.idx.gz
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6f5792317d7577220e0bd4b1aab84be1ed665c17f974ab7ba1cd8920f34c9955
|
| 3 |
+
size 13597012
|
hadith.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f1f514d7a2de2d4179d9588a861cedf07e81630886444f37d0c08c2cc777075a
|
| 3 |
+
size 48498788
|
idgham.py
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Idgham rendering of the Quran text in the published mushaf convention (used by the Subtask 1C gold corrections).
|
| 2 |
+
|
| 3 |
+
The flat Quran text writes assimilated letters without a shadda; the mushaf marks the assimilation. Covers noon
|
| 4 |
+
sakinah / tanween before ن م ر ل, meem sakinah before م and lam sakinah before ل ر, also across ayah-number markers
|
| 5 |
+
such as ``(20)``. The و / ي cases are excluded on purpose because the mushaf convention is inconsistent there."""
|
| 6 |
+
from __future__ import annotations
|
| 7 |
+
|
| 8 |
+
import re
|
| 9 |
+
|
| 10 |
+
_MARKER = re.compile(r"^\(\d+\)$")
|
| 11 |
+
|
| 12 |
+
_SUKUN, _SHADDA, _FATHATAN = "\u0652", "\u0651", "\u064B"
|
| 13 |
+
_TANWEEN = set("\u064B\u064C\u064D")
|
| 14 |
+
_IDGHAM_AFTER_NOON = set("نمرل")
|
| 15 |
+
_IDGHAM_AFTER_LAM = set("لر")
|
| 16 |
+
_DIACRITIC_CHARS = set(
|
| 17 |
+
"\u0610\u0611\u0612\u0613\u0614\u0615\u0616\u0617\u0618\u0619\u061A"
|
| 18 |
+
"\u064B\u064C\u064D\u064E\u064F\u0650\u0651\u0652\u0670"
|
| 19 |
+
"\u06D6\u06D7\u06D8\u06D9\u06DA\u06DB\u06DC\u06DF\u06E0\u06E1\u06E2\u06E3\u06E4"
|
| 20 |
+
"\u06E7\u06E8\u06EA\u06EB\u06EC\u06ED"
|
| 21 |
+
)
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def _base_letters(word: str) -> str:
|
| 25 |
+
return "".join(c for c in word if c not in _DIACRITIC_CHARS)
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
def _insert_shadda(word: str) -> str:
|
| 29 |
+
return word if not word else word[0] + _SHADDA + word[1:]
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def _ends_with_tanween(word: str) -> bool:
|
| 33 |
+
if not word:
|
| 34 |
+
return False
|
| 35 |
+
if word[-1] in _TANWEEN:
|
| 36 |
+
return True
|
| 37 |
+
return len(word) >= 2 and word[-1] in "اى" and word[-2] == _FATHATAN
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
def apply_idgham(text: str, extended: bool = True) -> str:
|
| 41 |
+
"""Convert the flat Quran text into the mushaf rendering that marks assimilation with a shadda.
|
| 42 |
+
|
| 43 |
+
Covers noon sakinah / tanween before ن م ر ل, meem sakinah before م and lam sakinah before ل ر, also across
|
| 44 |
+
ayah-number markers such as ``(20)``. The و / ي cases are excluded on purpose because the mushaf convention
|
| 45 |
+
is inconsistent there."""
|
| 46 |
+
words = text.split(" ")
|
| 47 |
+
noon_set = _IDGHAM_AFTER_NOON if extended else set("نم")
|
| 48 |
+
i = 0
|
| 49 |
+
while i < len(words):
|
| 50 |
+
word = words[i]
|
| 51 |
+
if _MARKER.match(word) or not word:
|
| 52 |
+
i += 1
|
| 53 |
+
continue
|
| 54 |
+
j = i + 1
|
| 55 |
+
while j < len(words) and _MARKER.match(words[j]):
|
| 56 |
+
j += 1
|
| 57 |
+
if j < len(words):
|
| 58 |
+
base = _base_letters(words[j])
|
| 59 |
+
first = base[0] if base else ""
|
| 60 |
+
if word.endswith("\u0646" + _SUKUN) and first in noon_set:
|
| 61 |
+
words[i], words[j] = word[:-1], _insert_shadda(words[j])
|
| 62 |
+
elif _ends_with_tanween(word) and first in noon_set:
|
| 63 |
+
words[j] = _insert_shadda(words[j])
|
| 64 |
+
elif word.endswith("\u0645" + _SUKUN) and first == "\u0645":
|
| 65 |
+
words[i], words[j] = word[:-1], _insert_shadda(words[j])
|
| 66 |
+
elif extended and word.endswith("\u0644" + _SUKUN) and first in _IDGHAM_AFTER_LAM:
|
| 67 |
+
words[i], words[j] = word[:-1], _insert_shadda(words[j])
|
| 68 |
+
i += 1
|
| 69 |
+
return " ".join(words)
|
| 70 |
+
|
| 71 |
+
|
index.html
CHANGED
|
@@ -1,19 +1,518 @@
|
|
| 1 |
<!doctype html>
|
| 2 |
-
<html>
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
|
| 13 |
-
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
|
| 17 |
-
|
| 18 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 19 |
</html>
|
|
|
|
| 1 |
<!doctype html>
|
| 2 |
+
<html lang="ar" dir="rtl">
|
| 3 |
+
<head>
|
| 4 |
+
<meta charset="utf-8">
|
| 5 |
+
<meta name="viewport" content="width=device-width, initial-scale=1">
|
| 6 |
+
<title>التحقق من هلوسة القرآن والحديث وتصحيحها</title>
|
| 7 |
+
<meta name="description" content="تحقّق من آيات القرآن والأحاديث النبوية بالدليل">
|
| 8 |
+
<link rel="preconnect" href="https://cdn.jsdelivr.net" crossorigin>
|
| 9 |
+
<link rel="preconnect" href="https://fonts.googleapis.com">
|
| 10 |
+
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
| 11 |
+
<link rel="preload" href="https://cdn.jsdelivr.net/pyodide/v0.26.4/full/pyodide.js" as="script" crossorigin>
|
| 12 |
+
<link rel="preload" href="index/quran.idx.gz" as="fetch" crossorigin>
|
| 13 |
+
<link rel="prefetch" href="index/hadith.idx.gz" as="fetch" crossorigin>
|
| 14 |
+
<style>
|
| 15 |
+
@import url('https://fonts.googleapis.com/css2?family=Cairo:wght@400;600;700&family=Amiri:wght@400;700&display=swap');
|
| 16 |
+
:root { --green:#0F4C3A; --green-2:#17694F; --gold:#B8912F; --gold-2:#E2C06E; --cream:#FBF6EA; --paper:#FFFFFF;
|
| 17 |
+
--ink:#1F2933; --muted:#55626D; --line:#E6DCC3; --ok:#1B7A4B; --ok-bg:#EAF6EF; --ok-line:#BFE3CD;
|
| 18 |
+
--bad:#B3261E; --bad-bg:#FDECEA; --bad-line:#F4B8B3; --warn:#7A5B0C; --warn-bg:#FFF6DA; --warn-line:#EBD28A; }
|
| 19 |
+
.gradio-container, body.icv-page { background:var(--cream) url("data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' width='88' height='88' viewBox='0 0 88 88'%3E%3Cg fill='none' stroke='%230F4C3A' stroke-opacity='0.07' stroke-width='1'%3E%3Crect x='22' y='22' width='44' height='44'/%3E%3Crect x='22' y='22' width='44' height='44' transform='rotate(45 44 44)'/%3E%3C/g%3E%3C/svg%3E") !important; font-family:'Cairo','Segoe UI',Tahoma,sans-serif; color:var(--ink); }
|
| 20 |
+
.gradio-container { max-width:1060px !important; --body-text-color:#1F2933; --block-background-fill:#FFFFFF; --block-border-color:#E6DCC3;
|
| 21 |
+
--input-background-fill:#FFFFFF; --block-label-text-color:#55626D; --button-primary-background-fill:#0F4C3A;
|
| 22 |
+
--button-primary-background-fill-hover:#17694F; --button-primary-text-color:#FFFFFF; --button-secondary-background-fill:#FFFFFF;
|
| 23 |
+
--button-secondary-text-color:#0F4C3A; --button-secondary-border-color:#B8912F; --color-accent:#B8912F; }
|
| 24 |
+
.icv { direction:rtl; text-align:right; color:var(--ink); line-height:1.9; }
|
| 25 |
+
.icv .hero { background:linear-gradient(135deg,#0F4C3A,#17694F); border-radius:20px; padding:30px 22px 26px; margin:10px 0 18px; text-align:center;
|
| 26 |
+
box-shadow:0 6px 22px rgba(15,76,58,.18); border-bottom:4px solid var(--gold); }
|
| 27 |
+
.icv .hero .mark { width:54px; height:54px; display:block; margin:0 auto 4px; }
|
| 28 |
+
.icv .hero h1 { font-size:2.3rem; margin:.1rem 0; color:#FFFFFF; font-weight:700; }
|
| 29 |
+
.icv .hero .tagline { color:var(--gold-2); font-size:1.2rem; font-weight:600; margin:.1rem 0 .6rem; }
|
| 30 |
+
.icv .hero .sub { max-width:720px; margin:0 auto; color:#E9F2EE; font-size:1rem; }
|
| 31 |
+
.icv .flow { display:flex; flex-wrap:wrap; justify-content:center; align-items:center; gap:8px; margin-top:16px; }
|
| 32 |
+
.icv .flow span { background:rgba(255,255,255,.12); border:1px solid rgba(226,192,110,.55); color:#FFF; border-radius:999px; padding:3px 16px; font-size:.9rem; }
|
| 33 |
+
.icv .flow i { color:var(--gold-2); font-style:normal; font-size:1.2rem; }
|
| 34 |
+
.icv .disclaimer { font-size:.85rem; color:var(--muted); text-align:center; padding:14px 8px; }
|
| 35 |
+
.icv .en-free { direction:rtl; }
|
| 36 |
+
.input-area textarea { direction:rtl; text-align:right; font-family:'Amiri','Cairo',serif !important; font-size:1.25rem !important; line-height:2.1 !important; background:#FFFFFF !important; color:#1F2933 !important; }
|
| 37 |
+
.llm-row label, .gradio-container label span { font-family:'Cairo',sans-serif; }
|
| 38 |
+
.results { direction:rtl; }
|
| 39 |
+
.icv .summary { display:grid; grid-template-columns:repeat(auto-fit,minmax(130px,1fr)); gap:10px; margin:8px 0 14px; }
|
| 40 |
+
.icv .tile { background:var(--paper); border:1px solid var(--line); border-radius:14px; padding:12px 8px; text-align:center; }
|
| 41 |
+
.icv .tile b { display:block; font-size:1.8rem; color:var(--green); line-height:1.3; } .icv .tile span { font-size:.92rem; color:var(--muted); }
|
| 42 |
+
.icv .tile.verified { background:var(--ok-bg); border-color:var(--ok-line); } .icv .tile.verified b { color:var(--ok); }
|
| 43 |
+
.icv .tile.mismatch { background:var(--bad-bg); border-color:var(--bad-line); } .icv .tile.mismatch b { color:var(--bad); }
|
| 44 |
+
.icv .tile.review { background:var(--warn-bg); border-color:var(--warn-line); } .icv .tile.review b { color:var(--warn); }
|
| 45 |
+
.icv .tile.zero { background:var(--paper); border-color:var(--line); opacity:.6; } .icv .tile.zero b { color:var(--muted); }
|
| 46 |
+
.icv .highlight, .icv .generated, .icv .final { background:var(--paper); border:1px solid var(--line); border-radius:14px; padding:14px 18px; margin-bottom:14px; }
|
| 47 |
+
.icv label { display:block; color:var(--muted); font-size:.82rem; font-weight:600; margin-bottom:4px; }
|
| 48 |
+
.icv .highlight p, .icv .generated p, .icv .final-text { font-family:'Amiri','Cairo',serif; font-size:1.2rem; line-height:2.2; margin:0; white-space:pre-wrap; }
|
| 49 |
+
.icv mark { color:var(--ink); border-radius:6px; padding:1px 5px; }
|
| 50 |
+
.icv mark.verified { background:#D5EEDF; } .icv mark.mismatch { background:#F8CFCB; } .icv mark.review { background:#F7E3A6; }
|
| 51 |
+
.icv .hl-legend { display:flex; flex-wrap:wrap; gap:8px; margin-top:10px; font-size:.78rem; }
|
| 52 |
+
.icv .qcard { background:var(--paper); border:1px solid var(--line); border-inline-start:6px solid var(--line); border-radius:16px; padding:16px 18px; margin:12px 0; }
|
| 53 |
+
.icv .qcard.verified { border-inline-start-color:var(--ok); } .icv .qcard.mismatch { border-inline-start-color:var(--bad); } .icv .qcard.review { border-inline-start-color:var(--gold); }
|
| 54 |
+
.icv .qhead { display:flex; flex-wrap:wrap; gap:8px; align-items:center; margin-bottom:8px; }
|
| 55 |
+
.icv .idx { background:var(--green); color:#FFF; border-radius:50%; width:28px; height:28px; display:inline-flex; align-items:center; justify-content:center; font-weight:700; font-size:.9rem; }
|
| 56 |
+
.icv .badge { padding:2px 14px; border-radius:999px; font-size:.84rem; background:#F3EEDD; border:1px solid var(--line); color:var(--ink); }
|
| 57 |
+
.icv .badge.st.verified { background:var(--ok-bg); border-color:var(--ok-line); color:var(--ok); font-weight:600; }
|
| 58 |
+
.icv .badge.st.mismatch { background:var(--bad-bg); border-color:var(--bad-line); color:var(--bad); font-weight:600; }
|
| 59 |
+
.icv .badge.st.review { background:var(--warn-bg); border-color:var(--warn-line); color:var(--warn); font-weight:600; }
|
| 60 |
+
.icv .badge.scan { background:#EEF3FB; border-color:#C9D8EE; color:#2F4F7F; }
|
| 61 |
+
.icv .conf { margin-inline-start:auto; color:var(--muted); font-size:.88rem; } .icv .conf b { color:var(--green); }
|
| 62 |
+
.icv .field { margin:8px 0; } .icv .field p { margin:0; }
|
| 63 |
+
.icv .quote { font-family:'Amiri','Cairo',serif; font-size:1.25rem; line-height:2.2; color:var(--ink); } .icv .quote.small { font-size:1.05rem; color:#3a4651; }
|
| 64 |
+
.icv .diff { font-family:'Amiri','Cairo',serif; font-size:1.2rem; line-height:2.2; }
|
| 65 |
+
.icv .w-extra { background:var(--bad-bg); color:var(--bad); border-radius:4px; padding:0 4px; text-decoration:line-through; }
|
| 66 |
+
.icv .w-missing { background:var(--ok-bg); color:var(--ok); border-radius:4px; padding:0 4px; font-weight:700; }
|
| 67 |
+
.icv .legend { color:var(--muted); font-size:.82rem; margin:4px 0 0; } .icv .legend span { text-decoration:none; font-size:.8rem; }
|
| 68 |
+
.icv .reason { color:var(--muted); font-size:.92rem; margin:6px 0 0; }
|
| 69 |
+
.icv .note { color:var(--warn); background:var(--warn-bg); border:1px solid var(--warn-line); border-radius:10px; padding:5px 12px; font-size:.88rem; margin:8px 0 0; }
|
| 70 |
+
.icv .action { border-radius:12px; padding:10px 14px; margin-top:10px; }
|
| 71 |
+
.icv .action.ok { background:var(--ok-bg); border:1px solid var(--ok-line); } .icv .action.ok b { color:var(--ok); }
|
| 72 |
+
.icv .action.warn { background:var(--warn-bg); border:1px solid var(--warn-line); } .icv .action.warn b { color:var(--warn); }
|
| 73 |
+
.icv .action.bad { background:var(--bad-bg); border:1px solid var(--bad-line); } .icv .action.bad b { color:var(--bad); }
|
| 74 |
+
.icv .evidence { margin-top:12px; border-top:1px dashed var(--line); padding-top:8px; }
|
| 75 |
+
.icv .evidence summary { cursor:pointer; color:var(--green); font-weight:700; }
|
| 76 |
+
.icv .ev-row { margin:10px 0; } .icv .ev-row b { color:var(--gold); font-size:.88rem; } .icv .ev-row p { margin:2px 0; }
|
| 77 |
+
.icv .indicators { display:grid; gap:6px; margin-top:6px; }
|
| 78 |
+
.icv .ind { display:grid; grid-template-columns:150px 1fr 48px; gap:10px; align-items:center; font-size:.88rem; }
|
| 79 |
+
.icv .bar { background:#EFE8D3; border-radius:999px; height:9px; overflow:hidden; direction:rtl; } .icv .bar span { display:block; height:100%; background:linear-gradient(270deg,var(--green),var(--gold)); }
|
| 80 |
+
.icv .ind-val { text-align:left; direction:ltr; color:var(--ink); } .icv .ind-val.wide { grid-column:2 / span 2; text-align:right; direction:rtl; }
|
| 81 |
+
.icv .final { border-color:var(--gold); background:#FFFDF6; }
|
| 82 |
+
.icv .final-head { display:flex; justify-content:space-between; align-items:center; } .icv .final-head b { color:var(--green); font-size:1.05rem; }
|
| 83 |
+
.icv .copy { background:var(--green); color:#FFF; border:0; border-radius:10px; padding:5px 16px; font-family:inherit; cursor:pointer; }
|
| 84 |
+
.icv .notice { background:var(--paper); border:1px solid var(--line); border-radius:14px; padding:16px; }
|
| 85 |
+
.icv .notice.warn { background:var(--warn-bg); border-color:var(--warn-line); color:var(--warn); } .icv .notice.bad { background:var(--bad-bg); border-color:var(--bad-line); color:var(--bad); }
|
| 86 |
+
@media (max-width:640px){ .icv .hero h1{font-size:1.7rem;} .icv .ind{grid-template-columns:104px 1fr 40px;} }
|
| 87 |
+
|
| 88 |
+
:root { --glass:rgba(255,255,255,.62); --glass-strong:rgba(255,255,255,.82); --glass-line:rgba(255,255,255,.75);
|
| 89 |
+
--shadow-1:0 1px 2px rgba(15,76,58,.06), 0 8px 24px rgba(15,76,58,.08); --shadow-2:0 2px 4px rgba(15,76,58,.08), 0 18px 44px rgba(15,76,58,.16); }
|
| 90 |
+
.gradio-container, body.icv-page { background:
|
| 91 |
+
radial-gradient(900px 520px at 88% -8%, rgba(226,192,110,.34), transparent 60%),
|
| 92 |
+
radial-gradient(760px 520px at 6% 4%, rgba(23,105,79,.20), transparent 62%),
|
| 93 |
+
linear-gradient(180deg,#FBF6EA 0%,#F3EBD3 100%) !important; background-attachment:fixed !important; }
|
| 94 |
+
/* RTL safety: long words, URLs and mixed Latin/digit runs wrap inside their box instead of spilling out */
|
| 95 |
+
.icv, .icv * { box-sizing:border-box; min-width:0; }
|
| 96 |
+
.icv p, .icv span, .icv b, .icv label, .icv summary, .icv button, .icv .badge, .icv .tile { overflow-wrap:anywhere; }
|
| 97 |
+
.icv .quote, .icv .diff, .icv .final-text, .icv .highlight p, .icv .generated p { unicode-bidi:plaintext; text-align:start; }
|
| 98 |
+
.icv .hero { position:relative; overflow:hidden; border-bottom:0; border:1px solid rgba(226,192,110,.45);
|
| 99 |
+
background:linear-gradient(120deg,#0B3B2D,#17694F 45%,#0F4C3A 70%,#1d7a5c); background-size:240% 240%; animation:icv-flow 16s ease-in-out infinite;
|
| 100 |
+
box-shadow:var(--shadow-2); }
|
| 101 |
+
.icv .hero::before { content:""; position:absolute; inset:-40% -10% auto auto; width:60%; aspect-ratio:1; border-radius:50%;
|
| 102 |
+
background:radial-gradient(circle, rgba(226,192,110,.38), transparent 65%); pointer-events:none; }
|
| 103 |
+
.icv .hero::after { content:""; position:absolute; inset:auto auto 0 0; width:100%; height:4px; background:linear-gradient(90deg,transparent,var(--gold-2),transparent); }
|
| 104 |
+
.icv .hero h1 { font-size:clamp(1.45rem,4.2vw,2.3rem); line-height:1.5; text-wrap:balance; position:relative; }
|
| 105 |
+
.icv .hero .tagline, .icv .hero .sub, .icv .flow { position:relative; }
|
| 106 |
+
.icv .flow span { backdrop-filter:blur(6px); -webkit-backdrop-filter:blur(6px); }
|
| 107 |
+
@keyframes icv-flow { 0%,100%{background-position:0% 50%} 50%{background-position:100% 50%} }
|
| 108 |
+
.icv .tile, .icv .highlight, .icv .generated, .icv .final, .icv .qcard, .icv .notice, .icv .export {
|
| 109 |
+
background:var(--glass); border:1px solid var(--glass-line); box-shadow:var(--shadow-1);
|
| 110 |
+
backdrop-filter:blur(14px) saturate(150%); -webkit-backdrop-filter:blur(14px) saturate(150%); transition:transform .25s ease, box-shadow .25s ease; }
|
| 111 |
+
.icv .qcard { border-inline-start:6px solid var(--line); }
|
| 112 |
+
.icv .qcard:hover, .icv .tile:hover { transform:translateY(-2px); box-shadow:var(--shadow-2); }
|
| 113 |
+
.icv .tile.verified { background:linear-gradient(160deg,rgba(234,246,239,.92),rgba(255,255,255,.6)); }
|
| 114 |
+
.icv .tile.mismatch { background:linear-gradient(160deg,rgba(253,236,234,.92),rgba(255,255,255,.6)); }
|
| 115 |
+
.icv .tile.review { background:linear-gradient(160deg,rgba(255,246,218,.95),rgba(255,255,255,.6)); }
|
| 116 |
+
.icv .final { background:linear-gradient(160deg,rgba(255,253,246,.92),rgba(250,240,208,.55)); border-color:rgba(184,145,47,.55); }
|
| 117 |
+
.icv .final-head { flex-wrap:wrap; gap:10px; }
|
| 118 |
+
.icv .qhead .badge { max-width:100%; white-space:normal; }
|
| 119 |
+
.icv .ind { grid-template-columns:minmax(96px,150px) minmax(0,1fr) 48px; }
|
| 120 |
+
.icv .bar span { background:linear-gradient(270deg,var(--green),var(--gold-2)); transition:width .6s ease; }
|
| 121 |
+
.icv .copy, .icv .dl { background:linear-gradient(135deg,var(--green),var(--green-2)); color:#FFF; border:0; border-radius:12px; padding:7px 18px;
|
| 122 |
+
font-family:inherit; font-weight:600; cursor:pointer; box-shadow:0 4px 12px rgba(15,76,58,.25); transition:transform .15s ease, box-shadow .15s ease, filter .15s ease; }
|
| 123 |
+
.icv .copy:hover, .icv .dl:hover { transform:translateY(-1px); filter:brightness(1.08); box-shadow:0 8px 18px rgba(15,76,58,.3); }
|
| 124 |
+
.icv .copy:active, .icv .dl:active { transform:translateY(0); }
|
| 125 |
+
.icv .copy:focus-visible, .icv .dl:focus-visible { outline:3px solid rgba(184,145,47,.55); outline-offset:2px; }
|
| 126 |
+
.icv .copy.done { background:linear-gradient(135deg,#1B7A4B,#2a9d66); }
|
| 127 |
+
.icv .export { border-radius:16px; padding:12px 18px; margin:12px 0; }
|
| 128 |
+
.icv .export summary { cursor:pointer; color:var(--green); font-weight:700; }
|
| 129 |
+
.icv .dls { display:flex; flex-wrap:wrap; gap:10px; margin-top:10px; }
|
| 130 |
+
/* toast */
|
| 131 |
+
#icv-toast { position:fixed; inset-inline:0; bottom:28px; margin-inline:auto; width:max-content; max-width:calc(100vw - 32px); z-index:99999;
|
| 132 |
+
direction:rtl; text-align:center; font:600 1rem 'Cairo','Segoe UI',Tahoma,sans-serif; color:#FFF; padding:12px 24px; border-radius:999px;
|
| 133 |
+
background:linear-gradient(135deg,rgba(15,76,58,.94),rgba(23,105,79,.94)); border:1px solid rgba(226,192,110,.6);
|
| 134 |
+
box-shadow:0 14px 40px rgba(15,76,58,.35); backdrop-filter:blur(10px); -webkit-backdrop-filter:blur(10px);
|
| 135 |
+
opacity:0; transform:translateY(18px) scale(.97); pointer-events:none; transition:opacity .28s ease, transform .28s cubic-bezier(.2,.9,.3,1.2); }
|
| 136 |
+
#icv-toast.show { opacity:1; transform:none; } #icv-toast.bad { background:linear-gradient(135deg,rgba(179,38,30,.95),rgba(214,69,58,.95)); }
|
| 137 |
+
/* skeleton loaders */
|
| 138 |
+
.skel { position:relative; overflow:hidden; background:rgba(15,76,58,.08); border-radius:10px; }
|
| 139 |
+
.skel::after { content:""; position:absolute; inset:0; transform:translateX(100%); animation:icv-shimmer 1.4s infinite;
|
| 140 |
+
background:linear-gradient(90deg,transparent,rgba(255,255,255,.75),transparent); }
|
| 141 |
+
@keyframes icv-shimmer { 100% { transform:translateX(-100%); } }
|
| 142 |
+
.skel-card { background:var(--glass); border:1px solid var(--glass-line); border-radius:16px; padding:16px 18px; margin:12px 0; box-shadow:var(--shadow-1); }
|
| 143 |
+
.skel-line { height:14px; margin:10px 0; } .skel-line.w60 { width:60%; } .skel-line.w85 { width:85%; } .skel-line.w40 { width:40%; }
|
| 144 |
+
.skel-tiles { display:grid; grid-template-columns:repeat(auto-fit,minmax(120px,1fr)); gap:10px; margin:8px 0 14px; } .skel-tile { height:78px; border-radius:14px; }
|
| 145 |
+
.icv-spinner { width:18px; height:18px; border-radius:50%; border:3px solid rgba(15,76,58,.18); border-top-color:var(--gold); display:inline-block;
|
| 146 |
+
vertical-align:middle; margin-inline-end:10px; animation:icv-spin .8s linear infinite; }
|
| 147 |
+
@keyframes icv-spin { to { transform:rotate(360deg); } }
|
| 148 |
+
@media (prefers-reduced-motion: reduce) { .icv .hero, .skel::after, .icv-spinner { animation:none; } .icv .qcard, .icv .tile, #icv-toast { transition:none; } }
|
| 149 |
+
@media (max-width:640px){ .icv .ind{grid-template-columns:minmax(84px,104px) minmax(0,1fr) 40px;} .icv .final-head{flex-direction:column; align-items:stretch;} .icv .copy{width:100%;} }
|
| 150 |
+
/* layout polish: centred, compact, professional */
|
| 151 |
+
.icv .notice, .icv .disclaimer, .icv .export { text-align:center; }
|
| 152 |
+
.icv .export .legend { max-width:640px; margin:8px auto 0; font-size:.9rem; line-height:1.9; color:var(--muted); }
|
| 153 |
+
.icv .dls { justify-content:center; }
|
| 154 |
+
.icv .dl { display:inline-flex; flex-direction:column; align-items:center; gap:2px; min-width:170px; }
|
| 155 |
+
.icv .dl .t { font-weight:700; } .icv .dl .s { font-size:.78rem; font-weight:500; opacity:.85; }
|
| 156 |
+
.icv .hero .flow { display:flex; justify-content:center; flex-wrap:wrap; gap:8px; }
|
| 157 |
+
|
| 158 |
+
body.icv-page{margin:0;min-height:100vh;background-attachment:fixed;} .page{max-width:980px;margin:0 auto;padding:10px 16px 28px;box-sizing:border-box;}
|
| 159 |
+
.glass{background:var(--glass);border:1px solid var(--glass-line);border-radius:20px;box-shadow:var(--shadow-1);
|
| 160 |
+
backdrop-filter:blur(16px) saturate(150%);-webkit-backdrop-filter:blur(16px) saturate(150%);}
|
| 161 |
+
.tabs{display:flex;gap:6px;margin:6px auto 14px;direction:rtl;padding:6px;border-radius:18px;max-width:560px;}
|
| 162 |
+
.tab{flex:1;border:0;background:transparent;color:var(--green);border-radius:13px;padding:11px 14px;font:700 1rem 'Cairo',sans-serif;cursor:pointer;transition:background .2s,color .2s,box-shadow .2s;}
|
| 163 |
+
.tab:hover{background:rgba(15,76,58,.07);}
|
| 164 |
+
.tab.active{background:linear-gradient(135deg,var(--green),var(--green-2));color:#FFF;box-shadow:0 6px 16px rgba(15,76,58,.28);}
|
| 165 |
+
.tab:focus-visible,.btns button:focus-visible{outline:3px solid rgba(184,145,47,.55);outline-offset:2px;}
|
| 166 |
+
section[hidden]{display:none;}
|
| 167 |
+
.form{padding:18px 20px 14px;margin-bottom:14px;} .form .icv{margin-bottom:6px;}
|
| 168 |
+
.form label{display:block;color:var(--muted);font-size:.88rem;font-weight:600;margin:10px 0 5px;direction:rtl;text-align:right;}
|
| 169 |
+
.form textarea,.form select{width:100%;box-sizing:border-box;background:rgba(255,255,255,.88);color:var(--ink);border:1px solid var(--line);border-radius:14px;padding:12px 14px;font:1rem 'Cairo',sans-serif;direction:rtl;text-align:right;outline:none;transition:border-color .2s,box-shadow .2s;}
|
| 170 |
+
.form textarea{min-height:200px;resize:vertical;font-family:'Amiri','Cairo',serif;font-size:1.25rem;line-height:2.1;}
|
| 171 |
+
.form textarea.short{min-height:96px;}
|
| 172 |
+
.form textarea:focus,.form select:focus{border-color:var(--gold);box-shadow:0 0 0 4px rgba(184,145,47,.18);}
|
| 173 |
+
.hint{font-size:.84rem;color:var(--muted);direction:rtl;margin:2px 0 0;}
|
| 174 |
+
.btns{display:flex;gap:12px;margin:14px 0 6px;flex-wrap:wrap;justify-content:center;}
|
| 175 |
+
.btns button{flex:1 1 200px;border:0;border-radius:14px;padding:13px 18px;font:700 1.02rem 'Cairo',sans-serif;cursor:pointer;transition:transform .15s,box-shadow .15s,filter .15s;}
|
| 176 |
+
.btn-main{background:linear-gradient(135deg,var(--green),var(--green-2));color:#FFF;box-shadow:0 8px 20px rgba(15,76,58,.28);}
|
| 177 |
+
.btn-main:hover{transform:translateY(-1px);filter:brightness(1.07);}
|
| 178 |
+
.btn-alt{background:rgba(255,255,255,.85);color:var(--green);border:1.5px solid var(--gold)!important;} .btn-alt:hover{background:#FFFBEF;transform:translateY(-1px);}
|
| 179 |
+
.btns button:disabled{opacity:.5;cursor:not-allowed;transform:none;}
|
| 180 |
+
#status{direction:rtl;text-align:center;color:var(--muted);font-size:.93rem;margin:8px 0;} #status:empty{display:none;}
|
| 181 |
+
.boot{padding:14px 18px;margin:0 0 14px;direction:rtl;text-align:center;}
|
| 182 |
+
.boot .step{text-align:right;}
|
| 183 |
+
.boot h2{margin:0 0 8px;font-size:.98rem;color:var(--green);}
|
| 184 |
+
.boot .step{display:grid;grid-template-columns:22px minmax(0,1fr) 54px;gap:10px;align-items:center;margin:8px 0;font-size:.92rem;color:var(--muted);}
|
| 185 |
+
.boot .dot{width:18px;height:18px;border-radius:50%;border:2px solid rgba(15,76,58,.22);display:inline-block;}
|
| 186 |
+
.boot .step.active .dot{border-color:rgba(15,76,58,.18);border-top-color:var(--gold);animation:icv-spin .8s linear infinite;}
|
| 187 |
+
.boot .step.done .dot{background:var(--ok);border-color:var(--ok);position:relative;} .boot .step.done .dot::after{content:"";position:absolute;inset:3px 5px 5px 5px;border:solid #FFF;border-width:0 2px 2px 0;transform:rotate(40deg);}
|
| 188 |
+
.boot .step.done{color:var(--ok);} .boot .step.active{color:var(--ink);font-weight:600;}
|
| 189 |
+
.boot .meter{height:8px;border-radius:999px;background:rgba(15,76,58,.1);overflow:hidden;} .boot .meter i{display:block;height:100%;width:0;background:linear-gradient(270deg,var(--green),var(--gold-2));transition:width .3s ease;}
|
| 190 |
+
.boot .step.pending .meter{position:relative;} .boot .pct{font-size:.8rem;direction:ltr;text-align:left;}
|
| 191 |
+
.boot.ready{padding:8px 16px;} .boot.ready h2{margin:0;} .boot.ready .step{display:none;}
|
| 192 |
+
.banner{margin:0 0 12px;}
|
| 193 |
+
</style>
|
| 194 |
+
</head>
|
| 195 |
+
<body class="icv-page">
|
| 196 |
+
<div class="page">
|
| 197 |
+
|
| 198 |
+
<div class="icv"><div class="hero">
|
| 199 |
+
<svg class="mark" viewBox="0 0 64 64" fill="none" stroke="#E2C06E" stroke-width="1.8" aria-hidden="true"><rect x="14" y="14" width="36" height="36"/><rect x="14" y="14" width="36" height="36" transform="rotate(45 32 32)"/><circle cx="32" cy="32" r="7" fill="#E2C06E" stroke="none"/></svg>
|
| 200 |
+
<h1>التحقق من هلوسة القرآن والحديث وتصحيحها</h1>
|
| 201 |
+
<p class="tagline">تحقّق من آيات القرآن والأحاديث النبوية بالدليل</p>
|
| 202 |
+
<p class="sub">يكتشف النظام الاقتباسات داخل أي نص، ويسترجع نصوصها من المصادر، ويقارنها كلمةً بكلمة، ثم يعرض الدليل.
|
| 203 |
+
وإذا لم تكفِ الأدلة فلن يختلق تصحيحًا، بل يحيل الحالة إلى المراجعة البشرية.</p>
|
| 204 |
+
<div class="flow"><span>كشف</span><i>‹</i><span>استرجاع</span><i>‹</i><span>محاذاة</span><i>‹</i><span>دليل</span><i>‹</i><span>قرار</span></div>
|
| 205 |
+
</div></div>
|
| 206 |
+
|
| 207 |
+
<div class="boot glass" id="boot" aria-live="polite">
|
| 208 |
+
<h2 id="boot-title">جارٍ تجهيز النظام في متصفحك…</h2>
|
| 209 |
+
<div class="step pending" data-step="python"><i class="dot"></i><span>بيئة بايثون</span><span class="pct"></span><div class="meter" style="grid-column:2 / span 2"><i></i></div></div>
|
| 210 |
+
<div class="step pending" data-step="code"><i class="dot"></i><span>الكود وفهرس القرآن</span><span class="pct"></span><div class="meter" style="grid-column:2 / span 2"><i></i></div></div>
|
| 211 |
+
<div class="step pending" data-step="hadith"><i class="dot"></i><span>كتب الحديث (تُحمَّل بالتوازي)</span><span class="pct"></span><div class="meter" style="grid-column:2 / span 2"><i></i></div></div>
|
| 212 |
+
</div>
|
| 213 |
+
<div class="tabs glass" role="tablist"><button class="tab active" role="tab" aria-selected="true" data-tab="direct">تحقّق مباشر</button><button class="tab" role="tab" aria-selected="false" data-tab="ask">اسأل ثم تحقّق</button></div>
|
| 214 |
+
<section id="tab-direct" class="form glass">
|
| 215 |
+
<label for="text">النص المراد التحقق منه</label>
|
| 216 |
+
<textarea id="text" placeholder="الصق هنا النص الذي ولّده نموذج لغوي. يكتشف النظام الآيات والأحاديث الواردة فيه، بعلامات تنصيص أو بدونها…"></textarea>
|
| 217 |
+
<div class="btns"><button id="verify" class="btn-main">تحقّق من النص</button><button id="example" class="btn-alt">جرّب مثالًا</button></div>
|
| 218 |
+
</section>
|
| 219 |
+
<section id="tab-ask" class="form glass" hidden>
|
| 220 |
+
<div class="icv"><div class="notice">اكتب سؤالًا، وسيجيب عنه النظام مباشرةً، ثم يفحص كل آية وحديث في الإجابة ويعرض الأخطاء والتصحيحات.</div></div>
|
| 221 |
+
<label for="prompt">سؤالك</label>
|
| 222 |
+
<textarea id="prompt" class="short" maxlength="1500" placeholder="اكتب سؤالك، مثل: اشرح لي فضل الصبر في القرآن والسنة مع ذكر الأدلة."></textarea>
|
| 223 |
+
<div class="btns"><button id="ask" class="btn-main">اسأل ثم تحقّق</button><button id="ask-example" class="btn-alt">جرّب مثالًا</button></div>
|
| 224 |
+
</section>
|
| 225 |
+
<div id="status" role="status"></div>
|
| 226 |
+
<div id="results" class="results"></div>
|
| 227 |
+
<div class="icv"><div class="disclaimer">أداة مساعدة للتدقيق النصي وليست فتوى ولا بديلًا عن المراجعة المتخصصة.
|
| 228 |
+
النتائج مبنية على مراجع القرآن الكريم والكتب الستة المضمّنة فقط.</div></div>
|
| 229 |
+
</div>
|
| 230 |
+
<script>
|
| 231 |
+
(function () {
|
| 232 |
+
if (window.icvCopy) return;
|
| 233 |
+
function toast(message, bad) {
|
| 234 |
+
var t = document.getElementById('icv-toast');
|
| 235 |
+
if (!t) { t = document.createElement('div'); t.id = 'icv-toast'; t.setAttribute('role', 'status'); t.setAttribute('aria-live', 'polite'); document.body.appendChild(t); }
|
| 236 |
+
t.textContent = message; t.className = 'show' + (bad ? ' bad' : '');
|
| 237 |
+
clearTimeout(window.__icvToast); window.__icvToast = setTimeout(function () { t.className = bad ? 'bad' : ''; }, 2400);
|
| 238 |
+
}
|
| 239 |
+
function legacyCopy(text) {
|
| 240 |
+
var area = document.createElement('textarea'); area.value = text; area.setAttribute('readonly', '');
|
| 241 |
+
area.style.cssText = 'position:fixed;top:0;left:0;opacity:0;pointer-events:none'; document.body.appendChild(area);
|
| 242 |
+
area.select(); area.setSelectionRange(0, text.length); var ok = false;
|
| 243 |
+
try { ok = document.execCommand('copy'); } catch (e) { ok = false; }
|
| 244 |
+
document.body.removeChild(area); return ok;
|
| 245 |
+
}
|
| 246 |
+
window.icvToast = toast;
|
| 247 |
+
window.icvCopy = function (button) {
|
| 248 |
+
var text = button.getAttribute('data-text') || '';
|
| 249 |
+
function finish(ok) {
|
| 250 |
+
toast(ok ? 'تم نسخ النص بنجاح' : 'تعذّر النسخ، حدّد النص وانسخه يدويًا', !ok);
|
| 251 |
+
if (ok) { button.classList.add('done'); setTimeout(function () { button.classList.remove('done'); }, 1600); }
|
| 252 |
+
}
|
| 253 |
+
if (navigator.clipboard && window.isSecureContext) {
|
| 254 |
+
navigator.clipboard.writeText(text).then(function () { finish(true); }, function () { finish(legacyCopy(text)); });
|
| 255 |
+
} else { finish(legacyCopy(text)); }
|
| 256 |
+
};
|
| 257 |
+
window.icvDownload = function (button) {
|
| 258 |
+
var blob = new Blob([button.getAttribute('data-text') || ''], { type: 'text/tab-separated-values;charset=utf-8' });
|
| 259 |
+
var link = document.createElement('a'); link.href = URL.createObjectURL(blob); link.download = button.getAttribute('data-name') || 'results.tsv';
|
| 260 |
+
document.body.appendChild(link); link.click(); document.body.removeChild(link); setTimeout(function () { URL.revokeObjectURL(link.href); }, 1000);
|
| 261 |
+
};
|
| 262 |
+
})();
|
| 263 |
+
</script>
|
| 264 |
+
<script type="text/plain" id="worker-src">
|
| 265 |
+
// Python (Pyodide) and the whole verification pipeline run here, off the main thread.
|
| 266 |
+
let py = null, booted = null, hadithReady = null;
|
| 267 |
+
const post = (message) => self.postMessage(message);
|
| 268 |
+
const step = (name, state, extra) => post(Object.assign({ type: "step", name, state }, extra || {}));
|
| 269 |
+
|
| 270 |
+
async function openCache(buildId) {
|
| 271 |
+
try {
|
| 272 |
+
const names = await caches.keys();
|
| 273 |
+
await Promise.all(names.filter((n) => n.startsWith("icv-") && n !== "icv-" + buildId).map((n) => caches.delete(n))); // purge old builds
|
| 274 |
+
return await caches.open("icv-" + buildId);
|
| 275 |
+
} catch (e) { return null; }
|
| 276 |
+
}
|
| 277 |
+
|
| 278 |
+
async function fetchBytes(url, cache, onProgress) {
|
| 279 |
+
if (cache) { try { const hit = await cache.match(url); if (hit) return { bytes: new Uint8Array(await hit.arrayBuffer()), cached: true }; } catch (e) { /* fall through */ } }
|
| 280 |
+
const response = await fetch(url);
|
| 281 |
+
if (!response.ok) throw new Error("تعذّر تحميل " + url + " (" + response.status + ")");
|
| 282 |
+
if (cache) { try { await cache.put(url, response.clone()); } catch (e) { /* the cache is an optimisation only */ } }
|
| 283 |
+
const total = Number(response.headers.get("Content-Length")) || 0;
|
| 284 |
+
if (!response.body || !onProgress) return { bytes: new Uint8Array(await response.arrayBuffer()), cached: false };
|
| 285 |
+
const reader = response.body.getReader(), chunks = []; let received = 0;
|
| 286 |
+
for (;;) { const { done, value } = await reader.read(); if (done) break; chunks.push(value); received += value.length; onProgress(received, total); }
|
| 287 |
+
const bytes = new Uint8Array(received); let offset = 0;
|
| 288 |
+
for (const chunk of chunks) { bytes.set(chunk, offset); offset += chunk.length; }
|
| 289 |
+
return { bytes, cached: false };
|
| 290 |
+
}
|
| 291 |
+
|
| 292 |
+
// Accept both layouts: files inside folders (index/, demo/) or uploaded flat next to index.html.
|
| 293 |
+
async function fetchAny(file, base, cache, onProgress) {
|
| 294 |
+
let lastError = null;
|
| 295 |
+
for (const path of [file, file.split("/").pop()]) {
|
| 296 |
+
try { return await fetchBytes(new URL(path, base).href, cache, onProgress); } catch (error) { lastError = error; }
|
| 297 |
+
}
|
| 298 |
+
throw lastError;
|
| 299 |
+
}
|
| 300 |
+
|
| 301 |
+
// A server that sets Content-Encoding: gzip on .gz files makes the browser inflate them; Python expects gzip, so re-pack.
|
| 302 |
+
async function ensureGzip(file, bytes) {
|
| 303 |
+
if (!file.endsWith(".gz") || (bytes[0] === 0x1f && bytes[1] === 0x8b)) return bytes;
|
| 304 |
+
const stream = new Blob([bytes]).stream().pipeThrough(new CompressionStream("gzip"));
|
| 305 |
+
return new Uint8Array(await new Response(stream).arrayBuffer());
|
| 306 |
+
}
|
| 307 |
+
|
| 308 |
+
async function boot(message) {
|
| 309 |
+
const cache = await openCache(message.buildId);
|
| 310 |
+
step("python", "active");
|
| 311 |
+
importScripts("https://cdn.jsdelivr.net/pyodide/v" + message.pyodideVersion + "/full/pyodide.js");
|
| 312 |
+
const hadithFile = message.lazy[0];
|
| 313 |
+
// the large Hadith index starts downloading now, in parallel with the Python runtime
|
| 314 |
+
const hadithDownload = fetchAny(hadithFile, message.base, cache, (got, total) => step("hadith", "active", { pct: total ? Math.round((got / total) * 100) : null, mb: (got / 1048576).toFixed(1) }));
|
| 315 |
+
hadithDownload.catch(() => {});
|
| 316 |
+
py = await loadPyodide();
|
| 317 |
+
step("python", "done");
|
| 318 |
+
step("code", "active", { pct: 0 });
|
| 319 |
+
for (const dir of ["/app", "/app/index", "/app/demo", "/app/data"]) py.FS.mkdir(dir);
|
| 320 |
+
let cached = true, n = 0;
|
| 321 |
+
for (const file of message.files) {
|
| 322 |
+
const result = await fetchAny(file, message.base, cache);
|
| 323 |
+
cached = cached && result.cached;
|
| 324 |
+
py.FS.writeFile("/app/" + file, await ensureGzip(file, result.bytes));
|
| 325 |
+
step("code", "active", { pct: Math.round((++n / message.files.length) * 100) });
|
| 326 |
+
}
|
| 327 |
+
const examplesText = new TextDecoder().decode(py.FS.readFile("/app/demo/examples.json"));
|
| 328 |
+
py.runPython("import sys; sys.path.insert(0, '/app')");
|
| 329 |
+
py.runPython("from app import get_pipeline, verify_text, verify_generated_answer; get_pipeline()");
|
| 330 |
+
step("code", "done", { cached });
|
| 331 |
+
post({ type: "ready", examples: JSON.parse(examplesText) });
|
| 332 |
+
hadithReady = (async () => {
|
| 333 |
+
const result = await hadithDownload;
|
| 334 |
+
step("hadith", "active", { pct: 100, prepare: true });
|
| 335 |
+
py.FS.writeFile("/app/" + hadithFile, await ensureGzip(hadithFile, result.bytes));
|
| 336 |
+
py.runPython("get_pipeline().retriever.warm()");
|
| 337 |
+
step("hadith", "done", { cached: result.cached });
|
| 338 |
+
post({ type: "hadith" });
|
| 339 |
+
})();
|
| 340 |
+
await hadithReady;
|
| 341 |
+
}
|
| 342 |
+
|
| 343 |
+
self.onmessage = async (event) => {
|
| 344 |
+
const message = event.data;
|
| 345 |
+
if (message.type === "boot") {
|
| 346 |
+
booted = boot(message);
|
| 347 |
+
try { await booted; } catch (error) { post({ type: "error", text: String((error && error.message) || error) }); }
|
| 348 |
+
} else if (message.type === "verify") {
|
| 349 |
+
try {
|
| 350 |
+
await booted;
|
| 351 |
+
const html = py.globals.get(message.generated ? "verify_generated_answer" : "verify_text")(message.text, message.entities || "");
|
| 352 |
+
post({ type: "result", id: message.id, html });
|
| 353 |
+
} catch (error) { post({ type: "result", id: message.id, html: null }); }
|
| 354 |
+
}
|
| 355 |
+
};
|
| 356 |
+
</script>
|
| 357 |
+
<script>
|
| 358 |
+
const DEFAULT_CONFIG = {"askEndpoint": "https://icv-ask-proxy.ghada-islamic-verifier-2026.workers.dev", "hf": {"model": "", "endpoint": "https://router.huggingface.co/hf-inference/models/{model}"}};
|
| 359 |
+
const BUILD_ID = "eae236d51019";
|
| 360 |
+
const SKELETON = "<div class=\"icv\" aria-busy=\"true\" aria-label=\"جارٍ التحقق\"><div class=\"skel-tiles\"><div class=\"skel skel-tile\"></div><div class=\"skel skel-tile\"></div><div class=\"skel skel-tile\"></div><div class=\"skel skel-tile\"></div></div><div class=\"skel-card\"><div class=\"skel skel-line w40\"></div><div class=\"skel skel-line w85\"></div><div class=\"skel skel-line\"></div><div class=\"skel skel-line w60\"></div></div><div class=\"skel-card\"><div class=\"skel skel-line w40\"></div><div class=\"skel skel-line w85\"></div><div class=\"skel skel-line\"></div><div class=\"skel skel-line w60\"></div></div></div>";
|
| 361 |
+
const MAX_PROMPT = 1500;
|
| 362 |
+
const $ = (id) => document.getElementById(id);
|
| 363 |
+
let cfg = DEFAULT_CONFIG, examples = [], exampleIndex = 0, askExampleIndex = 0, nextId = 0, worker = null, ready = false;
|
| 364 |
+
const pending = new Map();
|
| 365 |
+
|
| 366 |
+
const store = {
|
| 367 |
+
get(key) { try { return localStorage.getItem(key); } catch (e) { return null; } },
|
| 368 |
+
set(key, value) { try { localStorage.setItem(key, value); } catch (e) { /* private mode / quota: caching is optional */ } },
|
| 369 |
+
};
|
| 370 |
+
function purgeOldLocalCache() {
|
| 371 |
+
try { for (const key of Object.keys(localStorage)) if (key.startsWith("icv:examples:") && key !== "icv:examples:" + BUILD_ID) localStorage.removeItem(key); } catch (e) { /* ignore */ }
|
| 372 |
+
}
|
| 373 |
+
|
| 374 |
+
const esc = (s) => String(s).replace(/[&<>"]/g, (c) => ({ "&": "&", "<": "<", ">": ">", '"': """ }[c]));
|
| 375 |
+
function setStatus(text, busy = true) { $("status").innerHTML = (busy && text ? '<span class="icv-spinner"></span>' : "") + esc(text); }
|
| 376 |
+
function setBusy(flag) { for (const id of ["verify", "example", "ask", "ask-example"]) $(id).disabled = flag; }
|
| 377 |
+
function notice(text, kind) { return '<div class="icv banner"><div class="notice ' + kind + '">' + esc(text) + "</div></div>"; }
|
| 378 |
+
function show(html, banner) { $("results").innerHTML = (banner || "") + html; }
|
| 379 |
+
function showSkeleton() { $("results").innerHTML = SKELETON; }
|
| 380 |
+
|
| 381 |
+
function verifyRemote(payload) {
|
| 382 |
+
return new Promise((resolve) => { const id = ++nextId; pending.set(id, resolve); worker.postMessage(Object.assign({ type: "verify", id }, payload)); });
|
| 383 |
+
}
|
| 384 |
+
|
| 385 |
+
// ---- detection: the bundled rule + corpus detector always runs; a fine-tuned CAMeLBERT-MSA hosted on Hugging Face (when configured)
|
| 386 |
+
// runs alongside it and both results are merged inside the pipeline. Nothing here is visible to the visitor and any failure is silent.
|
| 387 |
+
async function hostedEntities(text) {
|
| 388 |
+
const model = (cfg.hf && cfg.hf.model || "").trim();
|
| 389 |
+
if (!model) return "";
|
| 390 |
+
const controller = new AbortController(), timer = setTimeout(() => controller.abort(), 6000);
|
| 391 |
+
try {
|
| 392 |
+
const response = await fetch(cfg.hf.endpoint.replace("{model}", encodeURIComponent(model)), {
|
| 393 |
+
method: "POST", headers: { "Content-Type": "application/json" }, signal: controller.signal,
|
| 394 |
+
body: JSON.stringify({ inputs: text, parameters: { aggregation_strategy: "simple" } }),
|
| 395 |
+
});
|
| 396 |
+
if (!response.ok) return "";
|
| 397 |
+
const data = await response.json();
|
| 398 |
+
return Array.isArray(data) ? JSON.stringify(data) : "";
|
| 399 |
+
} catch (e) { return ""; } finally { clearTimeout(timer); }
|
| 400 |
+
}
|
| 401 |
+
|
| 402 |
+
async function verifyAuto(text, generated) {
|
| 403 |
+
const entities = await hostedEntities(text);
|
| 404 |
+
return verifyRemote({ text, generated, entities });
|
| 405 |
+
}
|
| 406 |
+
|
| 407 |
+
// Small models sometimes drift into Chinese etc.; strip scripts that never belong in an Arabic answer (the server does it too).
|
| 408 |
+
const FOREIGN = /[\u0400-\u04ff\u0900-\u097f\u0e00-\u0e7f\u1100-\u11ff\u3000-\u303f\u3040-\u30ff\u3130-\u318f\u3400-\u4dbf\u4e00-\u9fff\uac00-\ud7af\uf900-\ufaff\uff00-\uffef]+/g;
|
| 409 |
+
const cleanAnswer = (t) => String(t).replace(FOREIGN, " ").replace(/([،,؛.])\s*[،,؛]+/g, "$1").replace(/[ \t]{2,}/g, " ").replace(/ +([،؛.:])/g, "$1").trim();
|
| 410 |
+
|
| 411 |
+
async function runDirect() {
|
| 412 |
+
if (!$("text").value.trim()) return show(notice("الرجاء إدخال نص للتحقق منه.", "warn"));
|
| 413 |
+
setBusy(true); showSkeleton(); setStatus(ready ? "جارٍ التحقق…" : "جارٍ تجهيز النظام ثم التحقق…");
|
| 414 |
+
const html = await verifyAuto($("text").value, false);
|
| 415 |
+
show(html !== null ? html : notice("حدث خطأ غير متوقع أثناء التحقق.", "bad"));
|
| 416 |
+
setStatus("", false); setBusy(false);
|
| 417 |
+
}
|
| 418 |
+
|
| 419 |
+
// ---- ask then verify: same-origin proxy that holds the key; saved sample answer when it is not reachable ----------------
|
| 420 |
+
async function askServer(prompt) {
|
| 421 |
+
const response = await fetch(cfg.askEndpoint, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ prompt }) });
|
| 422 |
+
if (!response.ok) {
|
| 423 |
+
const error = new Error("http_" + response.status); error.status = response.status;
|
| 424 |
+
try { error.code = (await response.json()).error; } catch (e) { error.code = ""; }
|
| 425 |
+
throw error;
|
| 426 |
+
}
|
| 427 |
+
const data = await response.json();
|
| 428 |
+
if (!data.answer) throw new Error("empty");
|
| 429 |
+
return cleanAnswer(data.answer);
|
| 430 |
+
}
|
| 431 |
+
|
| 432 |
+
async function runAsk() {
|
| 433 |
+
const prompt = $("prompt").value.trim();
|
| 434 |
+
if (!prompt) return show(notice("اكتب سؤالًا أولًا.", "warn"));
|
| 435 |
+
if (prompt.length > MAX_PROMPT) return show(notice("السؤال طويل جدًا (الحد الأقصى " + MAX_PROMPT + " حرف).", "warn"));
|
| 436 |
+
setBusy(true); showSkeleton(); setStatus("جارٍ إعداد الإجابة…");
|
| 437 |
+
let answer = null, banner = "";
|
| 438 |
+
try { answer = await askServer(prompt); }
|
| 439 |
+
catch (error) {
|
| 440 |
+
const stop = (message) => { show(notice(message, "warn")); setStatus("", false); setBusy(false); };
|
| 441 |
+
if (error.status === 402) return stop("رصيد حساب OpenAI المرتبط بالخدمة غير كافٍ؛ على صاحب الحساب مراجعة الفوترة ثم المحاولة مجددًا.");
|
| 442 |
+
if (error.status === 429) return stop("الطلبات كثيرة الآن؛ انتظر قليلًا ثم أعد المحاولة.");
|
| 443 |
+
if (error.status === 413) return stop("السؤال طويل جدًا.");
|
| 444 |
+
if (error.code === "upstream_auth") return stop("مفتاح الخدمة على الخادم غير صالح؛ يلزم تحديثه من صاحب الحساب.");
|
| 445 |
+
if (examples.length) {
|
| 446 |
+
answer = examples[exampleIndex++ % examples.length].text;
|
| 447 |
+
banner = notice("خدمة الإجابة المباشرة غير متاحة الآن؛ عُرضت إجابة تجريبية محفوظة لتوضيح خطوات التحقق.", "warn");
|
| 448 |
+
}
|
| 449 |
+
}
|
| 450 |
+
if (answer === null) { show(notice("تعذّر الحصول على إجابة الآن؛ حاول لاحقًا.", "warn")); setStatus("", false); return setBusy(false); }
|
| 451 |
+
setStatus("جارٍ التحقق من الإجابة…");
|
| 452 |
+
const html = await verifyAuto(answer, true);
|
| 453 |
+
show(html !== null ? html : notice("تعذّر التحقق من الإجابة.", "bad"), banner);
|
| 454 |
+
setStatus("", false); setBusy(false);
|
| 455 |
+
}
|
| 456 |
+
|
| 457 |
+
// Saved scenario: fills the question and verifies a stored model-style answer (no live call), so every case can be tried at once.
|
| 458 |
+
async function runAskExample() {
|
| 459 |
+
const samples = examples.filter((e) => e.question);
|
| 460 |
+
if (!samples.length) return show(notice("الأمثلة لم تُحمَّل بعد؛ انتظر لحظة ثم أعد المحاولة.", "warn"));
|
| 461 |
+
const sample = samples[askExampleIndex++ % samples.length];
|
| 462 |
+
$("prompt").value = sample.question;
|
| 463 |
+
setBusy(true); showSkeleton(); setStatus("جارٍ التحقق من الإجابة…");
|
| 464 |
+
const html = await verifyAuto(sample.text, true);
|
| 465 |
+
show(html !== null ? html : notice("تعذّر التحقق من الإجابة.", "bad"),
|
| 466 |
+
notice("إجابة تجريبية محفوظة لحالة «" + sample.title + "»، لم تُرسل إلى أي خدمة. لتجربة إجابة حيّة اكتب سؤالك واضغط «اسأل ثم تحقّق».", ""));
|
| 467 |
+
setStatus("", false); setBusy(false);
|
| 468 |
+
}
|
| 469 |
+
|
| 470 |
+
// ---- boot panel ------------------------------------------------------------------------------------------------------
|
| 471 |
+
function applyStep(m) {
|
| 472 |
+
const row = document.querySelector('.boot .step[data-step="' + m.name + '"]');
|
| 473 |
+
if (!row) return;
|
| 474 |
+
row.className = "step " + m.state;
|
| 475 |
+
const pct = m.state === "done" ? 100 : (m.pct == null ? 0 : m.pct);
|
| 476 |
+
row.querySelector(".meter i").style.width = pct + "%";
|
| 477 |
+
row.querySelector(".pct").textContent = m.state === "done" ? (m.cached ? "محفوظ" : "تم") : (m.prepare ? "…" : (m.pct == null ? (m.mb ? m.mb + " MB" : "") : pct + "%"));
|
| 478 |
+
if (document.querySelectorAll(".boot .step.done").length === 3) { $("boot").classList.add("ready"); $("boot-title").textContent = "النظام جاهز · يعمل كله داخل متصفحك ولا يُرسل نصك إلى أي خادم"; }
|
| 479 |
+
}
|
| 480 |
+
|
| 481 |
+
function start() {
|
| 482 |
+
try { worker = new Worker(URL.createObjectURL(new Blob([$("worker-src").textContent], { type: "text/javascript" }))); }
|
| 483 |
+
catch (error) { return setStatus("تعذّر تشغيل النظام: " + error.message, false); }
|
| 484 |
+
worker.onerror = (event) => setStatus("تعذّر تشغيل النظام: " + (event.message || ""), false);
|
| 485 |
+
worker.onmessage = (event) => {
|
| 486 |
+
const m = event.data;
|
| 487 |
+
if (m.type === "step") applyStep(m);
|
| 488 |
+
else if (m.type === "error") setStatus("تعذّر تشغيل النظام: " + m.text + ". جرّب متصفح كمبيوتر حديثًا ثم أعد تحميل الصفحة.", false);
|
| 489 |
+
else if (m.type === "result") { const resolve = pending.get(m.id); pending.delete(m.id); resolve(m.html); }
|
| 490 |
+
else if (m.type === "ready") { examples = m.examples; ready = true; store.set("icv:examples:" + BUILD_ID, JSON.stringify(m.examples)); }
|
| 491 |
+
};
|
| 492 |
+
worker.postMessage({ type: "boot", base: location.href, buildId: BUILD_ID, pyodideVersion: "0.26.4", files: ["app.py", "verifier.py", "retrieval.py", "index_builder.py", "normalization.py", "alignment.py", "similarity.py", "detector.py", "scanner.py", "idgham.py", "ui.py", "llm_client.py", "benchmark_format.py", "camelbert_adapter.py", "index/quran.idx.gz", "demo/examples.json"], lazy: ["index/hadith.idx.gz"] });
|
| 493 |
+
}
|
| 494 |
+
|
| 495 |
+
async function loadConfig() {
|
| 496 |
+
try {
|
| 497 |
+
const response = await fetch("config.json", { cache: "no-store" });
|
| 498 |
+
if (response.ok) { cfg = Object.assign({}, DEFAULT_CONFIG, await response.json()); store.set("icv:config", JSON.stringify(cfg)); return; }
|
| 499 |
+
} catch (e) { /* offline or missing: fall back to the last known configuration */ }
|
| 500 |
+
try { const saved = store.get("icv:config"); if (saved) cfg = Object.assign({}, DEFAULT_CONFIG, JSON.parse(saved)); } catch (e) { /* ignore */ }
|
| 501 |
+
}
|
| 502 |
+
|
| 503 |
+
document.querySelectorAll(".tab").forEach((tab) => tab.addEventListener("click", () => {
|
| 504 |
+
document.querySelectorAll(".tab").forEach((t) => { const on = t === tab; t.classList.toggle("active", on); t.setAttribute("aria-selected", on); });
|
| 505 |
+
for (const name of ["direct", "ask"]) $("tab-" + name).hidden = tab.dataset.tab !== name;
|
| 506 |
+
}));
|
| 507 |
+
$("verify").addEventListener("click", runDirect);
|
| 508 |
+
$("ask").addEventListener("click", runAsk);
|
| 509 |
+
$("ask-example").addEventListener("click", runAskExample);
|
| 510 |
+
$("example").addEventListener("click", () => { if (!examples.length) return; $("text").value = examples[exampleIndex++ % examples.length].text; runDirect(); });
|
| 511 |
+
|
| 512 |
+
purgeOldLocalCache();
|
| 513 |
+
try { const saved = store.get("icv:examples:" + BUILD_ID); if (saved) examples = JSON.parse(saved); } catch (e) { examples = []; }
|
| 514 |
+
loadConfig();
|
| 515 |
+
start();
|
| 516 |
+
</script>
|
| 517 |
+
</body>
|
| 518 |
</html>
|
index_builder.py
ADDED
|
@@ -0,0 +1,163 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Build the pre-tokenised search indexes shipped in ``index/``.
|
| 2 |
+
|
| 3 |
+
python index_builder.py # writes index/quran.idx.gz and index/hadith.idx.gz
|
| 4 |
+
|
| 5 |
+
An index is a gzip-compressed pickle holding the raw records, their phonetic skeletons and BM25 postings, so the
|
| 6 |
+
application never re-tokenises the corpora at start-up. The pickles are produced locally by this script and read
|
| 7 |
+
back only by this project; never load an index file from an untrusted source.
|
| 8 |
+
"""
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
import gzip
|
| 12 |
+
import json
|
| 13 |
+
import math
|
| 14 |
+
import pickle
|
| 15 |
+
import sys
|
| 16 |
+
import time
|
| 17 |
+
import zlib
|
| 18 |
+
from array import array
|
| 19 |
+
from collections import Counter, defaultdict
|
| 20 |
+
from pathlib import Path
|
| 21 |
+
from typing import Dict, List, Sequence, Tuple
|
| 22 |
+
|
| 23 |
+
from normalization import content_words, normalize_for_matching, normalize_strict, phonetic_key, tokenize
|
| 24 |
+
|
| 25 |
+
INDEX_VERSION = 3
|
| 26 |
+
BM25_K1, BM25_B = 1.2, 0.75
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
class BM25Index:
|
| 30 |
+
"""Okapi BM25 over a postings table ``term -> (doc ids, term frequencies)``."""
|
| 31 |
+
|
| 32 |
+
def __init__(self, postings: Dict[str, Tuple[array, array]], doc_len: array) -> None:
|
| 33 |
+
self.postings, self.doc_len = postings, doc_len
|
| 34 |
+
self.n_docs = len(doc_len)
|
| 35 |
+
self.avg_len = (sum(doc_len) / self.n_docs) if self.n_docs else 1.0
|
| 36 |
+
self.idf = {t: math.log(1 + (self.n_docs - len(d) + 0.5) / (len(d) + 0.5)) for t, (d, _) in postings.items()}
|
| 37 |
+
|
| 38 |
+
def search(self, terms: Sequence[str], top_k: int) -> List[Tuple[int, float]]:
|
| 39 |
+
scores: Dict[int, float] = defaultdict(float)
|
| 40 |
+
for term in set(terms):
|
| 41 |
+
entry = self.postings.get(term)
|
| 42 |
+
if entry is None:
|
| 43 |
+
continue
|
| 44 |
+
idf = self.idf[term]
|
| 45 |
+
for doc, tf in zip(*entry):
|
| 46 |
+
norm = 1 - BM25_B + BM25_B * self.doc_len[doc] / self.avg_len
|
| 47 |
+
scores[doc] += idf * tf * (BM25_K1 + 1) / (tf + BM25_K1 * norm)
|
| 48 |
+
return sorted(scores.items(), key=lambda item: item[1], reverse=True)[:top_k]
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
def _postings(documents: List[List[str]]) -> Tuple[Dict[str, Tuple[array, array]], array]:
|
| 52 |
+
docs: Dict[str, array] = defaultdict(lambda: array("I"))
|
| 53 |
+
tfs: Dict[str, array] = defaultdict(lambda: array("H"))
|
| 54 |
+
doc_len = array("I")
|
| 55 |
+
for doc_id, tokens in enumerate(documents):
|
| 56 |
+
doc_len.append(len(tokens))
|
| 57 |
+
for term, tf in Counter(tokens).items():
|
| 58 |
+
docs[term].append(doc_id)
|
| 59 |
+
tfs[term].append(min(tf, 65535))
|
| 60 |
+
return {term: (docs[term], tfs[term]) for term in docs}, doc_len
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
def _read_json(path: Path) -> list:
|
| 64 |
+
opener = gzip.open if path.suffix == ".gz" else open
|
| 65 |
+
with opener(path, "rt", encoding="utf-8") as handle:
|
| 66 |
+
data = json.load(handle)
|
| 67 |
+
if not isinstance(data, list):
|
| 68 |
+
raise ValueError(f"Unexpected corpus format in {path}: expected a JSON list")
|
| 69 |
+
return data
|
| 70 |
+
|
| 71 |
+
|
| 72 |
+
|
| 73 |
+
def anchor_keys(tokens: Sequence[str]) -> List[Tuple[int, int]]:
|
| 74 |
+
"""Seed keys for unannounced-quotation search: ``(crc32, position)`` for every phonetic trigram plus two gapped
|
| 75 |
+
trigrams that survive a single substituted or inserted word."""
|
| 76 |
+
keys = [phonetic_key(t) for t in tokens]
|
| 77 |
+
out = []
|
| 78 |
+
for i in range(len(keys) - 2):
|
| 79 |
+
out.append((zlib.crc32(f"{keys[i]} {keys[i + 1]} {keys[i + 2]}".encode()), i))
|
| 80 |
+
if i + 3 < len(keys):
|
| 81 |
+
out.append((zlib.crc32(f"{keys[i]} {keys[i + 1]} _ {keys[i + 3]}".encode()), i))
|
| 82 |
+
out.append((zlib.crc32(f"{keys[i]} _ {keys[i + 2]} {keys[i + 3]}".encode()), i))
|
| 83 |
+
return out
|
| 84 |
+
|
| 85 |
+
|
| 86 |
+
def build_quran_index(path: Path) -> dict:
|
| 87 |
+
records, norm, word_count = [], [], []
|
| 88 |
+
all_index: Dict[str, List[int]] = defaultdict(list)
|
| 89 |
+
by_surah: Dict[int, Dict[int, int]] = defaultdict(dict)
|
| 90 |
+
documents: List[List[str]] = []
|
| 91 |
+
anchors: Dict[int, List[Tuple[int, int]]] = defaultdict(list)
|
| 92 |
+
for entry in _read_json(Path(path)):
|
| 93 |
+
text = (entry.get("ayah_text") or "").strip()
|
| 94 |
+
if not text:
|
| 95 |
+
continue
|
| 96 |
+
idx = len(records)
|
| 97 |
+
records.append({"surah_id": entry.get("surah_id"), "surah_name": entry.get("surah_name", ""),
|
| 98 |
+
"ayah_id": entry.get("ayah_id"), "text": text})
|
| 99 |
+
by_surah[entry.get("surah_id")][entry.get("ayah_id")] = idx
|
| 100 |
+
norm.append(normalize_for_matching(text))
|
| 101 |
+
strict_tokens = tokenize(normalize_strict(text))
|
| 102 |
+
word_count.append(len(strict_tokens))
|
| 103 |
+
for word in set(strict_tokens):
|
| 104 |
+
all_index[word].append(idx)
|
| 105 |
+
documents.append(content_words(norm[-1].split()))
|
| 106 |
+
for key, pos in anchor_keys(norm[-1].split()):
|
| 107 |
+
anchors[key].append((idx, pos))
|
| 108 |
+
postings, doc_len = _postings(documents)
|
| 109 |
+
return {"version": INDEX_VERSION, "records": records, "norm": norm, "word_count": word_count,
|
| 110 |
+
"all_index": dict(all_index), "by_surah": dict(by_surah), "postings": postings, "doc_len": doc_len,
|
| 111 |
+
"anchors": dict(anchors)}
|
| 112 |
+
|
| 113 |
+
|
| 114 |
+
def build_hadith_index(path: Path) -> dict:
|
| 115 |
+
records, documents = [], []
|
| 116 |
+
for entry in _read_json(Path(path)):
|
| 117 |
+
if not entry:
|
| 118 |
+
continue
|
| 119 |
+
matn = (entry.get("Matn") or "").strip() or None
|
| 120 |
+
full = (entry.get("hadithTxt") or "").strip() or None
|
| 121 |
+
if not (matn or full):
|
| 122 |
+
continue
|
| 123 |
+
records.append({"hadithID": entry.get("hadithID"), "book": entry.get("BookID"), "title": entry.get("title"),
|
| 124 |
+
"matn": matn, "full": full})
|
| 125 |
+
words = content_words(normalize_for_matching(full or matn).split())
|
| 126 |
+
if matn and full: # words of the matn that the full text may lack
|
| 127 |
+
extra = set(content_words(normalize_for_matching(matn).split())) - set(words)
|
| 128 |
+
words += sorted(extra)
|
| 129 |
+
documents.append(words)
|
| 130 |
+
postings, doc_len = _postings(documents)
|
| 131 |
+
return {"version": INDEX_VERSION, "records": records, "postings": postings, "doc_len": doc_len}
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
def save_index(index: dict, path: Path) -> None:
|
| 135 |
+
path = Path(path)
|
| 136 |
+
path.parent.mkdir(parents=True, exist_ok=True)
|
| 137 |
+
with gzip.open(path, "wb", compresslevel=6) as handle:
|
| 138 |
+
pickle.dump(index, handle, protocol=pickle.HIGHEST_PROTOCOL)
|
| 139 |
+
|
| 140 |
+
|
| 141 |
+
def load_index(path: Path) -> dict:
|
| 142 |
+
with gzip.open(path, "rb") as handle:
|
| 143 |
+
index = pickle.load(handle) # produced by save_index() from this project's own data
|
| 144 |
+
if index.get("version") != INDEX_VERSION:
|
| 145 |
+
raise ValueError("Index version mismatch; rebuild with index_builder.py")
|
| 146 |
+
return index
|
| 147 |
+
|
| 148 |
+
|
| 149 |
+
def main() -> None:
|
| 150 |
+
base = Path(__file__).resolve().parent
|
| 151 |
+
out = base / "index"
|
| 152 |
+
for name, builder, source in (("quran", build_quran_index, base / "data" / "quran.json"),
|
| 153 |
+
("hadith", build_hadith_index, base / "data" / "hadith.json")):
|
| 154 |
+
started = time.time()
|
| 155 |
+
source = source if source.is_file() else source.with_name(source.name + ".gz")
|
| 156 |
+
index = builder(source)
|
| 157 |
+
save_index(index, out / f"{name}.idx.gz")
|
| 158 |
+
size = (out / f"{name}.idx.gz").stat().st_size / 1e6
|
| 159 |
+
print(f"{name}: {len(index['records'])} records -> index/{name}.idx.gz ({size:.1f} MB, {time.time() - started:.1f}s)")
|
| 160 |
+
|
| 161 |
+
|
| 162 |
+
if __name__ == "__main__":
|
| 163 |
+
sys.exit(main())
|
islamic_unified_dataset.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
islamiceval_dev_subset.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
llm_client.py
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Ask-then-verify backend: one pre-configured OpenAI (ChatGPT) client.
|
| 2 |
+
|
| 3 |
+
There is no provider or model choice for the user. The API key is **never** part of the source code, the page or the
|
| 4 |
+
repository (the competition rules forbid secrets in the repository, and a key shipped to a browser is public): it is read
|
| 5 |
+
from the ``OPENAI_API_KEY`` environment variable on the server side. The browser page talks to a same-origin proxy
|
| 6 |
+
(``functions/api/ask.js``, a Cloudflare Pages Function) that holds the key as a platform secret; this module serves the
|
| 7 |
+
local Gradio app and the tests with the same request shapes.
|
| 8 |
+
|
| 9 |
+
Environment: OPENAI_API_KEY (required) OPENAI_MODEL (optional, default below) OPENAI_BASE_URL (optional, tests / gateways)
|
| 10 |
+
"""
|
| 11 |
+
from __future__ import annotations
|
| 12 |
+
|
| 13 |
+
import json
|
| 14 |
+
import os
|
| 15 |
+
import urllib.error
|
| 16 |
+
import urllib.request
|
| 17 |
+
from dataclasses import dataclass
|
| 18 |
+
from typing import Dict, Optional, Tuple
|
| 19 |
+
|
| 20 |
+
SYSTEM_PROMPT = (
|
| 21 |
+
"أنت مساعد معرفي في العلوم الإسلامية. أجب بالعربية بإيجاز ودقة. عند الاستشهاد بآية قرآنية أو حديث نبوي اكتب نصه كاملًا "
|
| 22 |
+
"بين علامتي تنصيص مزدوجتين \"...\" بعد عبارة تمهيدية مثل: قال الله تعالى: أو قال رسول الله ﷺ:. "
|
| 23 |
+
"لا تضع بين علامات التنصيص إلا نص الآية أو الحديث، واذكر السورة ورقم الآية أو مصدر الحديث بعد الاقتباس."
|
| 24 |
+
)
|
| 25 |
+
|
| 26 |
+
DEFAULT_MODEL = "gpt-4o-mini"
|
| 27 |
+
DEFAULT_BASE_URL = "https://api.openai.com/v1"
|
| 28 |
+
MAX_PROMPT_CHARS = 1500
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
class LLMError(RuntimeError):
|
| 32 |
+
"""Raised with a user-presentable Arabic message."""
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
@dataclass
|
| 36 |
+
class LLMSettings:
|
| 37 |
+
api_key: str = ""
|
| 38 |
+
model: str = ""
|
| 39 |
+
base_url: str = "" # override for tests or a compatible gateway
|
| 40 |
+
timeout: float = 60.0
|
| 41 |
+
|
| 42 |
+
@classmethod
|
| 43 |
+
def from_env(cls) -> "LLMSettings":
|
| 44 |
+
return cls(api_key=os.environ.get("OPENAI_API_KEY", ""), model=os.environ.get("OPENAI_MODEL", ""),
|
| 45 |
+
base_url=os.environ.get("OPENAI_BASE_URL", ""))
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
def build_request(settings: LLMSettings, prompt: str) -> Tuple[str, Dict[str, str], dict]:
|
| 49 |
+
"""``(url, headers, json_body)`` for OpenAI chat completions."""
|
| 50 |
+
base = (settings.base_url or DEFAULT_BASE_URL).rstrip("/")
|
| 51 |
+
body = {"model": settings.model.strip() or DEFAULT_MODEL,
|
| 52 |
+
"messages": [{"role": "system", "content": SYSTEM_PROMPT}, {"role": "user", "content": prompt}]}
|
| 53 |
+
headers = {"Content-Type": "application/json", "Authorization": f"Bearer {settings.api_key}"}
|
| 54 |
+
return f"{base}/chat/completions", headers, body
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
def parse_response(payload: dict) -> str:
|
| 58 |
+
try:
|
| 59 |
+
text = payload["choices"][0]["message"]["content"]
|
| 60 |
+
except (KeyError, IndexError, TypeError) as exc:
|
| 61 |
+
raise LLMError("وصلت استجابة غير متوقعة من النموذج.") from exc
|
| 62 |
+
if not text or not text.strip():
|
| 63 |
+
raise LLMError("لم يُرجع النموذج أي نص.")
|
| 64 |
+
return text.strip()
|
| 65 |
+
|
| 66 |
+
|
| 67 |
+
_HTTP_MESSAGES = {
|
| 68 |
+
400: "رفض المزوّد الطلب.",
|
| 69 |
+
401: "خدمة الإجابة غير مهيّأة بعد (المفتاح غير صالح).",
|
| 70 |
+
403: "الخدمة غير مصرّح لها باستخدام هذا النموذج.",
|
| 71 |
+
404: "النموذج غير متاح لدى المزوّد.",
|
| 72 |
+
429: "تجاوزت حد الاستخدام المسموح؛ حاول لاحقًا.",
|
| 73 |
+
}
|
| 74 |
+
|
| 75 |
+
|
| 76 |
+
def generate(settings: Optional[LLMSettings], prompt: str) -> str:
|
| 77 |
+
"""Return the model's answer. Raises ``LLMError`` with an Arabic message on any failure."""
|
| 78 |
+
settings = settings or LLMSettings.from_env()
|
| 79 |
+
if not prompt or not prompt.strip():
|
| 80 |
+
raise LLMError("اكتب سؤالًا أولًا.")
|
| 81 |
+
if len(prompt) > MAX_PROMPT_CHARS:
|
| 82 |
+
raise LLMError(f"السؤال طويل جدًا (الحد الأقصى {MAX_PROMPT_CHARS} حرف).")
|
| 83 |
+
if not settings.api_key or not settings.api_key.strip():
|
| 84 |
+
raise LLMError("خدمة الإجابة غير مهيّأة: لم يُضبط مفتاح الخادم.")
|
| 85 |
+
url, headers, body = build_request(settings, prompt.strip())
|
| 86 |
+
request = urllib.request.Request(url, data=json.dumps(body).encode("utf-8"), headers=headers, method="POST")
|
| 87 |
+
try:
|
| 88 |
+
with urllib.request.urlopen(request, timeout=settings.timeout) as response:
|
| 89 |
+
payload = json.loads(response.read().decode("utf-8"))
|
| 90 |
+
except urllib.error.HTTPError as exc:
|
| 91 |
+
raise LLMError(_HTTP_MESSAGES.get(exc.code, f"فشل الطلب (الرمز {exc.code}).")) from exc
|
| 92 |
+
except (urllib.error.URLError, TimeoutError) as exc:
|
| 93 |
+
raise LLMError("تعذّر الاتصال بالمزوّد؛ تحقق من الإنترنت.") from exc
|
| 94 |
+
except json.JSONDecodeError as exc:
|
| 95 |
+
raise LLMError("وصلت استجابة غير مقروءة من المزوّد.") from exc
|
| 96 |
+
return parse_response(payload)
|
normalization.py
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Phonetic-aware Arabic normalisation shared by indexing, retrieval, alignment and verification.
|
| 2 |
+
|
| 3 |
+
Three levels, from gentle to aggressive:
|
| 4 |
+
|
| 5 |
+
* ``normalize_strict`` diacritics / tatweel / Quranic marks removed, alef forms unified (hamza on waw / ya kept).
|
| 6 |
+
* ``normalize_lenient`` strict + ta marbuta -> ha and alef maqsura -> ya (used for Hadith, whose spelling varies).
|
| 7 |
+
* ``normalize_for_matching`` the phonetic skeleton used for retrieval and word alignment: additionally folds hamza
|
| 8 |
+
carriers and keeps Arabic letters only.
|
| 9 |
+
|
| 10 |
+
All levels also reconcile the Uthmani mushaf script with Modern Standard Arabic: alef wasla (ٱ) becomes alef, the
|
| 11 |
+
dagger alef / "alif khanjariyah" (ـٰ) and tatweel are removed, Persian ya / kaf and the ligature ﷲ are mapped to Arabic
|
| 12 |
+
letters, and zero-width / bidi control characters are dropped.
|
| 13 |
+
"""
|
| 14 |
+
from __future__ import annotations
|
| 15 |
+
|
| 16 |
+
import re
|
| 17 |
+
import unicodedata
|
| 18 |
+
from typing import List, Tuple
|
| 19 |
+
|
| 20 |
+
_ZERO_WIDTH = re.compile("[\u200b-\u200f\u202a-\u202e\u2066-\u2069\ufeff]")
|
| 21 |
+
_MARKS = re.compile(r"[\u0610-\u061A\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E4\u06E7\u06E8\u06EA-\u06ED\u0640]")
|
| 22 |
+
_MATCH_MARKS = re.compile(r"[\u0610-\u061A\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E8\u06EA-\u06ED\u0640]")
|
| 23 |
+
_PUNCTUATION = re.compile(r"[،؛؟!،.,:;'\"()\[\]{}<>«»\-_/\\|@#$%^&*+=~`﴿﴾]")
|
| 24 |
+
_SPACES = re.compile(r"\s+")
|
| 25 |
+
_ALEF_FORMS = re.compile(r"[أإآٱ]")
|
| 26 |
+
_PERSIAN = str.maketrans({"ی": "ي", "ې": "ي", "ک": "ك", "ە": "ه", "ہ": "ه", "ۀ": "ه"})
|
| 27 |
+
|
| 28 |
+
STOPWORDS = frozenset(
|
| 29 |
+
"""من في على ان أن إن الى إلى عن مع ما لا لم لن قد و ثم أو او هو هي هم انت أنتم كان كانت يكون تكون قال قالت
|
| 30 |
+
هذا هذه ذلك تلك الذي التي الذين اللاتي اللائي كل بعض غير عند بين حتى إذا اذا لو لكن بل يا أيها ايها""".split()
|
| 31 |
+
)
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def _prefold(text: str) -> str:
|
| 35 |
+
"""Script-level clean-up common to every normalisation level."""
|
| 36 |
+
text = unicodedata.normalize("NFC", text).replace("ﷲ", "الله")
|
| 37 |
+
return _ZERO_WIDTH.sub("", text).translate(_PERSIAN)
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
def normalize_strict(text) -> str:
|
| 41 |
+
if not text:
|
| 42 |
+
return ""
|
| 43 |
+
text = _MARKS.sub("", _prefold(text))
|
| 44 |
+
text = _ALEF_FORMS.sub("ا", text)
|
| 45 |
+
return _SPACES.sub(" ", _PUNCTUATION.sub(" ", text)).strip()
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
def normalize_lenient(text) -> str:
|
| 49 |
+
return normalize_strict(text).replace("ة", "ه").replace("ى", "ي")
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
def normalize_for_matching(text) -> str:
|
| 53 |
+
"""Phonetic skeleton: Arabic letters only, hamza carriers / ta marbuta / alef maqsura folded."""
|
| 54 |
+
if not text:
|
| 55 |
+
return ""
|
| 56 |
+
text = _MATCH_MARKS.sub("", _prefold(text))
|
| 57 |
+
for source, target in (("أ", "ا"), ("إ", "ا"), ("آ", "ا"), ("ٱ", "ا"), ("ؤ", "و"), ("ئ", "ي"), ("ة", "ه"), ("ى", "ي")):
|
| 58 |
+
text = text.replace(source, target)
|
| 59 |
+
text = re.sub(r"[^\u0621-\u064A\s]", " ", text) # Arabic letters only: drops Arabic punctuation, digits, Latin
|
| 60 |
+
return _SPACES.sub(" ", text).strip()
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
def tokenize(text: str) -> List[str]:
|
| 64 |
+
return [token for token in text.split() if token]
|
| 65 |
+
|
| 66 |
+
|
| 67 |
+
def content_words(tokens) -> List[str]:
|
| 68 |
+
"""Drop stop-words and single-letter tokens."""
|
| 69 |
+
return [t for t in tokens if t not in STOPWORDS and len(t) > 1]
|
| 70 |
+
|
| 71 |
+
|
| 72 |
+
def char_ngrams(text: str, n: int = 4) -> set:
|
| 73 |
+
return {text[i : i + n] for i in range(len(text) - n + 1)}
|
| 74 |
+
|
| 75 |
+
|
| 76 |
+
# ---- word-level helpers used by the aligner -----------------------------------------------------------------------
|
| 77 |
+
_ARABIC_LETTER = re.compile(r"[\u0621-\u064A]")
|
| 78 |
+
|
| 79 |
+
|
| 80 |
+
def aligned_words(text: str) -> List[Tuple[str, str]]:
|
| 81 |
+
"""``[(original_word, phonetic_skeleton)]``; ayah markers like ``(12)`` and mark-only tokens are dropped."""
|
| 82 |
+
pairs = []
|
| 83 |
+
for word in text.split():
|
| 84 |
+
skeleton = normalize_for_matching(word)
|
| 85 |
+
if skeleton and _ARABIC_LETTER.search(skeleton):
|
| 86 |
+
pairs.append((word, skeleton.replace(" ", "")))
|
| 87 |
+
return pairs
|
| 88 |
+
|
| 89 |
+
|
| 90 |
+
_VOWELS = {"\u064E": "a", "\u064F": "u", "\u0650": "i", "\u064B": "A", "\u064C": "U", "\u064D": "I", "\u0651": "~"}
|
| 91 |
+
|
| 92 |
+
|
| 93 |
+
def vowel_signature(word: str) -> List[Tuple[str, str]]:
|
| 94 |
+
"""Per base letter, the set of short vowels / tanween / shadda written on it (sukun and Quranic marks ignored)."""
|
| 95 |
+
word = _prefold(word).replace("ٱ", "ا")
|
| 96 |
+
signature: List[Tuple[str, str]] = []
|
| 97 |
+
for char in word:
|
| 98 |
+
if char in _VOWELS:
|
| 99 |
+
if signature:
|
| 100 |
+
letter, marks = signature[-1]
|
| 101 |
+
signature[-1] = (letter, "".join(sorted(set(marks + _VOWELS[char]))))
|
| 102 |
+
elif "\u0621" <= char <= "\u064A":
|
| 103 |
+
signature.append((char, ""))
|
| 104 |
+
return signature
|
| 105 |
+
|
| 106 |
+
|
| 107 |
+
# ---- phonetic sequence matching --------------------------------------------------------------------------------
|
| 108 |
+
# Letters that are commonly confused in writing or dictation are folded into one class, so an altered or mis-spelled
|
| 109 |
+
# quotation can still be *found*; the verifier then reports the exact differences.
|
| 110 |
+
_PHONETIC_CLASSES = {"ص": "س", "ث": "س", "ذ": "ز", "ظ": "ز", "ض": "ز", "ط": "ت", "ك": "ق", "ح": "ه", "غ": "ع"}
|
| 111 |
+
_PHONETIC_TABLE = str.maketrans(_PHONETIC_CLASSES)
|
| 112 |
+
|
| 113 |
+
|
| 114 |
+
def phonetic_key(skeleton_word: str) -> str:
|
| 115 |
+
"""Sound-alike key of a phonetic-skeleton word (confusable consonants folded, doubled letters collapsed)."""
|
| 116 |
+
word = skeleton_word.translate(_PHONETIC_TABLE)
|
| 117 |
+
return "".join(ch for i, ch in enumerate(word) if i == 0 or ch != word[i - 1])
|
quran.idx.gz
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:87eb1db1dffa15d95d757b157a1569cad93e6265c15736c9f4ee62c17b2a2c98
|
| 3 |
+
size 2181000
|
quran.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
retrieval.py
ADDED
|
@@ -0,0 +1,187 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Source retrieval over the Quran and the six Hadith books: BM25 candidate search with pre-built, lazily loaded indexes.
|
| 2 |
+
|
| 3 |
+
Start-up cost is kept small by shipping the corpora as pre-tokenised, gzip-compressed pickle indexes
|
| 4 |
+
(``index/quran.idx.gz``, ``index/hadith.idx.gz``, produced by ``index_builder.py``):
|
| 5 |
+
|
| 6 |
+
* the small Quran index loads with the retriever;
|
| 7 |
+
* the large Hadith index loads on first use (or via ``warm()``);
|
| 8 |
+
* if an index file is missing it is rebuilt from ``data/*.json`` (slower, a few seconds) so the code never breaks.
|
| 9 |
+
|
| 10 |
+
Retrieval pipeline
|
| 11 |
+
Quran : word-vote F1 (coverage x precision, exact-quote friendly) + BM25 candidates
|
| 12 |
+
Hadith: BM25 recall, then character 4-gram cosine re-rank
|
| 13 |
+
"""
|
| 14 |
+
from __future__ import annotations
|
| 15 |
+
|
| 16 |
+
import logging
|
| 17 |
+
import threading
|
| 18 |
+
from collections import OrderedDict
|
| 19 |
+
from pathlib import Path
|
| 20 |
+
from typing import Dict, Iterable, List, Optional, Sequence, Tuple
|
| 21 |
+
|
| 22 |
+
from index_builder import BM25Index, build_hadith_index, build_quran_index, load_index, save_index
|
| 23 |
+
from normalization import char_ngrams, content_words, normalize_for_matching, normalize_lenient, normalize_strict, tokenize
|
| 24 |
+
|
| 25 |
+
logger = logging.getLogger(__name__)
|
| 26 |
+
|
| 27 |
+
BASE_DIR = Path(__file__).resolve().parent
|
| 28 |
+
DATA_DIR = BASE_DIR / "data"
|
| 29 |
+
INDEX_DIR = BASE_DIR / "index"
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
class CorpusError(RuntimeError):
|
| 33 |
+
"""Raised when a corpus or index file is missing or malformed."""
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
class SourceRetriever:
|
| 37 |
+
"""Quran + Hadith retriever. ``quran`` is available immediately; ``hadith`` is loaded lazily."""
|
| 38 |
+
|
| 39 |
+
def __init__(self, index_dir: Path = INDEX_DIR, data_dir: Path = DATA_DIR) -> None:
|
| 40 |
+
self.index_dir, self.data_dir = Path(index_dir), Path(data_dir)
|
| 41 |
+
self._hadith_lock = threading.Lock()
|
| 42 |
+
self._hadith: Optional[dict] = None
|
| 43 |
+
self._norm_cache: "OrderedDict[Tuple[int, str], str]" = OrderedDict()
|
| 44 |
+
|
| 45 |
+
quran = self._load("quran", build_quran_index)
|
| 46 |
+
self.quran: List[dict] = quran["records"]
|
| 47 |
+
self.q_norm_match: List[str] = quran["norm"]
|
| 48 |
+
self.q_word_count: List[int] = quran["word_count"]
|
| 49 |
+
self.q_all_index: Dict[str, List[int]] = quran["all_index"]
|
| 50 |
+
self.quran_by_surah: Dict[int, Dict[int, int]] = quran["by_surah"]
|
| 51 |
+
self.quran_bm25 = BM25Index(quran["postings"], quran["doc_len"])
|
| 52 |
+
self.quran_anchors: Dict[int, List[Tuple[int, int]]] = quran["anchors"]
|
| 53 |
+
logger.info("Quran index ready: %d ayahs", len(self.quran))
|
| 54 |
+
|
| 55 |
+
# ---- loading ----------------------------------------------------------------------------------------------
|
| 56 |
+
def _load(self, name: str, builder) -> dict:
|
| 57 |
+
path = self.index_dir / f"{name}.idx.gz"
|
| 58 |
+
if path.is_file():
|
| 59 |
+
try:
|
| 60 |
+
return load_index(path)
|
| 61 |
+
except Exception:
|
| 62 |
+
logger.warning("Index %s is unreadable; rebuilding from data/", path)
|
| 63 |
+
source = self.data_dir / ("quran.json" if name == "quran" else "hadith.json")
|
| 64 |
+
gz = source.with_name(source.name + ".gz")
|
| 65 |
+
source = source if source.is_file() else gz
|
| 66 |
+
if not source.is_file():
|
| 67 |
+
raise CorpusError(f"Neither {path} nor the source corpus {source} was found")
|
| 68 |
+
index = builder(source)
|
| 69 |
+
try:
|
| 70 |
+
save_index(index, path)
|
| 71 |
+
except OSError:
|
| 72 |
+
logger.info("Could not cache %s (read-only file system); continuing in memory", path)
|
| 73 |
+
return index
|
| 74 |
+
|
| 75 |
+
def warm(self) -> None:
|
| 76 |
+
"""Load the Hadith index now (otherwise it loads on the first Hadith query)."""
|
| 77 |
+
_ = self.hadith_data
|
| 78 |
+
|
| 79 |
+
@property
|
| 80 |
+
def hadith_loaded(self) -> bool:
|
| 81 |
+
return self._hadith is not None
|
| 82 |
+
|
| 83 |
+
@property
|
| 84 |
+
def hadith_data(self) -> dict:
|
| 85 |
+
if self._hadith is None:
|
| 86 |
+
with self._hadith_lock:
|
| 87 |
+
if self._hadith is None:
|
| 88 |
+
data = self._load("hadith", build_hadith_index)
|
| 89 |
+
data["bm25"] = BM25Index(data["postings"], data["doc_len"])
|
| 90 |
+
self._hadith = data
|
| 91 |
+
logger.info("Hadith index ready: %d records", len(data["records"]))
|
| 92 |
+
return self._hadith
|
| 93 |
+
|
| 94 |
+
@property
|
| 95 |
+
def hadith(self) -> List[dict]:
|
| 96 |
+
return self.hadith_data["records"]
|
| 97 |
+
|
| 98 |
+
# ---- Quran ------------------------------------------------------------------------------------------------
|
| 99 |
+
@property
|
| 100 |
+
def quran_vocabulary(self) -> frozenset:
|
| 101 |
+
"""Phonetic-skeleton words that occur in the Quran (lets the aligner tell spelling variants from real words)."""
|
| 102 |
+
if getattr(self, "_vocab", None) is None:
|
| 103 |
+
self._vocab = frozenset(word for text in self.q_norm_match for word in text.split())
|
| 104 |
+
return self._vocab
|
| 105 |
+
|
| 106 |
+
def search_quran_ayahs(self, query: str, top_k: int = 25, extra_bm25: int = 10) -> List[dict]:
|
| 107 |
+
"""Single-ayah candidates: word-vote F1 first, then the best BM25 matches (rare-word hits for partial quotes)."""
|
| 108 |
+
query_words = tokenize(normalize_strict(query))
|
| 109 |
+
if not query_words:
|
| 110 |
+
return []
|
| 111 |
+
votes: Dict[int, int] = {}
|
| 112 |
+
for word in query_words:
|
| 113 |
+
for idx in self.q_all_index.get(word, ()):
|
| 114 |
+
votes[idx] = votes.get(idx, 0) + 1
|
| 115 |
+
scored: List[Tuple[int, float]] = []
|
| 116 |
+
for idx, vote in votes.items():
|
| 117 |
+
coverage = vote / len(query_words)
|
| 118 |
+
precision = vote / self.q_word_count[idx] if self.q_word_count[idx] else 0.0
|
| 119 |
+
f1 = 2 * coverage * precision / (coverage + precision) if coverage + precision > 0 else 0.0
|
| 120 |
+
scored.append((idx, f1))
|
| 121 |
+
scored.sort(key=lambda item: item[1], reverse=True)
|
| 122 |
+
scores = dict(scored)
|
| 123 |
+
ranked = [idx for idx, _ in scored[:top_k]]
|
| 124 |
+
seen = set(ranked)
|
| 125 |
+
bm25_hits = self.quran_bm25.search(content_words(normalize_for_matching(query).split()), extra_bm25)
|
| 126 |
+
ranked += [idx for idx, _ in bm25_hits if idx not in seen]
|
| 127 |
+
results = []
|
| 128 |
+
for idx in ranked:
|
| 129 |
+
candidate = dict(self.quran[idx])
|
| 130 |
+
candidate.update(type="Quran", retrieval_score=scores.get(idx, 0.0))
|
| 131 |
+
results.append(candidate)
|
| 132 |
+
return results
|
| 133 |
+
|
| 134 |
+
def quran_seed_ayahs(self, query_words: Sequence[str], top_k: int = 25) -> List[int]:
|
| 135 |
+
"""BM25-ranked ayah indices used to seed the multi-ayah window search."""
|
| 136 |
+
return [idx for idx, _ in self.quran_bm25.search(query_words, top_k)]
|
| 137 |
+
|
| 138 |
+
# ---- Hadith -----------------------------------------------------------------------------------------------
|
| 139 |
+
def hadith_candidates(self, query_words: Sequence[str], top_k: int) -> List[int]:
|
| 140 |
+
return [idx for idx, _ in self.hadith_data["bm25"].search(query_words, top_k)]
|
| 141 |
+
|
| 142 |
+
def hadith_norm(self, idx: int, field: str) -> Optional[str]:
|
| 143 |
+
"""Phonetic skeleton of a Hadith field (``matn`` or ``full``), computed on demand and cached."""
|
| 144 |
+
record = self.hadith[idx]
|
| 145 |
+
raw = record.get(field)
|
| 146 |
+
if not raw:
|
| 147 |
+
return None
|
| 148 |
+
key = (idx, field)
|
| 149 |
+
if key in self._norm_cache:
|
| 150 |
+
self._norm_cache.move_to_end(key)
|
| 151 |
+
return self._norm_cache[key]
|
| 152 |
+
value = normalize_for_matching(raw)
|
| 153 |
+
self._norm_cache[key] = value
|
| 154 |
+
if len(self._norm_cache) > 4096:
|
| 155 |
+
self._norm_cache.popitem(last=False)
|
| 156 |
+
return value
|
| 157 |
+
|
| 158 |
+
def search_hadith(self, query: str, top_k: int = 15, pool: int = 60) -> List[dict]:
|
| 159 |
+
"""BM25 recall, then character 4-gram cosine re-rank."""
|
| 160 |
+
words = content_words(normalize_for_matching(query).split())
|
| 161 |
+
if not words:
|
| 162 |
+
return []
|
| 163 |
+
query_grams = char_ngrams(normalize_lenient(query))
|
| 164 |
+
results = []
|
| 165 |
+
for idx in self.hadith_candidates(words, pool):
|
| 166 |
+
record = self.hadith[idx]
|
| 167 |
+
text = record["matn"] or record["full"]
|
| 168 |
+
doc_grams = char_ngrams(normalize_lenient(text))
|
| 169 |
+
cosine = (
|
| 170 |
+
len(query_grams & doc_grams) / ((len(query_grams) * len(doc_grams)) ** 0.5)
|
| 171 |
+
if query_grams and doc_grams
|
| 172 |
+
else 0.0
|
| 173 |
+
)
|
| 174 |
+
results.append(
|
| 175 |
+
{
|
| 176 |
+
"type": "Hadith",
|
| 177 |
+
"idx": idx,
|
| 178 |
+
"hadithID": record["hadithID"],
|
| 179 |
+
"book": record["book"],
|
| 180 |
+
"title": record["title"],
|
| 181 |
+
"text": text,
|
| 182 |
+
"has_matn": bool(record["matn"]),
|
| 183 |
+
"retrieval_score": cosine,
|
| 184 |
+
}
|
| 185 |
+
)
|
| 186 |
+
results.sort(key=lambda c: c["retrieval_score"], reverse=True)
|
| 187 |
+
return results[:top_k]
|
scanner.py
ADDED
|
@@ -0,0 +1,349 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Detection of unannounced and embedded quotations (no quotation marks, no introductory phrase needed).
|
| 2 |
+
|
| 3 |
+
The scanner reads the text as a stream of phonetic-skeleton words and looks for stretches that follow the Quran or a
|
| 4 |
+
Hadith closely:
|
| 5 |
+
|
| 6 |
+
1. *Seeds* a sliding window of 3 words (plus two gapped variants that survive one changed word) is looked up in the
|
| 7 |
+
pre-built phonetic n-gram index of the Quran.
|
| 8 |
+
2. *Chaining* seeds of the same ayah on a consistent diagonal are chained into a candidate region; regions of adjacent
|
| 9 |
+
ayahs that touch in the text are merged.
|
| 10 |
+
3. *Boundaries* the region grows word by word (tolerating one substituted word) while the text keeps following the
|
| 11 |
+
source, and never crosses a sentence boundary on its own; this is the contextual boundary step.
|
| 12 |
+
4. *Evidence* a region is kept only if enough of its words match and the matched words are rare enough (summed IDF), so
|
| 13 |
+
everyday phrases that merely occur in the Quran are not reported.
|
| 14 |
+
5. *Hadith* clause-sized windows are sent to BM25, and the best record is aligned word by word with the same rules.
|
| 15 |
+
|
| 16 |
+
Phonetic keys make the search tolerant to sound-alike spelling, but the verifier still reports every real difference.
|
| 17 |
+
"""
|
| 18 |
+
from __future__ import annotations
|
| 19 |
+
|
| 20 |
+
import re
|
| 21 |
+
from collections import defaultdict
|
| 22 |
+
from dataclasses import dataclass
|
| 23 |
+
from difflib import SequenceMatcher
|
| 24 |
+
from typing import Dict, List, Optional, Sequence, Tuple
|
| 25 |
+
|
| 26 |
+
from alignment import aligned_words
|
| 27 |
+
from detector import DetectedSpan, RuleDetector, trim_span
|
| 28 |
+
from index_builder import anchor_keys
|
| 29 |
+
from normalization import content_words, normalize_for_matching, phonetic_key
|
| 30 |
+
from retrieval import SourceRetriever
|
| 31 |
+
|
| 32 |
+
_ARABIC = re.compile(r"[\u0621-\u064A]")
|
| 33 |
+
_STRONG_BOUNDARY = re.compile(r"[.؟?!؛;:\n…]")
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
@dataclass
|
| 37 |
+
class Token:
|
| 38 |
+
skeleton: str
|
| 39 |
+
key: str
|
| 40 |
+
start: int
|
| 41 |
+
end: int
|
| 42 |
+
boundary_after: bool # a sentence-level punctuation mark or line break follows this word
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
def tokenize_with_offsets(text: str) -> List[Token]:
|
| 46 |
+
matches = list(re.finditer(r"\S+", text))
|
| 47 |
+
tokens: List[Token] = []
|
| 48 |
+
for i, match in enumerate(matches):
|
| 49 |
+
skeleton = normalize_for_matching(match.group()).replace(" ", "")
|
| 50 |
+
if not skeleton or not _ARABIC.search(skeleton):
|
| 51 |
+
if tokens and _STRONG_BOUNDARY.search(match.group()):
|
| 52 |
+
tokens[-1].boundary_after = True
|
| 53 |
+
continue
|
| 54 |
+
gap_end = matches[i + 1].start() if i + 1 < len(matches) else len(text)
|
| 55 |
+
trailing = text[match.end():gap_end] + match.group()[-2:]
|
| 56 |
+
tokens.append(Token(skeleton, phonetic_key(skeleton), match.start(), match.end(),
|
| 57 |
+
bool(_STRONG_BOUNDARY.search(trailing))))
|
| 58 |
+
return tokens
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
@dataclass
|
| 62 |
+
class _Region:
|
| 63 |
+
start: int # token index in the run (inclusive)
|
| 64 |
+
end: int # exclusive
|
| 65 |
+
label: str
|
| 66 |
+
matched: int
|
| 67 |
+
ratio: float
|
| 68 |
+
idf: float
|
| 69 |
+
surah: Optional[int] = None
|
| 70 |
+
first_ayah: Optional[int] = None
|
| 71 |
+
last_ayah: Optional[int] = None
|
| 72 |
+
|
| 73 |
+
|
| 74 |
+
class CorpusScanner:
|
| 75 |
+
"""Finds Quran / Hadith stretches in free text. Thresholds are deliberately conservative."""
|
| 76 |
+
|
| 77 |
+
def __init__(self, retriever: SourceRetriever, min_tokens: int = 4, min_ratio: float = 0.7, min_idf: float = 8.0,
|
| 78 |
+
hadith_min_tokens: int = 6, hadith_min_idf: float = 14.0, scan_hadith: bool = True) -> None:
|
| 79 |
+
self.kb = retriever
|
| 80 |
+
self.min_tokens, self.min_ratio, self.min_idf = min_tokens, min_ratio, min_idf
|
| 81 |
+
self.hadith_min_tokens, self.hadith_min_idf = hadith_min_tokens, hadith_min_idf
|
| 82 |
+
self.scan_hadith = scan_hadith
|
| 83 |
+
|
| 84 |
+
# ---- public -----------------------------------------------------------------------------------------------
|
| 85 |
+
def scan(self, text: str, exclude: Sequence[Tuple[int, int]] = ()) -> List[DetectedSpan]:
|
| 86 |
+
tokens = tokenize_with_offsets(text)
|
| 87 |
+
runs = self._runs(tokens, exclude)
|
| 88 |
+
spans: List[DetectedSpan] = []
|
| 89 |
+
for run in runs:
|
| 90 |
+
for region in self._merge(self._scan_quran(run)):
|
| 91 |
+
spans.append(self._to_span(text, run, region))
|
| 92 |
+
if self.scan_hadith:
|
| 93 |
+
for run in runs:
|
| 94 |
+
taken = [(s.start, s.end) for s in spans]
|
| 95 |
+
for region in self._scan_hadith(run, taken):
|
| 96 |
+
spans.append(self._to_span(text, run, region))
|
| 97 |
+
return sorted(spans, key=lambda s: s.start)
|
| 98 |
+
|
| 99 |
+
# ---- helpers ----------------------------------------------------------------------------------------------
|
| 100 |
+
@staticmethod
|
| 101 |
+
def _runs(tokens: List[Token], exclude: Sequence[Tuple[int, int]]) -> List[List[Token]]:
|
| 102 |
+
runs, current = [], []
|
| 103 |
+
for token in tokens:
|
| 104 |
+
if any(token.start < e and token.end > s for s, e in exclude):
|
| 105 |
+
if current:
|
| 106 |
+
runs.append(current)
|
| 107 |
+
current = []
|
| 108 |
+
else:
|
| 109 |
+
current.append(token)
|
| 110 |
+
if current:
|
| 111 |
+
runs.append(current)
|
| 112 |
+
return runs
|
| 113 |
+
|
| 114 |
+
@staticmethod
|
| 115 |
+
def _to_span(text: str, run: List[Token], region: _Region) -> DetectedSpan:
|
| 116 |
+
start, end = trim_span(text, run[region.start].start, run[region.end - 1].end)
|
| 117 |
+
return DetectedSpan(start, end, region.label, round(region.ratio, 3), "scan", text[start:end])
|
| 118 |
+
|
| 119 |
+
# ---- Quran ------------------------------------------------------------------------------------------------
|
| 120 |
+
def _scan_quran(self, run: List[Token]) -> List[_Region]:
|
| 121 |
+
kb = self.kb
|
| 122 |
+
if len(run) < self.min_tokens:
|
| 123 |
+
return []
|
| 124 |
+
hits: Dict[int, List[Tuple[int, int]]] = defaultdict(list)
|
| 125 |
+
for key_hash, pos in anchor_keys([t.skeleton for t in run]):
|
| 126 |
+
postings = kb.quran_anchors.get(key_hash)
|
| 127 |
+
if not postings or len(postings) > 60: # skip phrases that occur everywhere
|
| 128 |
+
continue
|
| 129 |
+
for ayah, source_pos in postings:
|
| 130 |
+
hits[ayah].append((pos, source_pos))
|
| 131 |
+
|
| 132 |
+
regions: List[_Region] = self._whole_ayahs(run)
|
| 133 |
+
for ayah, found in hits.items():
|
| 134 |
+
found.sort()
|
| 135 |
+
chains: List[dict] = []
|
| 136 |
+
for pos, source_pos in found:
|
| 137 |
+
diagonal = source_pos - pos
|
| 138 |
+
for chain in chains:
|
| 139 |
+
if abs(diagonal - chain["d"]) <= 2 and pos - chain["last"] <= 5:
|
| 140 |
+
chain["hits"].append((pos, source_pos))
|
| 141 |
+
chain["last"], chain["d"] = pos, diagonal
|
| 142 |
+
break
|
| 143 |
+
else:
|
| 144 |
+
chains.append({"d": diagonal, "last": pos, "hits": [(pos, source_pos)]})
|
| 145 |
+
for chain in chains:
|
| 146 |
+
region = self._grow(run, ayah, chain)
|
| 147 |
+
if region is not None:
|
| 148 |
+
regions.append(region)
|
| 149 |
+
return self._non_overlapping(regions)
|
| 150 |
+
|
| 151 |
+
def _whole_ayah_index(self) -> Dict[str, list]:
|
| 152 |
+
"""first word -> [(words, ayah index)] for every ayah of at least ``min_tokens`` words (built once)."""
|
| 153 |
+
if getattr(self, "_whole", None) is None:
|
| 154 |
+
index: Dict[str, list] = defaultdict(list)
|
| 155 |
+
for i, norm in enumerate(self.kb.q_norm_match):
|
| 156 |
+
words = tuple(w for w in norm.split() if w)
|
| 157 |
+
if len(words) >= self.min_tokens:
|
| 158 |
+
index[words[0]].append((words, i))
|
| 159 |
+
for entries in index.values():
|
| 160 |
+
entries.sort(key=lambda e: -len(e[0])) # longest first
|
| 161 |
+
self._whole = index
|
| 162 |
+
return self._whole
|
| 163 |
+
|
| 164 |
+
def _whole_ayahs(self, run: List[Token]) -> List[_Region]:
|
| 165 |
+
"""A complete ayah typed as it is: found by exact word sequence, whatever the rarity of its words (so a short
|
| 166 |
+
ayah made of common words, like ``قل هو الله احد``, is not missed)."""
|
| 167 |
+
index, skeletons, found, i = self._whole_ayah_index(), [t.skeleton for t in run], [], 0
|
| 168 |
+
while i < len(run):
|
| 169 |
+
for words, ayah in index.get(skeletons[i], ()):
|
| 170 |
+
n = len(words)
|
| 171 |
+
if tuple(skeletons[i:i + n]) == words and not any(t.boundary_after for t in run[i:i + n - 1]):
|
| 172 |
+
record = self.kb.quran[ayah]
|
| 173 |
+
found.append(_Region(i, i + n, "Ayah", n, 1.0, 99.0, record["surah_id"], record["ayah_id"], record["ayah_id"]))
|
| 174 |
+
i += n - 1
|
| 175 |
+
break
|
| 176 |
+
i += 1
|
| 177 |
+
return found
|
| 178 |
+
|
| 179 |
+
def _grow(self, run: List[Token], ayah: int, chain: dict) -> Optional[_Region]:
|
| 180 |
+
kb = self.kb
|
| 181 |
+
source = kb.q_norm_match[ayah].split()
|
| 182 |
+
source_keys = [phonetic_key(w) for w in source]
|
| 183 |
+
diagonal = sorted(source_pos - pos for pos, source_pos in chain["hits"])[len(chain["hits"]) // 2]
|
| 184 |
+
start = min(pos for pos, _ in chain["hits"])
|
| 185 |
+
end = min(len(run), max(pos for pos, _ in chain["hits"]) + 3)
|
| 186 |
+
|
| 187 |
+
def key_at(i: int) -> Optional[str]:
|
| 188 |
+
return source_keys[i + diagonal] if 0 <= i + diagonal < len(source_keys) else None
|
| 189 |
+
|
| 190 |
+
while start > 0 and not run[start - 1].boundary_after: # grow left while the text keeps following the source
|
| 191 |
+
if run[start - 1].key == key_at(start - 1):
|
| 192 |
+
start -= 1
|
| 193 |
+
elif start >= 2 and not run[start - 2].boundary_after and run[start - 2].key == key_at(start - 2):
|
| 194 |
+
start -= 2 # one substituted word
|
| 195 |
+
else:
|
| 196 |
+
break
|
| 197 |
+
while end < len(run) and not run[end - 1].boundary_after:
|
| 198 |
+
if run[end].key == key_at(end):
|
| 199 |
+
end += 1
|
| 200 |
+
elif end + 1 < len(run) and run[end + 1].key == key_at(end + 1):
|
| 201 |
+
end += 2
|
| 202 |
+
else:
|
| 203 |
+
break
|
| 204 |
+
|
| 205 |
+
start = self._soft_left(run, start, diagonal, source_keys)
|
| 206 |
+
end = self._soft_right(run, end, diagonal, source_keys)
|
| 207 |
+
positions = range(start, end)
|
| 208 |
+
matched_idx = [i for i in positions if run[i].key == key_at(i)]
|
| 209 |
+
matched = len(matched_idx)
|
| 210 |
+
n = end - start
|
| 211 |
+
idf = sum(kb.quran_bm25.idf.get(run[i].skeleton, 0.0) for i in matched_idx)
|
| 212 |
+
ratio = matched / n if n else 0.0
|
| 213 |
+
needed_idf = self.min_idf if n >= 5 else self.min_idf + 6.0 # very short stretches must be rare phrases
|
| 214 |
+
if n < self.min_tokens or matched < self.min_tokens or ratio < self.min_ratio or idf < needed_idf:
|
| 215 |
+
return None
|
| 216 |
+
record = kb.quran[ayah]
|
| 217 |
+
return _Region(start, end, "Ayah", matched, ratio, idf, record["surah_id"], record["ayah_id"], record["ayah_id"])
|
| 218 |
+
|
| 219 |
+
@staticmethod
|
| 220 |
+
def _closest(run: List[Token], candidates, target: str):
|
| 221 |
+
"""Among candidate token ranges, the one whose joined phonetic key best resembles ``target``; near-ties prefer the
|
| 222 |
+
longer range (words next to a changed word usually belong to the same altered quotation)."""
|
| 223 |
+
scored = [(SequenceMatcher(None, "".join(t.key for t in run[a:b]), target).ratio(), a, b) for a, b in candidates]
|
| 224 |
+
if not scored:
|
| 225 |
+
return None
|
| 226 |
+
top = max(score for score, _, _ in scored)
|
| 227 |
+
if top < 0.6:
|
| 228 |
+
return None
|
| 229 |
+
return max((c for c in scored if c[0] >= top - 0.2), key=lambda c: c[2] - c[1])
|
| 230 |
+
|
| 231 |
+
def _soft_left(self, run: List[Token], start: int, diagonal: int, source_keys: List[str]) -> int:
|
| 232 |
+
"""Pull in up to two words before the region that look like the source words missing at its beginning."""
|
| 233 |
+
missing = min(start + diagonal, 2)
|
| 234 |
+
if missing <= 0:
|
| 235 |
+
return start
|
| 236 |
+
target = "".join(source_keys[start + diagonal - missing : start + diagonal])
|
| 237 |
+
candidates = [(start - n, start) for n in range(max(1, missing - 1), missing + 2)
|
| 238 |
+
if start - n >= 0 and not any(run[i].boundary_after for i in range(start - n, start))]
|
| 239 |
+
best = self._closest(run, candidates, target)
|
| 240 |
+
return best[1] if best else start
|
| 241 |
+
|
| 242 |
+
def _soft_right(self, run: List[Token], end: int, diagonal: int, source_keys: List[str]) -> int:
|
| 243 |
+
missing = min(len(source_keys) - (end + diagonal), 2)
|
| 244 |
+
if missing <= 0 or end >= len(run) or run[end - 1].boundary_after:
|
| 245 |
+
return end
|
| 246 |
+
target = "".join(source_keys[end + diagonal : end + diagonal + missing])
|
| 247 |
+
candidates = [(end, end + n) for n in range(max(1, missing - 1), missing + 2)
|
| 248 |
+
if end + n <= len(run) and not any(run[i].boundary_after for i in range(end, end + n - 1))]
|
| 249 |
+
best = self._closest(run, candidates, target)
|
| 250 |
+
return best[2] if best else end
|
| 251 |
+
|
| 252 |
+
@staticmethod
|
| 253 |
+
def _non_overlapping(regions: List[_Region]) -> List[_Region]:
|
| 254 |
+
chosen: List[_Region] = []
|
| 255 |
+
for region in sorted(regions, key=lambda r: (r.matched, r.ratio), reverse=True):
|
| 256 |
+
if all(region.end <= c.start or region.start >= c.end for c in chosen):
|
| 257 |
+
chosen.append(region)
|
| 258 |
+
return sorted(chosen, key=lambda r: r.start)
|
| 259 |
+
|
| 260 |
+
@staticmethod
|
| 261 |
+
def _merge(regions: List[_Region]) -> List[_Region]:
|
| 262 |
+
"""Join regions of consecutive ayahs that follow each other in the text (a quotation spanning several ayahs)."""
|
| 263 |
+
merged: List[_Region] = []
|
| 264 |
+
for region in regions:
|
| 265 |
+
last = merged[-1] if merged else None
|
| 266 |
+
if (last and last.surah == region.surah and region.first_ayah == last.last_ayah + 1
|
| 267 |
+
and region.start - last.end <= 1):
|
| 268 |
+
last.end, last.matched = region.end, last.matched + region.matched
|
| 269 |
+
last.ratio = last.matched / (last.end - last.start)
|
| 270 |
+
last.idf += region.idf
|
| 271 |
+
last.last_ayah = region.last_ayah
|
| 272 |
+
else:
|
| 273 |
+
merged.append(region)
|
| 274 |
+
return merged
|
| 275 |
+
|
| 276 |
+
# ---- Hadith -----------------------------------------------------------------------------------------------
|
| 277 |
+
def _scan_hadith(self, run: List[Token], taken: Sequence[Tuple[int, int]]) -> List[_Region]:
|
| 278 |
+
kb = self.kb
|
| 279 |
+
regions: List[_Region] = []
|
| 280 |
+
segment_start = 0
|
| 281 |
+
for i, token in enumerate(run):
|
| 282 |
+
if token.boundary_after or i == len(run) - 1:
|
| 283 |
+
segment = (segment_start, i + 1)
|
| 284 |
+
segment_start = i + 1
|
| 285 |
+
if segment[1] - segment[0] < self.hadith_min_tokens:
|
| 286 |
+
continue
|
| 287 |
+
for a, b in self._windows(*segment):
|
| 288 |
+
region = self._match_hadith(run, a, b)
|
| 289 |
+
if region is not None:
|
| 290 |
+
regions.append(region)
|
| 291 |
+
return self._non_overlapping(regions)
|
| 292 |
+
|
| 293 |
+
@staticmethod
|
| 294 |
+
def _windows(start: int, end: int, size: int = 24, stride: int = 12):
|
| 295 |
+
if end - start <= 40:
|
| 296 |
+
yield start, end
|
| 297 |
+
else:
|
| 298 |
+
for a in range(start, end - 6, stride):
|
| 299 |
+
yield a, min(end, a + size)
|
| 300 |
+
|
| 301 |
+
def _match_hadith(self, run: List[Token], a: int, b: int) -> Optional[_Region]:
|
| 302 |
+
kb = self.kb
|
| 303 |
+
window = run[a:b]
|
| 304 |
+
words = content_words(t.skeleton for t in window)
|
| 305 |
+
if len(words) < 4:
|
| 306 |
+
return None
|
| 307 |
+
window_keys = [t.key for t in window]
|
| 308 |
+
best = None
|
| 309 |
+
for idx in kb.hadith_candidates(words, 3):
|
| 310 |
+
record = kb.hadith[idx]
|
| 311 |
+
source = [p[1] for p in aligned_words(record["matn"] or record["full"])]
|
| 312 |
+
matcher = SequenceMatcher(None, window_keys, [phonetic_key(w) for w in source], autojunk=False)
|
| 313 |
+
blocks = [blk for blk in matcher.get_matching_blocks() if blk.size >= 3]
|
| 314 |
+
matched = sum(blk.size for blk in blocks)
|
| 315 |
+
if matched >= self.hadith_min_tokens and (best is None or matched > best[0]):
|
| 316 |
+
best = (matched, blocks)
|
| 317 |
+
if best is None:
|
| 318 |
+
return None
|
| 319 |
+
matched, blocks = best
|
| 320 |
+
start, end = blocks[0].a, blocks[-1].a + blocks[-1].size
|
| 321 |
+
ratio = matched / (end - start)
|
| 322 |
+
idf = sum(kb.hadith_data["bm25"].idf.get(window[i].skeleton, 0.0) for blk in blocks for i in range(blk.a, blk.a + blk.size))
|
| 323 |
+
if ratio < self.min_ratio or idf < self.hadith_min_idf:
|
| 324 |
+
return None
|
| 325 |
+
return _Region(a + start, a + end, "Hadith", matched, ratio, idf)
|
| 326 |
+
|
| 327 |
+
|
| 328 |
+
class HybridDetector:
|
| 329 |
+
"""Rule-based detection first (quotation marks / brackets, typed by the corpora; phrases are only a hint), then the corpus scanner on the rest of the
|
| 330 |
+
text to find unannounced and embedded quotations. Rule spans always win overlaps: an author-delimited quotation is
|
| 331 |
+
verified exactly as written."""
|
| 332 |
+
|
| 333 |
+
def __init__(self, retriever: SourceRetriever, use_scanner: bool = True, rules_use_corpus: bool = True,
|
| 334 |
+
decouple_triggers: bool = True) -> None:
|
| 335 |
+
"""``rules_use_corpus`` / ``decouple_triggers``: type delimited quotations from the corpora instead of from
|
| 336 |
+
introductory phrases (set both to False to reproduce the earlier trigger-driven behaviour for ablations)."""
|
| 337 |
+
self.rules = RuleDetector(retriever if rules_use_corpus else None, decouple_triggers=decouple_triggers)
|
| 338 |
+
self.scanner = CorpusScanner(retriever) if use_scanner else None
|
| 339 |
+
|
| 340 |
+
def detect(self, text: str) -> List[DetectedSpan]:
|
| 341 |
+
spans = self.rules.detect(text)
|
| 342 |
+
if self.scanner is not None:
|
| 343 |
+
found = sorted(self.scanner.scan(text, [(s.start, s.end) for s in spans]), key=lambda s: s.end - s.start, reverse=True)
|
| 344 |
+
kept: List[DetectedSpan] = []
|
| 345 |
+
for span in found: # the same words can follow both corpora: keep the longer span, never report one text twice
|
| 346 |
+
if all(span.end <= k.start or span.start >= k.end for k in kept):
|
| 347 |
+
kept.append(span)
|
| 348 |
+
spans += kept
|
| 349 |
+
return sorted(spans, key=lambda s: s.start)
|
similarity.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Similarity indicators between a quotation and a candidate source (the signals behind the verification score)."""
|
| 2 |
+
from __future__ import annotations
|
| 3 |
+
|
| 4 |
+
import re
|
| 5 |
+
import unicodedata
|
| 6 |
+
from typing import Dict, List
|
| 7 |
+
|
| 8 |
+
from alignment import best_region, lcs_length
|
| 9 |
+
from normalization import normalize_lenient, normalize_strict, tokenize
|
| 10 |
+
|
| 11 |
+
try: # optional accelerator; identical formula (1 - distance / max_len)
|
| 12 |
+
from rapidfuzz.distance import Levenshtein as _RapidLevenshtein
|
| 13 |
+
except ImportError: # pragma: no cover
|
| 14 |
+
_RapidLevenshtein = None
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
def _edit_similarity(a: str, b: str, max_len: int = 600) -> float:
|
| 18 |
+
"""Normalised Levenshtein similarity in [0, 1] on the first ``max_len`` characters."""
|
| 19 |
+
a, b = a[:max_len], b[:max_len]
|
| 20 |
+
if a == b:
|
| 21 |
+
return 1.0
|
| 22 |
+
if not a or not b:
|
| 23 |
+
return 0.0
|
| 24 |
+
if _RapidLevenshtein is not None:
|
| 25 |
+
return float(_RapidLevenshtein.normalized_similarity(a, b))
|
| 26 |
+
prev = list(range(len(b) + 1))
|
| 27 |
+
for i, ca in enumerate(a):
|
| 28 |
+
curr = [i + 1]
|
| 29 |
+
for j, cb in enumerate(b):
|
| 30 |
+
curr.append(min(curr[j] + 1, prev[j + 1] + 1, prev[j] + (ca != cb)))
|
| 31 |
+
prev = curr
|
| 32 |
+
return 1.0 - prev[-1] / max(len(a), len(b))
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
def _light_normalize(text: str) -> str:
|
| 36 |
+
"""Keeps diacritics (so diacritic changes lower the score) but drops punctuation and tatweel."""
|
| 37 |
+
text = unicodedata.normalize("NFC", text)
|
| 38 |
+
text = re.sub(r"[،؛؟!.,:;'\"()\[\]{}<>«»\-_/\\|@#$%^&*+=~`\u0640]", " ", text)
|
| 39 |
+
return re.sub(r"\s+", " ", text).strip()
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
def compute_signals(claim: str, candidate: str, content_type: str = "Ayah", local: bool = True) -> Dict[str, float]:
|
| 43 |
+
"""Word-, character- and sequence-level similarity indicators between a quotation and a candidate source."""
|
| 44 |
+
is_quran = content_type == "Ayah"
|
| 45 |
+
full_candidate = candidate
|
| 46 |
+
if local: # score the snippet against the region it refers to, not against the whole verse / Hadith
|
| 47 |
+
candidate = best_region(claim, candidate)
|
| 48 |
+
normalize = normalize_strict if is_quran else normalize_lenient
|
| 49 |
+
norm_claim, norm_cand = normalize(claim), normalize(candidate)
|
| 50 |
+
|
| 51 |
+
claim_set, cand_set = set(tokenize(norm_claim)), set(tokenize(norm_cand))
|
| 52 |
+
if claim_set and cand_set:
|
| 53 |
+
shared = claim_set & cand_set
|
| 54 |
+
token_overlap = len(shared) / len(claim_set | cand_set)
|
| 55 |
+
coverage = len(shared) / len(claim_set)
|
| 56 |
+
else:
|
| 57 |
+
token_overlap = coverage = 0.0
|
| 58 |
+
|
| 59 |
+
claim_tokens, cand_tokens = tokenize(norm_claim), tokenize(norm_cand)
|
| 60 |
+
lcs_ratio = lcs_length(claim_tokens, cand_tokens) / len(claim_tokens) if claim_tokens else 0.0
|
| 61 |
+
edit_sim = _edit_similarity(norm_claim, norm_cand)
|
| 62 |
+
|
| 63 |
+
claim_chars, cand_chars = set(norm_claim.replace(" ", "")), set(norm_cand.replace(" ", ""))
|
| 64 |
+
char_overlap = len(claim_chars & cand_chars) / len(claim_chars | cand_chars) if (claim_chars or cand_chars) else 0.0
|
| 65 |
+
|
| 66 |
+
claim_flat, cand_flat = norm_claim.replace(" ", ""), norm_cand.replace(" ", "")
|
| 67 |
+
is_substring = int(bool(claim_flat) and bool(cand_flat) and (claim_flat in cand_flat or cand_flat in claim_flat))
|
| 68 |
+
|
| 69 |
+
diacritic_sim = (
|
| 70 |
+
_edit_similarity(_light_normalize(claim), _light_normalize(candidate), max_len=800) if is_quran else edit_sim
|
| 71 |
+
)
|
| 72 |
+
short = len(claim_tokens) < 4
|
| 73 |
+
|
| 74 |
+
if is_quran:
|
| 75 |
+
w = (
|
| 76 |
+
dict(coverage=0.35, diacritic_sim=0.30, lcs_ratio=0.15, token_overlap=0.10, char_overlap=0.05, edit_sim=0.05)
|
| 77 |
+
if short
|
| 78 |
+
else dict(coverage=0.25, diacritic_sim=0.30, lcs_ratio=0.20, token_overlap=0.10, char_overlap=0.05, edit_sim=0.10)
|
| 79 |
+
)
|
| 80 |
+
composite = (
|
| 81 |
+
w["coverage"] * coverage + w["diacritic_sim"] * diacritic_sim + w["lcs_ratio"] * lcs_ratio
|
| 82 |
+
+ w["token_overlap"] * token_overlap + w["char_overlap"] * char_overlap + w["edit_sim"] * edit_sim
|
| 83 |
+
)
|
| 84 |
+
if is_substring and diacritic_sim >= 0.60:
|
| 85 |
+
composite = max(composite, 0.88)
|
| 86 |
+
elif is_substring:
|
| 87 |
+
composite = max(composite, 0.75)
|
| 88 |
+
else:
|
| 89 |
+
w = (
|
| 90 |
+
dict(coverage=0.40, lcs_ratio=0.20, token_overlap=0.20, char_overlap=0.10, edit_sim=0.10)
|
| 91 |
+
if short
|
| 92 |
+
else dict(coverage=0.30, lcs_ratio=0.28, token_overlap=0.18, char_overlap=0.12, edit_sim=0.12)
|
| 93 |
+
)
|
| 94 |
+
composite = (
|
| 95 |
+
w["coverage"] * coverage + w["lcs_ratio"] * lcs_ratio + w["token_overlap"] * token_overlap
|
| 96 |
+
+ w["char_overlap"] * char_overlap + w["edit_sim"] * edit_sim
|
| 97 |
+
)
|
| 98 |
+
if is_substring:
|
| 99 |
+
composite = max(composite, 0.82)
|
| 100 |
+
|
| 101 |
+
return {
|
| 102 |
+
"token_overlap": round(token_overlap, 4),
|
| 103 |
+
"coverage": round(coverage, 4),
|
| 104 |
+
"lcs_ratio": round(lcs_ratio, 4),
|
| 105 |
+
"edit_sim": round(edit_sim, 4),
|
| 106 |
+
"diacritic_sim": round(diacritic_sim, 4),
|
| 107 |
+
"char_overlap": round(char_overlap, 4),
|
| 108 |
+
"is_substring": is_substring,
|
| 109 |
+
"source_fraction": round(min(1.0, len(claim_tokens) / max(len(tokenize(normalize(full_candidate))), 1)), 4),
|
| 110 |
+
"composite": round(composite, 4),
|
| 111 |
+
}
|
| 112 |
+
|
| 113 |
+
|
| 114 |
+
def best_match_score(claim: str, candidates: List[dict], content_type: str = "Ayah", local: bool = True):
|
| 115 |
+
"""Return ``(score, candidate_with_signals)`` for the best-scoring candidate."""
|
| 116 |
+
best_score, best_candidate = 0.0, None
|
| 117 |
+
for candidate in candidates:
|
| 118 |
+
signals = compute_signals(claim, candidate.get("text", ""), content_type, local)
|
| 119 |
+
if signals["composite"] > best_score:
|
| 120 |
+
best_score, best_candidate = signals["composite"], {**candidate, "signals": signals}
|
| 121 |
+
return best_score, best_candidate
|
ui.py
ADDED
|
@@ -0,0 +1,534 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Arabic-only presentation layer shared by the Gradio app and the static (in-browser) page.
|
| 2 |
+
|
| 3 |
+
Colour rules: green = verified, red = a real mismatch or a missing reference, amber = needs a human. Verified items
|
| 4 |
+
never use red. Every user-facing string is Arabic.
|
| 5 |
+
"""
|
| 6 |
+
from __future__ import annotations
|
| 7 |
+
|
| 8 |
+
import html
|
| 9 |
+
from typing import List, Optional
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
_e = html.escape
|
| 13 |
+
|
| 14 |
+
APP_TITLE = "التحقق من هلوسة القرآن والحديث وتصحيحها"
|
| 15 |
+
APP_TAGLINE = "تحقّق من آيات القرآن والأحاديث النبوية بالدليل"
|
| 16 |
+
COPIED_MESSAGE = "تم نسخ النص بنجاح"
|
| 17 |
+
|
| 18 |
+
GROUP_LABEL = {"verified": "موثّق", "mismatch": "غير مطابق", "review": "تحتاج مراجعة بشرية"}
|
| 19 |
+
TYPE_LABEL = {"Ayah": "آية قرآنية", "Hadith": "حديث نبوي"}
|
| 20 |
+
INDICATORS = [
|
| 21 |
+
("composite", "الدرجة المركبة"),
|
| 22 |
+
("coverage", "تغطية الكلمات"),
|
| 23 |
+
("lcs_ratio", "التسلسل النصي"),
|
| 24 |
+
("token_overlap", "تداخل الكلمات"),
|
| 25 |
+
("edit_sim", "تشابه الحروف"),
|
| 26 |
+
("diacritic_sim", "التشكيل"),
|
| 27 |
+
]
|
| 28 |
+
METHODS = {
|
| 29 |
+
"substring_match": "الاقتباس واردٌ كاملًا في المصدر",
|
| 30 |
+
"threshold_pass": "تجاوز عتبة التطابق",
|
| 31 |
+
"threshold_fail": "دون عتبة التطابق",
|
| 32 |
+
"borderline_multi_cov": "حالة حدّية: عدة مصادر متقاربة",
|
| 33 |
+
"borderline_default": "حالة حدّية",
|
| 34 |
+
"borderline_low_retrieval": "حالة حدّية: استرجاع ضعيف",
|
| 35 |
+
"no_candidates": "لا توجد مصادر مرشحة",
|
| 36 |
+
"empty_span": "نص فارغ",
|
| 37 |
+
"error": "تعذّرت المعالجة",
|
| 38 |
+
}
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
def _pct(value: float) -> str:
|
| 42 |
+
return f"{round(value * 100)}%"
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
def source_label(source: dict) -> str:
|
| 46 |
+
if source["type"] == "Quran":
|
| 47 |
+
start, end = source["ayah_start"], source["ayah_end"]
|
| 48 |
+
verses = f"{start}" if start == end else f"{start}–{end}"
|
| 49 |
+
return f"سورة {source['surah_name']} — الآية {verses}"
|
| 50 |
+
return f"حديث رقم {source['hadithID']} — {source['title']}"
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
def reason_text(reason: dict) -> str:
|
| 54 |
+
"""Arabic explanation of a decision code produced by the pipeline."""
|
| 55 |
+
code = reason["code"]
|
| 56 |
+
src = reason.get("source", "")
|
| 57 |
+
if code == "exact_match":
|
| 58 |
+
return f"تطابق تام مع {src} بعد تجاهل اختلافات الرسم والتشكيل."
|
| 59 |
+
if code == "close_match":
|
| 60 |
+
return f"تطابق شبه كامل مع {src}."
|
| 61 |
+
if code == "altered_passage":
|
| 62 |
+
n = reason.get("n", 0)
|
| 63 |
+
how = "تغيّر ترتيب الكلمات" if reason.get("reordered") else f"{n} موضع مختلف"
|
| 64 |
+
return f"النص يخالف {src} ({how}). التصحيح المقترح هو نص المصدر حرفيًا ولم يُولَّد."
|
| 65 |
+
if code == "weak_match":
|
| 66 |
+
return "وُجد مصدر قريب لكن التطابق غير كافٍ للحكم الآلي."
|
| 67 |
+
if code == "too_short":
|
| 68 |
+
return "الاقتباس قصير جدًا (كلمتان فقط) فلا يمكن تحديد مصدره بيقين؛ يلزم الرجوع إلى مختص."
|
| 69 |
+
if code == "no_source":
|
| 70 |
+
return "لا يوجد في المراجع المضمّنة نصٌّ يشبه هذا الاقتباس، وقد يكون مختلَقًا."
|
| 71 |
+
if code == "insufficient_evidence":
|
| 72 |
+
return "الأدلة غير كافية: لا مصدر واضح، ودرجة اليقين منخفضة."
|
| 73 |
+
if code == "candidate_not_strong":
|
| 74 |
+
return f"وُجد مرشح ({src}، قوة المطابقة {round(reason.get('strength', 0) * 100)}%) لكن الدليل لا يكفي للتصحيح الآلي."
|
| 75 |
+
if code == "hadith_altered":
|
| 76 |
+
return (f"النص قريب جدًا من {src} لكن يختلف عنه في {reason.get('n', 1)} موضع (زيادة أو نقص أو استبدال كلمة)، وقد يغيّر ذلك المعنى؛ "
|
| 77 |
+
"لا يُوثَّق الحديث إلا إذا طابق النص حرفيًا، ولا يُصحَّح آليًا فتلزم مراجعة مختص.")
|
| 78 |
+
if code == "hadith_candidate":
|
| 79 |
+
return f"وُجدت رواية مشابهة ({src}) لكن لا يُصحَّح الحديث آليًا لاختلاف الروايات؛ يلزم الرجوع إلى مختص."
|
| 80 |
+
return "تعذّرت معالجة هذا الاقتباس آليًا."
|
| 81 |
+
|
| 82 |
+
|
| 83 |
+
def note_text(note: dict) -> str:
|
| 84 |
+
n = note.get("n", 0)
|
| 85 |
+
code = note["code"]
|
| 86 |
+
if code == "diacritic_conflict":
|
| 87 |
+
return f"تنبيه: تشكيل {n} كلمة يخالف المصحف (الكلمات صحيحة لكن الحركات مختلفة)."
|
| 88 |
+
if code == "orthographic_variant":
|
| 89 |
+
return f"ملاحظة: {n} اختلاف إملائي (رسم أو مسافات) لا يغيّر الكلمة."
|
| 90 |
+
if note.get("misattributed"):
|
| 91 |
+
if code == "is_ayah":
|
| 92 |
+
return f"تنبيه: النص قرآني ({note['source']}) لكن عبارة التقديم تنسبه إلى الحديث."
|
| 93 |
+
return f"تنبيه: النص حديث ({note['source']}) لكن عبارة التقديم تنسبه إلى القرآن."
|
| 94 |
+
if code == "is_hadith":
|
| 95 |
+
return f"تنبيه: هذا النص وارد في الحديث ({note['source']}) وليس في القرآن؛ ربما نُسب إلى الله تعالى خطأً."
|
| 96 |
+
if code == "is_ayah":
|
| 97 |
+
return f"تنبيه: هذا النص وارد في القرآن ({note['source']}) وليس في الحديث؛ ربما نُسب إلى النبي ﷺ خطأً."
|
| 98 |
+
if code == "hadith_minor_diffs":
|
| 99 |
+
return f"ملاحظة: {n} اختلاف طفيف عن أقرب رواية في المراجع."
|
| 100 |
+
return ""
|
| 101 |
+
|
| 102 |
+
|
| 103 |
+
def _diff_html(comparison: dict) -> str:
|
| 104 |
+
"""Red = words only in the quotation; green = the source words that should be there."""
|
| 105 |
+
parts = []
|
| 106 |
+
for op in comparison["word_diff"]:
|
| 107 |
+
if op["op"] == "equal":
|
| 108 |
+
parts.append(f'<span class="w-eq">{_e(op["span"])}</span>')
|
| 109 |
+
continue
|
| 110 |
+
if op["span"]:
|
| 111 |
+
parts.append(f'<span class="w-extra">{_e(op["span"])}</span>')
|
| 112 |
+
if op["source"]:
|
| 113 |
+
parts.append(f'<span class="w-missing">{_e(op["source"])}</span>')
|
| 114 |
+
return " ".join(parts)
|
| 115 |
+
|
| 116 |
+
|
| 117 |
+
def _legend(group: str, comparison: Optional[dict]) -> str:
|
| 118 |
+
if group == "verified" or not comparison or all(op["op"] == "equal" for op in comparison["word_diff"]):
|
| 119 |
+
return ""
|
| 120 |
+
return '<p class="legend"><span class="w-extra">كلمات في الاقتباس تخالف المصدر</span> <span class="w-missing">الصواب من المصدر</span></p>'
|
| 121 |
+
|
| 122 |
+
|
| 123 |
+
def _indicator_table(signals: Optional[dict]) -> str:
|
| 124 |
+
if not signals:
|
| 125 |
+
return ""
|
| 126 |
+
rows = []
|
| 127 |
+
for key, label in INDICATORS:
|
| 128 |
+
value = float(signals.get(key, 0.0))
|
| 129 |
+
rows.append(
|
| 130 |
+
f'<div class="ind"><div class="ind-name">{label}</div>'
|
| 131 |
+
f'<div class="bar"><span style="width:{_pct(value)}"></span></div><div class="ind-val">{_pct(value)}</div></div>'
|
| 132 |
+
)
|
| 133 |
+
rows.append(f'<div class="ind"><div class="ind-name">وروده كاملًا في المصدر</div><div class="ind-val wide">{"نعم" if signals.get("is_substring") else "لا"}</div></div>')
|
| 134 |
+
return '<div class="indicators">' + "".join(rows) + "</div>"
|
| 135 |
+
|
| 136 |
+
|
| 137 |
+
def _match_percent(span: dict) -> int:
|
| 138 |
+
evidence = span["evidence"]
|
| 139 |
+
if evidence and evidence.get("comparison"):
|
| 140 |
+
return round(evidence["comparison"]["word_similarity"] * 100)
|
| 141 |
+
if evidence and evidence.get("signals"):
|
| 142 |
+
return round(evidence["signals"]["composite"] * 100)
|
| 143 |
+
return 0
|
| 144 |
+
|
| 145 |
+
|
| 146 |
+
def _evidence_panel(span: dict) -> str:
|
| 147 |
+
evidence, verification = span["evidence"], span["verification"]
|
| 148 |
+
rows = [
|
| 149 |
+
f'<div class="ev-row"><b>الاقتباس المكتشف</b><p class="quote">{_e(span["text"])}</p></div>',
|
| 150 |
+
f'<div class="ev-row"><b>النوع</b><p>{TYPE_LABEL[span["type"]]}</p></div>',
|
| 151 |
+
]
|
| 152 |
+
if evidence:
|
| 153 |
+
comparison = evidence["comparison"]
|
| 154 |
+
rows += [
|
| 155 |
+
f'<div class="ev-row"><b>المصدر المرشح</b><p>{_e(source_label(evidence["source"]))}</p></div>',
|
| 156 |
+
f'<div class="ev-row"><b>نص المصدر الأصلي</b><p class="quote">{_e(comparison["source_excerpt"])}</p></div>',
|
| 157 |
+
f'<div class="ev-row"><b>المقارنة كلمةً بكلمة</b><p class="diff">{_diff_html(comparison)}</p>{_legend(span["group"], comparison)}</div>',
|
| 158 |
+
f'<div class="ev-row"><b>مؤشرات التحقق</b>{_indicator_table(evidence.get("signals"))}</div>',
|
| 159 |
+
]
|
| 160 |
+
if comparison["diacritic_notes"]:
|
| 161 |
+
items = "، ".join(f"{_e(n['word'])} ← {_e(n['source_word'])}" for n in comparison["diacritic_notes"])
|
| 162 |
+
rows.append(f'<div class="ev-row"><b>اختلافات التشكيل</b><p>{items}</p></div>')
|
| 163 |
+
else:
|
| 164 |
+
rows.append('<div class="ev-row"><b>المصدر المرشح</b><p>لم يُسترجع أي مصدر.</p></div>')
|
| 165 |
+
rows += [
|
| 166 |
+
f'<div class="ev-row"><b>درجة اليقين في الحكم</b><p>{_pct(verification["confidence"])} · {METHODS.get(verification["method"], "")}</p></div>',
|
| 167 |
+
f'<div class="ev-row"><b>القرار</b><p>{_e(span["status_ar"])}</p></div>',
|
| 168 |
+
f'<div class="ev-row"><b>سبب القرار</b><p>{_e(reason_text(span["reason"]))}</p></div>',
|
| 169 |
+
]
|
| 170 |
+
return "".join(rows)
|
| 171 |
+
|
| 172 |
+
|
| 173 |
+
def _card(span: dict) -> str:
|
| 174 |
+
group, evidence = span["group"], span["evidence"]
|
| 175 |
+
label = GROUP_LABEL[group]
|
| 176 |
+
chip = f'<span class="conf">نسبة المطابقة <b>{_match_percent(span)}%</b></span>'
|
| 177 |
+
scan_badge = ('<span class="badge scan" title="اكتُشف بمطابقة النص مع المراجع دون علامات تنصيص">اقتباس غير معلَن</span>'
|
| 178 |
+
if span["detection"]["backend"] == "scan" else "")
|
| 179 |
+
|
| 180 |
+
fields = [f'<div class="field"><label>الاقتباس المكتشف</label><p class="quote">{_e(span["text"])}</p></div>']
|
| 181 |
+
if evidence:
|
| 182 |
+
comparison = evidence["comparison"]
|
| 183 |
+
fields.append(f'<div class="field"><label>المصدر المرشح</label><p>{_e(source_label(evidence["source"]))}</p></div>')
|
| 184 |
+
if group != "verified": # a verified quotation equals its source: no need to print it twice
|
| 185 |
+
fields.append(
|
| 186 |
+
f'<div class="field"><label>المقارنة بالمصدر</label><p class="diff">{_diff_html(comparison)}</p>{_legend(group, comparison)}</div>'
|
| 187 |
+
)
|
| 188 |
+
else:
|
| 189 |
+
fields.append('<div class="field"><label>المصدر المرشح</label><p>لا يوجد</p></div>')
|
| 190 |
+
fields.append(f'<p class="reason">{_e(reason_text(span["reason"]))}</p>')
|
| 191 |
+
|
| 192 |
+
notes = "".join(f'<p class="note">{_e(note_text(n))}</p>' for n in span["notes"] if note_text(n))
|
| 193 |
+
action = ""
|
| 194 |
+
correction, suggestion = span["correction"], span["suggestion"]
|
| 195 |
+
if correction:
|
| 196 |
+
action = (
|
| 197 |
+
'<div class="action ok"><b>التصحيح المقترح من المصدر</b>'
|
| 198 |
+
f'<p class="quote">{_e(correction["display_text"])}</p>'
|
| 199 |
+
f'<p class="legend">{_e(source_label(correction["source"]))}</p></div>'
|
| 200 |
+
)
|
| 201 |
+
elif group == "review":
|
| 202 |
+
closest = ""
|
| 203 |
+
if suggestion:
|
| 204 |
+
closest = (f'<p class="legend">أقرب مصدر وُجد للمراجِع، وليس تصحيحًا آليًا: {_e(source_label(suggestion["source"]))}</p>'
|
| 205 |
+
f'<p class="quote small">{_e(suggestion["display_text"][:420])}</p>')
|
| 206 |
+
action = ('<div class="action warn"><b>الأدلة غير كافية للتصحيح الآلي.</b> يُوصى بالمراجعة البشرية.' + closest + "</div>")
|
| 207 |
+
elif span["status"] == "UNSUPPORTED":
|
| 208 |
+
action = '<div class="action bad"><b>لا يوجد مصدر مطابق في المراجع المتاحة.</b> لم يُقترح أي نص بديل.</div>'
|
| 209 |
+
|
| 210 |
+
return f"""
|
| 211 |
+
<div class="qcard {group}">
|
| 212 |
+
<div class="qhead">
|
| 213 |
+
<span class="idx">{span["id"]}</span>
|
| 214 |
+
<span class="badge type">{TYPE_LABEL[span["type"]]}</span>{scan_badge}
|
| 215 |
+
<span class="badge st {group}">{label}</span>
|
| 216 |
+
{chip}
|
| 217 |
+
</div>
|
| 218 |
+
{"".join(fields)}
|
| 219 |
+
{notes}
|
| 220 |
+
{action}
|
| 221 |
+
<details class="evidence"><summary>عرض الدليل</summary><div class="ev-body">{_evidence_panel(span)}</div></details>
|
| 222 |
+
</div>"""
|
| 223 |
+
|
| 224 |
+
|
| 225 |
+
def _summary(summary: dict) -> str:
|
| 226 |
+
tiles = [
|
| 227 |
+
(summary["n_spans"], "إجمالي الاقتباسات", ""),
|
| 228 |
+
(summary["n_ayah"], "آيات قرآنية", ""),
|
| 229 |
+
(summary["n_hadith"], "أحاديث نبوية", ""),
|
| 230 |
+
(summary["VERIFIED"], "موثّق", "verified"),
|
| 231 |
+
(summary["CORRECTED"] + summary["UNSUPPORTED"], "غير مطابق", "mismatch"),
|
| 232 |
+
(summary["HUMAN_REVIEW"], "تحتاج مراجعة", "review"),
|
| 233 |
+
]
|
| 234 |
+
return '<div class="summary">' + "".join(
|
| 235 |
+
f'<div class="tile {cls}{" zero" if value == 0 and cls else ""}"><b>{value}</b><span>{label}</span></div>'
|
| 236 |
+
for value, label, cls in tiles
|
| 237 |
+
) + "</div>"
|
| 238 |
+
|
| 239 |
+
|
| 240 |
+
def _highlighted_text(result: dict, title: str) -> str:
|
| 241 |
+
text, pieces, cursor = result["input_text"], [], 0
|
| 242 |
+
for span in result["spans"]:
|
| 243 |
+
pieces.append(_e(text[cursor:span["start"]]))
|
| 244 |
+
pieces.append(f'<mark class="{span["group"]}">{_e(text[span["start"]:span["end"]])}</mark>')
|
| 245 |
+
cursor = span["end"]
|
| 246 |
+
pieces.append(_e(text[cursor:]))
|
| 247 |
+
legend = ('<div class="hl-legend"><mark class="verified">موثّق</mark><mark class="mismatch">غير مطابق</mark>'
|
| 248 |
+
'<mark class="review">مراجعة بشرية</mark></div>')
|
| 249 |
+
return f'<div class="highlight"><label>{title}</label><p>' + "".join(pieces) + "</p>" + legend + "</div>"
|
| 250 |
+
|
| 251 |
+
|
| 252 |
+
def final_text(result: dict) -> str:
|
| 253 |
+
"""Corrected text with an inline Arabic flag after every quotation that still needs attention."""
|
| 254 |
+
text, reports = result["input_text"], result["spans"]
|
| 255 |
+
for report in sorted(reports, key=lambda r: r["start"], reverse=True):
|
| 256 |
+
if report["status"] == "CORRECTED" and report["correction"]:
|
| 257 |
+
text = text[: report["start"]] + report["correction"]["display_text"] + text[report["end"]:]
|
| 258 |
+
elif report["status"] == "UNSUPPORTED":
|
| 259 |
+
text = text[: report["end"]] + " [⚠ لا يوجد مصدر مطابق]" + text[report["end"]:]
|
| 260 |
+
elif report["status"] == "HUMAN_REVIEW":
|
| 261 |
+
text = text[: report["end"]] + " [⚠ يحتاج مراجعة بشرية]" + text[report["end"]:]
|
| 262 |
+
return text
|
| 263 |
+
|
| 264 |
+
|
| 265 |
+
def _final_block(result: dict) -> str:
|
| 266 |
+
summary = result["summary"]
|
| 267 |
+
fixed = summary["CORRECTED"]
|
| 268 |
+
open_items = summary["UNSUPPORTED"] + summary["HUMAN_REVIEW"]
|
| 269 |
+
if fixed == 0 and open_items == 0:
|
| 270 |
+
headline = "كل الاقتباسات المكتشفة مطابقة للمصادر."
|
| 271 |
+
else:
|
| 272 |
+
headline = f"صُحِّح {fixed} اقتباس من نص المصدر، وبقي {open_items} اقتباس يحتاج مراجعة بشرية ومعلَّم بعلامة تنبيه."
|
| 273 |
+
text = final_text(result)
|
| 274 |
+
return (
|
| 275 |
+
'<div class="final"><div class="final-head"><b>النسخة المصحّحة</b>'
|
| 276 |
+
f'<button class="copy" type="button" data-text="{_e(text, quote=True)}" '
|
| 277 |
+
'onclick="window.icvCopy&&window.icvCopy(this)">نسخ النص</button></div>'
|
| 278 |
+
f'<p class="legend">{headline}</p><p class="final-text">{_e(text)}</p></div>'
|
| 279 |
+
)
|
| 280 |
+
|
| 281 |
+
|
| 282 |
+
def render_results(result: dict, generated_answer: Optional[str] = None) -> str:
|
| 283 |
+
"""HTML report for a pipeline result. Pass the model's answer (mode B) to show it first and label the text as generated."""
|
| 284 |
+
header = render_generated_header(generated_answer) if generated_answer is not None else ""
|
| 285 |
+
if not result["spans"]:
|
| 286 |
+
body = (
|
| 287 |
+
'<div class="notice">لم يُعثر على اقتباسات قرآنية أو حديثية في هذا النص. يتعرّف النظام على الاقتباسات بمطابقتها مع المراجع، '
|
| 288 |
+
'سواء وُضعت بين علامات تنصيص أو أقواس أو وردت داخل الكلام دون أي عبارة تمهيدية.</div>'
|
| 289 |
+
)
|
| 290 |
+
return f'<div class="icv">{header}{body}</div>'
|
| 291 |
+
generated = generated_answer is not None
|
| 292 |
+
title = "إجابة النموذج مع الاقتباسات المكتشفة" if generated else "النص مع الاقتباسات المكتشفة"
|
| 293 |
+
parts = [_summary(result["summary"]), _highlighted_text(result, title), "".join(_card(s) for s in result["spans"])]
|
| 294 |
+
summary = result["summary"]
|
| 295 |
+
if generated or summary["CORRECTED"] or summary["UNSUPPORTED"] or summary["HUMAN_REVIEW"]:
|
| 296 |
+
parts.append(_final_block(result))
|
| 297 |
+
return f'<div class="icv">{header}{"".join(parts)}</div>'
|
| 298 |
+
|
| 299 |
+
|
| 300 |
+
def render_generated_header(answer: str) -> str:
|
| 301 |
+
return f'<div class="generated"><label>إجابة النموذج اللغوي كما وصلت</label><p>{_e(answer)}</p></div>'
|
| 302 |
+
|
| 303 |
+
|
| 304 |
+
def render_message(message: str, kind: str = "info") -> str:
|
| 305 |
+
return f'<div class="icv"><div class="notice {kind}">{_e(message)}</div></div>'
|
| 306 |
+
|
| 307 |
+
|
| 308 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 309 |
+
# Static markup and CSS
|
| 310 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 311 |
+
STAR = (
|
| 312 |
+
'<svg class="mark" viewBox="0 0 64 64" fill="none" stroke="#E2C06E" stroke-width="1.8" aria-hidden="true">'
|
| 313 |
+
'<rect x="14" y="14" width="36" height="36"/><rect x="14" y="14" width="36" height="36" transform="rotate(45 32 32)"/>'
|
| 314 |
+
'<circle cx="32" cy="32" r="7" fill="#E2C06E" stroke="none"/></svg>'
|
| 315 |
+
)
|
| 316 |
+
|
| 317 |
+
HERO = f"""
|
| 318 |
+
<div class="icv"><div class="hero">
|
| 319 |
+
{STAR}
|
| 320 |
+
<h1>{APP_TITLE}</h1>
|
| 321 |
+
<p class="tagline">{APP_TAGLINE}</p>
|
| 322 |
+
<p class="sub">يكتشف النظام الاقتباسات داخل أي نص، ويسترجع نصوصها من المصادر، ويقارنها كلمةً بكلمة، ثم يعرض الدليل.
|
| 323 |
+
وإذا لم تكفِ الأدلة فلن يختلق تصحيحًا، بل يحيل الحالة إلى المراجعة البشرية.</p>
|
| 324 |
+
<div class="flow"><span>كشف</span><i>‹</i><span>استرجاع</span><i>‹</i><span>محاذاة</span><i>‹</i><span>دليل</span><i>‹</i><span>قرار</span></div>
|
| 325 |
+
</div></div>
|
| 326 |
+
"""
|
| 327 |
+
|
| 328 |
+
DISCLAIMER = """<div class="icv"><div class="disclaimer">أداة مساعدة للتدقيق النصي وليست فتوى ولا بديلًا عن المراجعة المتخصصة.
|
| 329 |
+
النتائج مبنية على مراجع القرآن الكريم والكتب الستة المضمّنة فقط.</div></div>"""
|
| 330 |
+
|
| 331 |
+
PLACEHOLDER = "الصق هنا النص الذي ولّده نموذج لغوي. يكتشف النظام الآيات والأحاديث الواردة فيه، بعلامات تنصيص أو بدونها…"
|
| 332 |
+
PROMPT_PLACEHOLDER = "اكتب سؤالك، مثل: اشرح لي فضل الصبر في القرآن والسنة مع ذكر الأدلة."
|
| 333 |
+
|
| 334 |
+
_PATTERN = (
|
| 335 |
+
"url(\"data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' width='88' height='88' viewBox='0 0 88 88'%3E"
|
| 336 |
+
"%3Cg fill='none' stroke='%230F4C3A' stroke-opacity='0.07' stroke-width='1'%3E"
|
| 337 |
+
"%3Crect x='22' y='22' width='44' height='44'/%3E%3Crect x='22' y='22' width='44' height='44' transform='rotate(45 44 44)'/%3E"
|
| 338 |
+
"%3C/g%3E%3C/svg%3E\")"
|
| 339 |
+
)
|
| 340 |
+
|
| 341 |
+
CSS = """
|
| 342 |
+
@import url('https://fonts.googleapis.com/css2?family=Cairo:wght@400;600;700&family=Amiri:wght@400;700&display=swap');
|
| 343 |
+
:root { --green:#0F4C3A; --green-2:#17694F; --gold:#B8912F; --gold-2:#E2C06E; --cream:#FBF6EA; --paper:#FFFFFF;
|
| 344 |
+
--ink:#1F2933; --muted:#55626D; --line:#E6DCC3; --ok:#1B7A4B; --ok-bg:#EAF6EF; --ok-line:#BFE3CD;
|
| 345 |
+
--bad:#B3261E; --bad-bg:#FDECEA; --bad-line:#F4B8B3; --warn:#7A5B0C; --warn-bg:#FFF6DA; --warn-line:#EBD28A; }
|
| 346 |
+
.gradio-container, body.icv-page { background:var(--cream) PATTERN !important; font-family:'Cairo','Segoe UI',Tahoma,sans-serif; color:var(--ink); }
|
| 347 |
+
.gradio-container { max-width:1060px !important; --body-text-color:#1F2933; --block-background-fill:#FFFFFF; --block-border-color:#E6DCC3;
|
| 348 |
+
--input-background-fill:#FFFFFF; --block-label-text-color:#55626D; --button-primary-background-fill:#0F4C3A;
|
| 349 |
+
--button-primary-background-fill-hover:#17694F; --button-primary-text-color:#FFFFFF; --button-secondary-background-fill:#FFFFFF;
|
| 350 |
+
--button-secondary-text-color:#0F4C3A; --button-secondary-border-color:#B8912F; --color-accent:#B8912F; }
|
| 351 |
+
.icv { direction:rtl; text-align:right; color:var(--ink); line-height:1.9; }
|
| 352 |
+
.icv .hero { background:linear-gradient(135deg,#0F4C3A,#17694F); border-radius:20px; padding:30px 22px 26px; margin:10px 0 18px; text-align:center;
|
| 353 |
+
box-shadow:0 6px 22px rgba(15,76,58,.18); border-bottom:4px solid var(--gold); }
|
| 354 |
+
.icv .hero .mark { width:54px; height:54px; display:block; margin:0 auto 4px; }
|
| 355 |
+
.icv .hero h1 { font-size:2.3rem; margin:.1rem 0; color:#FFFFFF; font-weight:700; }
|
| 356 |
+
.icv .hero .tagline { color:var(--gold-2); font-size:1.2rem; font-weight:600; margin:.1rem 0 .6rem; }
|
| 357 |
+
.icv .hero .sub { max-width:720px; margin:0 auto; color:#E9F2EE; font-size:1rem; }
|
| 358 |
+
.icv .flow { display:flex; flex-wrap:wrap; justify-content:center; align-items:center; gap:8px; margin-top:16px; }
|
| 359 |
+
.icv .flow span { background:rgba(255,255,255,.12); border:1px solid rgba(226,192,110,.55); color:#FFF; border-radius:999px; padding:3px 16px; font-size:.9rem; }
|
| 360 |
+
.icv .flow i { color:var(--gold-2); font-style:normal; font-size:1.2rem; }
|
| 361 |
+
.icv .disclaimer { font-size:.85rem; color:var(--muted); text-align:center; padding:14px 8px; }
|
| 362 |
+
.icv .en-free { direction:rtl; }
|
| 363 |
+
.input-area textarea { direction:rtl; text-align:right; font-family:'Amiri','Cairo',serif !important; font-size:1.25rem !important; line-height:2.1 !important; background:#FFFFFF !important; color:#1F2933 !important; }
|
| 364 |
+
.llm-row label, .gradio-container label span { font-family:'Cairo',sans-serif; }
|
| 365 |
+
.results { direction:rtl; }
|
| 366 |
+
.icv .summary { display:grid; grid-template-columns:repeat(auto-fit,minmax(130px,1fr)); gap:10px; margin:8px 0 14px; }
|
| 367 |
+
.icv .tile { background:var(--paper); border:1px solid var(--line); border-radius:14px; padding:12px 8px; text-align:center; }
|
| 368 |
+
.icv .tile b { display:block; font-size:1.8rem; color:var(--green); line-height:1.3; } .icv .tile span { font-size:.92rem; color:var(--muted); }
|
| 369 |
+
.icv .tile.verified { background:var(--ok-bg); border-color:var(--ok-line); } .icv .tile.verified b { color:var(--ok); }
|
| 370 |
+
.icv .tile.mismatch { background:var(--bad-bg); border-color:var(--bad-line); } .icv .tile.mismatch b { color:var(--bad); }
|
| 371 |
+
.icv .tile.review { background:var(--warn-bg); border-color:var(--warn-line); } .icv .tile.review b { color:var(--warn); }
|
| 372 |
+
.icv .tile.zero { background:var(--paper); border-color:var(--line); opacity:.6; } .icv .tile.zero b { color:var(--muted); }
|
| 373 |
+
.icv .highlight, .icv .generated, .icv .final { background:var(--paper); border:1px solid var(--line); border-radius:14px; padding:14px 18px; margin-bottom:14px; }
|
| 374 |
+
.icv label { display:block; color:var(--muted); font-size:.82rem; font-weight:600; margin-bottom:4px; }
|
| 375 |
+
.icv .highlight p, .icv .generated p, .icv .final-text { font-family:'Amiri','Cairo',serif; font-size:1.2rem; line-height:2.2; margin:0; white-space:pre-wrap; }
|
| 376 |
+
.icv mark { color:var(--ink); border-radius:6px; padding:1px 5px; }
|
| 377 |
+
.icv mark.verified { background:#D5EEDF; } .icv mark.mismatch { background:#F8CFCB; } .icv mark.review { background:#F7E3A6; }
|
| 378 |
+
.icv .hl-legend { display:flex; flex-wrap:wrap; gap:8px; margin-top:10px; font-size:.78rem; }
|
| 379 |
+
.icv .qcard { background:var(--paper); border:1px solid var(--line); border-inline-start:6px solid var(--line); border-radius:16px; padding:16px 18px; margin:12px 0; }
|
| 380 |
+
.icv .qcard.verified { border-inline-start-color:var(--ok); } .icv .qcard.mismatch { border-inline-start-color:var(--bad); } .icv .qcard.review { border-inline-start-color:var(--gold); }
|
| 381 |
+
.icv .qhead { display:flex; flex-wrap:wrap; gap:8px; align-items:center; margin-bottom:8px; }
|
| 382 |
+
.icv .idx { background:var(--green); color:#FFF; border-radius:50%; width:28px; height:28px; display:inline-flex; align-items:center; justify-content:center; font-weight:700; font-size:.9rem; }
|
| 383 |
+
.icv .badge { padding:2px 14px; border-radius:999px; font-size:.84rem; background:#F3EEDD; border:1px solid var(--line); color:var(--ink); }
|
| 384 |
+
.icv .badge.st.verified { background:var(--ok-bg); border-color:var(--ok-line); color:var(--ok); font-weight:600; }
|
| 385 |
+
.icv .badge.st.mismatch { background:var(--bad-bg); border-color:var(--bad-line); color:var(--bad); font-weight:600; }
|
| 386 |
+
.icv .badge.st.review { background:var(--warn-bg); border-color:var(--warn-line); color:var(--warn); font-weight:600; }
|
| 387 |
+
.icv .badge.scan { background:#EEF3FB; border-color:#C9D8EE; color:#2F4F7F; }
|
| 388 |
+
.icv .conf { margin-inline-start:auto; color:var(--muted); font-size:.88rem; } .icv .conf b { color:var(--green); }
|
| 389 |
+
.icv .field { margin:8px 0; } .icv .field p { margin:0; }
|
| 390 |
+
.icv .quote { font-family:'Amiri','Cairo',serif; font-size:1.25rem; line-height:2.2; color:var(--ink); } .icv .quote.small { font-size:1.05rem; color:#3a4651; }
|
| 391 |
+
.icv .diff { font-family:'Amiri','Cairo',serif; font-size:1.2rem; line-height:2.2; }
|
| 392 |
+
.icv .w-extra { background:var(--bad-bg); color:var(--bad); border-radius:4px; padding:0 4px; text-decoration:line-through; }
|
| 393 |
+
.icv .w-missing { background:var(--ok-bg); color:var(--ok); border-radius:4px; padding:0 4px; font-weight:700; }
|
| 394 |
+
.icv .legend { color:var(--muted); font-size:.82rem; margin:4px 0 0; } .icv .legend span { text-decoration:none; font-size:.8rem; }
|
| 395 |
+
.icv .reason { color:var(--muted); font-size:.92rem; margin:6px 0 0; }
|
| 396 |
+
.icv .note { color:var(--warn); background:var(--warn-bg); border:1px solid var(--warn-line); border-radius:10px; padding:5px 12px; font-size:.88rem; margin:8px 0 0; }
|
| 397 |
+
.icv .action { border-radius:12px; padding:10px 14px; margin-top:10px; }
|
| 398 |
+
.icv .action.ok { background:var(--ok-bg); border:1px solid var(--ok-line); } .icv .action.ok b { color:var(--ok); }
|
| 399 |
+
.icv .action.warn { background:var(--warn-bg); border:1px solid var(--warn-line); } .icv .action.warn b { color:var(--warn); }
|
| 400 |
+
.icv .action.bad { background:var(--bad-bg); border:1px solid var(--bad-line); } .icv .action.bad b { color:var(--bad); }
|
| 401 |
+
.icv .evidence { margin-top:12px; border-top:1px dashed var(--line); padding-top:8px; }
|
| 402 |
+
.icv .evidence summary { cursor:pointer; color:var(--green); font-weight:700; }
|
| 403 |
+
.icv .ev-row { margin:10px 0; } .icv .ev-row b { color:var(--gold); font-size:.88rem; } .icv .ev-row p { margin:2px 0; }
|
| 404 |
+
.icv .indicators { display:grid; gap:6px; margin-top:6px; }
|
| 405 |
+
.icv .ind { display:grid; grid-template-columns:150px 1fr 48px; gap:10px; align-items:center; font-size:.88rem; }
|
| 406 |
+
.icv .bar { background:#EFE8D3; border-radius:999px; height:9px; overflow:hidden; direction:rtl; } .icv .bar span { display:block; height:100%; background:linear-gradient(270deg,var(--green),var(--gold)); }
|
| 407 |
+
.icv .ind-val { text-align:left; direction:ltr; color:var(--ink); } .icv .ind-val.wide { grid-column:2 / span 2; text-align:right; direction:rtl; }
|
| 408 |
+
.icv .final { border-color:var(--gold); background:#FFFDF6; }
|
| 409 |
+
.icv .final-head { display:flex; justify-content:space-between; align-items:center; } .icv .final-head b { color:var(--green); font-size:1.05rem; }
|
| 410 |
+
.icv .copy { background:var(--green); color:#FFF; border:0; border-radius:10px; padding:5px 16px; font-family:inherit; cursor:pointer; }
|
| 411 |
+
.icv .notice { background:var(--paper); border:1px solid var(--line); border-radius:14px; padding:16px; }
|
| 412 |
+
.icv .notice.warn { background:var(--warn-bg); border-color:var(--warn-line); color:var(--warn); } .icv .notice.bad { background:var(--bad-bg); border-color:var(--bad-line); color:var(--bad); }
|
| 413 |
+
@media (max-width:640px){ .icv .hero h1{font-size:1.7rem;} .icv .ind{grid-template-columns:104px 1fr 40px;} }
|
| 414 |
+
""".replace("PATTERN", _PATTERN)
|
| 415 |
+
|
| 416 |
+
|
| 417 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 418 |
+
# Visual layer: soft glassmorphism, dynamic gradients, RTL-safe text, toast and skeleton loaders
|
| 419 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 420 |
+
GLASS_CSS = """
|
| 421 |
+
:root { --glass:rgba(255,255,255,.62); --glass-strong:rgba(255,255,255,.82); --glass-line:rgba(255,255,255,.75);
|
| 422 |
+
--shadow-1:0 1px 2px rgba(15,76,58,.06), 0 8px 24px rgba(15,76,58,.08); --shadow-2:0 2px 4px rgba(15,76,58,.08), 0 18px 44px rgba(15,76,58,.16); }
|
| 423 |
+
.gradio-container, body.icv-page { background:
|
| 424 |
+
radial-gradient(900px 520px at 88% -8%, rgba(226,192,110,.34), transparent 60%),
|
| 425 |
+
radial-gradient(760px 520px at 6% 4%, rgba(23,105,79,.20), transparent 62%),
|
| 426 |
+
linear-gradient(180deg,#FBF6EA 0%,#F3EBD3 100%) !important; background-attachment:fixed !important; }
|
| 427 |
+
/* RTL safety: long words, URLs and mixed Latin/digit runs wrap inside their box instead of spilling out */
|
| 428 |
+
.icv, .icv * { box-sizing:border-box; min-width:0; }
|
| 429 |
+
.icv p, .icv span, .icv b, .icv label, .icv summary, .icv button, .icv .badge, .icv .tile { overflow-wrap:anywhere; }
|
| 430 |
+
.icv .quote, .icv .diff, .icv .final-text, .icv .highlight p, .icv .generated p { unicode-bidi:plaintext; text-align:start; }
|
| 431 |
+
.icv .hero { position:relative; overflow:hidden; border-bottom:0; border:1px solid rgba(226,192,110,.45);
|
| 432 |
+
background:linear-gradient(120deg,#0B3B2D,#17694F 45%,#0F4C3A 70%,#1d7a5c); background-size:240% 240%; animation:icv-flow 16s ease-in-out infinite;
|
| 433 |
+
box-shadow:var(--shadow-2); }
|
| 434 |
+
.icv .hero::before { content:""; position:absolute; inset:-40% -10% auto auto; width:60%; aspect-ratio:1; border-radius:50%;
|
| 435 |
+
background:radial-gradient(circle, rgba(226,192,110,.38), transparent 65%); pointer-events:none; }
|
| 436 |
+
.icv .hero::after { content:""; position:absolute; inset:auto auto 0 0; width:100%; height:4px; background:linear-gradient(90deg,transparent,var(--gold-2),transparent); }
|
| 437 |
+
.icv .hero h1 { font-size:clamp(1.45rem,4.2vw,2.3rem); line-height:1.5; text-wrap:balance; position:relative; }
|
| 438 |
+
.icv .hero .tagline, .icv .hero .sub, .icv .flow { position:relative; }
|
| 439 |
+
.icv .flow span { backdrop-filter:blur(6px); -webkit-backdrop-filter:blur(6px); }
|
| 440 |
+
@keyframes icv-flow { 0%,100%{background-position:0% 50%} 50%{background-position:100% 50%} }
|
| 441 |
+
.icv .tile, .icv .highlight, .icv .generated, .icv .final, .icv .qcard, .icv .notice, .icv .export {
|
| 442 |
+
background:var(--glass); border:1px solid var(--glass-line); box-shadow:var(--shadow-1);
|
| 443 |
+
backdrop-filter:blur(14px) saturate(150%); -webkit-backdrop-filter:blur(14px) saturate(150%); transition:transform .25s ease, box-shadow .25s ease; }
|
| 444 |
+
.icv .qcard { border-inline-start:6px solid var(--line); }
|
| 445 |
+
.icv .qcard:hover, .icv .tile:hover { transform:translateY(-2px); box-shadow:var(--shadow-2); }
|
| 446 |
+
.icv .tile.verified { background:linear-gradient(160deg,rgba(234,246,239,.92),rgba(255,255,255,.6)); }
|
| 447 |
+
.icv .tile.mismatch { background:linear-gradient(160deg,rgba(253,236,234,.92),rgba(255,255,255,.6)); }
|
| 448 |
+
.icv .tile.review { background:linear-gradient(160deg,rgba(255,246,218,.95),rgba(255,255,255,.6)); }
|
| 449 |
+
.icv .final { background:linear-gradient(160deg,rgba(255,253,246,.92),rgba(250,240,208,.55)); border-color:rgba(184,145,47,.55); }
|
| 450 |
+
.icv .final-head { flex-wrap:wrap; gap:10px; }
|
| 451 |
+
.icv .qhead .badge { max-width:100%; white-space:normal; }
|
| 452 |
+
.icv .ind { grid-template-columns:minmax(96px,150px) minmax(0,1fr) 48px; }
|
| 453 |
+
.icv .bar span { background:linear-gradient(270deg,var(--green),var(--gold-2)); transition:width .6s ease; }
|
| 454 |
+
.icv .copy, .icv .dl { background:linear-gradient(135deg,var(--green),var(--green-2)); color:#FFF; border:0; border-radius:12px; padding:7px 18px;
|
| 455 |
+
font-family:inherit; font-weight:600; cursor:pointer; box-shadow:0 4px 12px rgba(15,76,58,.25); transition:transform .15s ease, box-shadow .15s ease, filter .15s ease; }
|
| 456 |
+
.icv .copy:hover, .icv .dl:hover { transform:translateY(-1px); filter:brightness(1.08); box-shadow:0 8px 18px rgba(15,76,58,.3); }
|
| 457 |
+
.icv .copy:active, .icv .dl:active { transform:translateY(0); }
|
| 458 |
+
.icv .copy:focus-visible, .icv .dl:focus-visible { outline:3px solid rgba(184,145,47,.55); outline-offset:2px; }
|
| 459 |
+
.icv .copy.done { background:linear-gradient(135deg,#1B7A4B,#2a9d66); }
|
| 460 |
+
.icv .export { border-radius:16px; padding:12px 18px; margin:12px 0; }
|
| 461 |
+
.icv .export summary { cursor:pointer; color:var(--green); font-weight:700; }
|
| 462 |
+
.icv .dls { display:flex; flex-wrap:wrap; gap:10px; margin-top:10px; }
|
| 463 |
+
/* toast */
|
| 464 |
+
#icv-toast { position:fixed; inset-inline:0; bottom:28px; margin-inline:auto; width:max-content; max-width:calc(100vw - 32px); z-index:99999;
|
| 465 |
+
direction:rtl; text-align:center; font:600 1rem 'Cairo','Segoe UI',Tahoma,sans-serif; color:#FFF; padding:12px 24px; border-radius:999px;
|
| 466 |
+
background:linear-gradient(135deg,rgba(15,76,58,.94),rgba(23,105,79,.94)); border:1px solid rgba(226,192,110,.6);
|
| 467 |
+
box-shadow:0 14px 40px rgba(15,76,58,.35); backdrop-filter:blur(10px); -webkit-backdrop-filter:blur(10px);
|
| 468 |
+
opacity:0; transform:translateY(18px) scale(.97); pointer-events:none; transition:opacity .28s ease, transform .28s cubic-bezier(.2,.9,.3,1.2); }
|
| 469 |
+
#icv-toast.show { opacity:1; transform:none; } #icv-toast.bad { background:linear-gradient(135deg,rgba(179,38,30,.95),rgba(214,69,58,.95)); }
|
| 470 |
+
/* skeleton loaders */
|
| 471 |
+
.skel { position:relative; overflow:hidden; background:rgba(15,76,58,.08); border-radius:10px; }
|
| 472 |
+
.skel::after { content:""; position:absolute; inset:0; transform:translateX(100%); animation:icv-shimmer 1.4s infinite;
|
| 473 |
+
background:linear-gradient(90deg,transparent,rgba(255,255,255,.75),transparent); }
|
| 474 |
+
@keyframes icv-shimmer { 100% { transform:translateX(-100%); } }
|
| 475 |
+
.skel-card { background:var(--glass); border:1px solid var(--glass-line); border-radius:16px; padding:16px 18px; margin:12px 0; box-shadow:var(--shadow-1); }
|
| 476 |
+
.skel-line { height:14px; margin:10px 0; } .skel-line.w60 { width:60%; } .skel-line.w85 { width:85%; } .skel-line.w40 { width:40%; }
|
| 477 |
+
.skel-tiles { display:grid; grid-template-columns:repeat(auto-fit,minmax(120px,1fr)); gap:10px; margin:8px 0 14px; } .skel-tile { height:78px; border-radius:14px; }
|
| 478 |
+
.icv-spinner { width:18px; height:18px; border-radius:50%; border:3px solid rgba(15,76,58,.18); border-top-color:var(--gold); display:inline-block;
|
| 479 |
+
vertical-align:middle; margin-inline-end:10px; animation:icv-spin .8s linear infinite; }
|
| 480 |
+
@keyframes icv-spin { to { transform:rotate(360deg); } }
|
| 481 |
+
@media (prefers-reduced-motion: reduce) { .icv .hero, .skel::after, .icv-spinner { animation:none; } .icv .qcard, .icv .tile, #icv-toast { transition:none; } }
|
| 482 |
+
@media (max-width:640px){ .icv .ind{grid-template-columns:minmax(84px,104px) minmax(0,1fr) 40px;} .icv .final-head{flex-direction:column; align-items:stretch;} .icv .copy{width:100%;} }
|
| 483 |
+
/* layout polish: centred, compact, professional */
|
| 484 |
+
.icv .notice, .icv .disclaimer, .icv .export { text-align:center; }
|
| 485 |
+
.icv .export .legend { max-width:640px; margin:8px auto 0; font-size:.9rem; line-height:1.9; color:var(--muted); }
|
| 486 |
+
.icv .dls { justify-content:center; }
|
| 487 |
+
.icv .dl { display:inline-flex; flex-direction:column; align-items:center; gap:2px; min-width:170px; }
|
| 488 |
+
.icv .dl .t { font-weight:700; } .icv .dl .s { font-size:.78rem; font-weight:500; opacity:.85; }
|
| 489 |
+
.icv .hero .flow { display:flex; justify-content:center; flex-wrap:wrap; gap:8px; }
|
| 490 |
+
"""
|
| 491 |
+
CSS += GLASS_CSS
|
| 492 |
+
|
| 493 |
+
SKELETON = (
|
| 494 |
+
'<div class="icv" aria-busy="true" aria-label="جارٍ التحقق">'
|
| 495 |
+
'<div class="skel-tiles">' + '<div class="skel skel-tile"></div>' * 4 + '</div>'
|
| 496 |
+
+ ('<div class="skel-card"><div class="skel skel-line w40"></div><div class="skel skel-line w85"></div>'
|
| 497 |
+
'<div class="skel skel-line"></div><div class="skel skel-line w60"></div></div>') * 2
|
| 498 |
+
+ '</div>'
|
| 499 |
+
)
|
| 500 |
+
|
| 501 |
+
COPY_JS = r"""
|
| 502 |
+
(function () {
|
| 503 |
+
if (window.icvCopy) return;
|
| 504 |
+
function toast(message, bad) {
|
| 505 |
+
var t = document.getElementById('icv-toast');
|
| 506 |
+
if (!t) { t = document.createElement('div'); t.id = 'icv-toast'; t.setAttribute('role', 'status'); t.setAttribute('aria-live', 'polite'); document.body.appendChild(t); }
|
| 507 |
+
t.textContent = message; t.className = 'show' + (bad ? ' bad' : '');
|
| 508 |
+
clearTimeout(window.__icvToast); window.__icvToast = setTimeout(function () { t.className = bad ? 'bad' : ''; }, 2400);
|
| 509 |
+
}
|
| 510 |
+
function legacyCopy(text) {
|
| 511 |
+
var area = document.createElement('textarea'); area.value = text; area.setAttribute('readonly', '');
|
| 512 |
+
area.style.cssText = 'position:fixed;top:0;left:0;opacity:0;pointer-events:none'; document.body.appendChild(area);
|
| 513 |
+
area.select(); area.setSelectionRange(0, text.length); var ok = false;
|
| 514 |
+
try { ok = document.execCommand('copy'); } catch (e) { ok = false; }
|
| 515 |
+
document.body.removeChild(area); return ok;
|
| 516 |
+
}
|
| 517 |
+
window.icvToast = toast;
|
| 518 |
+
window.icvCopy = function (button) {
|
| 519 |
+
var text = button.getAttribute('data-text') || '';
|
| 520 |
+
function finish(ok) {
|
| 521 |
+
toast(ok ? '__COPIED__' : 'تعذّر النسخ، حدّد النص وانسخه يدويًا', !ok);
|
| 522 |
+
if (ok) { button.classList.add('done'); setTimeout(function () { button.classList.remove('done'); }, 1600); }
|
| 523 |
+
}
|
| 524 |
+
if (navigator.clipboard && window.isSecureContext) {
|
| 525 |
+
navigator.clipboard.writeText(text).then(function () { finish(true); }, function () { finish(legacyCopy(text)); });
|
| 526 |
+
} else { finish(legacyCopy(text)); }
|
| 527 |
+
};
|
| 528 |
+
window.icvDownload = function (button) {
|
| 529 |
+
var blob = new Blob([button.getAttribute('data-text') || ''], { type: 'text/tab-separated-values;charset=utf-8' });
|
| 530 |
+
var link = document.createElement('a'); link.href = URL.createObjectURL(blob); link.download = button.getAttribute('data-name') || 'results.tsv';
|
| 531 |
+
document.body.appendChild(link); link.click(); document.body.removeChild(link); setTimeout(function () { URL.revokeObjectURL(link.href); }, 1000);
|
| 532 |
+
};
|
| 533 |
+
})();
|
| 534 |
+
""".replace("__COPIED__", COPIED_MESSAGE)
|
verifier.py
ADDED
|
@@ -0,0 +1,571 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Evidence-based verification and source-backed correction of Quran and Hadith quotations.
|
| 2 |
+
|
| 3 |
+
Pipeline (see ``IslamicContentVerifier``):
|
| 4 |
+
|
| 5 |
+
input text -> detection -> BM25 retrieval -> alignment (phonetic skeleton, sliding window, LCS, dynamic gap)
|
| 6 |
+
-> exact match : verified
|
| 7 |
+
-> altered Quran passage : mismatch with the exact source text as the correction
|
| 8 |
+
-> anything uncertain : human review (no correction is produced)
|
| 9 |
+
-> nothing similar found : no matching source
|
| 10 |
+
|
| 11 |
+
Safety rule: a correction is only ever the exact text of a retrieved source. Nothing is generated.
|
| 12 |
+
"""
|
| 13 |
+
from __future__ import annotations
|
| 14 |
+
|
| 15 |
+
import difflib
|
| 16 |
+
import logging
|
| 17 |
+
import time
|
| 18 |
+
from dataclasses import dataclass, field
|
| 19 |
+
from typing import Dict, List, Optional
|
| 20 |
+
|
| 21 |
+
from alignment import align
|
| 22 |
+
from detector import DetectedSpan
|
| 23 |
+
from scanner import HybridDetector
|
| 24 |
+
from idgham import apply_idgham
|
| 25 |
+
from normalization import content_words, normalize_for_matching
|
| 26 |
+
from retrieval import SourceRetriever
|
| 27 |
+
from similarity import best_match_score, compute_signals
|
| 28 |
+
|
| 29 |
+
logger = logging.getLogger(__name__)
|
| 30 |
+
|
| 31 |
+
MAX_INPUT_CHARS = 20_000
|
| 32 |
+
SHORT_QUOTE_WORDS = 3 # quotations shorter than this (found by the rules) always go to human review
|
| 33 |
+
MIN_EXACT_TOKENS = 3 # shorter quotations are too generic to be verified by exact containment alone
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 37 |
+
# Configuration (all tunable numbers live here)
|
| 38 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 39 |
+
@dataclass
|
| 40 |
+
class VerifierConfig:
|
| 41 |
+
"""Thresholds calibrated in Subtask 1B of the research notebook."""
|
| 42 |
+
|
| 43 |
+
quran_correct_threshold: float = 0.94
|
| 44 |
+
quran_uncertain_low: float = 0.45
|
| 45 |
+
quran_min_coverage: float = 0.40
|
| 46 |
+
hadith_correct_threshold: float = 0.88
|
| 47 |
+
hadith_uncertain_low: float = 0.30
|
| 48 |
+
hadith_min_coverage: float = 0.70
|
| 49 |
+
quran_top_k: int = 25
|
| 50 |
+
hadith_top_k: int = 15
|
| 51 |
+
hadith_retrieval_guard: float = 0.20
|
| 52 |
+
|
| 53 |
+
|
| 54 |
+
@dataclass
|
| 55 |
+
class CorrectorConfig:
|
| 56 |
+
"""Correction is proposed only when match strength >= ``*_strong``; between ``*_low`` and ``*_strong`` the
|
| 57 |
+
case goes to human review. Hadith is never auto-corrected (``hadith_strong`` > 1), by design: the exact
|
| 58 |
+
fragment boundaries of a Hadith quotation cannot be reproduced reliably, and narrations differ legitimately."""
|
| 59 |
+
|
| 60 |
+
max_window: int = 8
|
| 61 |
+
hadith_top_k: int = 40
|
| 62 |
+
quran_strong: float = 0.65
|
| 63 |
+
min_full_ratio: float = 0.40
|
| 64 |
+
quran_low: float = 0.55
|
| 65 |
+
hadith_strong: float = 1.01
|
| 66 |
+
hadith_low: float = 0.45
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
@dataclass
|
| 70 |
+
class PipelineConfig:
|
| 71 |
+
verifier: VerifierConfig = field(default_factory=VerifierConfig)
|
| 72 |
+
corrector: CorrectorConfig = field(default_factory=CorrectorConfig)
|
| 73 |
+
verified_min_conf: float = 0.75 # "Correct" verdicts weaker than this -> human review
|
| 74 |
+
unsupported_min_conf: float = 0.70 # "Incorrect + no source" needs this confidence to abstain confidently
|
| 75 |
+
unsupported_strength: float = 0.35 # ...or the best source match is this weak (nothing similar exists)
|
| 76 |
+
|
| 77 |
+
|
| 78 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 79 |
+
# Verification (Subtask 1B)
|
| 80 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 81 |
+
@dataclass
|
| 82 |
+
class Verification:
|
| 83 |
+
verdict: str # 'Correct' | 'Incorrect'
|
| 84 |
+
confidence: float
|
| 85 |
+
best_score: float
|
| 86 |
+
method: str
|
| 87 |
+
source: Optional[dict] = None # best matching source record (with 'signals')
|
| 88 |
+
n_candidates: int = 0
|
| 89 |
+
retrieval_top: float = 0.0
|
| 90 |
+
|
| 91 |
+
|
| 92 |
+
class Verifier:
|
| 93 |
+
"""Scores a quotation against retrieved candidates and returns a verdict with its supporting source."""
|
| 94 |
+
|
| 95 |
+
def __init__(self, retriever: SourceRetriever, config: Optional[VerifierConfig] = None) -> None:
|
| 96 |
+
self.kb = retriever
|
| 97 |
+
self.cfg = config or VerifierConfig()
|
| 98 |
+
|
| 99 |
+
def verify(self, span_text: str, content_type: str) -> Verification:
|
| 100 |
+
if not span_text or not span_text.strip():
|
| 101 |
+
return self._result("Incorrect", 0.95, 0.0, None, 0, "empty_span")
|
| 102 |
+
if content_type == "Ayah":
|
| 103 |
+
return self._verify_quran(span_text)
|
| 104 |
+
if content_type == "Hadith":
|
| 105 |
+
return self._verify_hadith(span_text)
|
| 106 |
+
return self._result("Incorrect", 0.5, 0.0, None, 0, "unknown_type")
|
| 107 |
+
|
| 108 |
+
def _verify_quran(self, span: str) -> Verification:
|
| 109 |
+
cfg = self.cfg
|
| 110 |
+
candidates = self.kb.search_quran_ayahs(span, top_k=cfg.quran_top_k)
|
| 111 |
+
if not candidates:
|
| 112 |
+
return self._result("Incorrect", 0.8, 0.0, None, 0, "no_candidates")
|
| 113 |
+
score, best = best_match_score(span, candidates, "Ayah")
|
| 114 |
+
signals = best.get("signals", {}) if best else {}
|
| 115 |
+
coverage, is_substring = signals.get("coverage", 0.0), signals.get("is_substring", 0)
|
| 116 |
+
n = len(candidates)
|
| 117 |
+
|
| 118 |
+
if is_substring and coverage >= cfg.quran_min_coverage:
|
| 119 |
+
return self._result("Correct", min(0.98, 0.85 + score * 0.15), score, best, n, "substring_match")
|
| 120 |
+
if score >= cfg.quran_correct_threshold and coverage >= cfg.quran_min_coverage:
|
| 121 |
+
return self._result("Correct", min(0.95, 0.70 + score * 0.25), score, best, n, "threshold_pass")
|
| 122 |
+
if score <= cfg.quran_uncertain_low:
|
| 123 |
+
return self._result("Incorrect", min(0.95, 0.70 + (1 - score) * 0.25), score, best, n, "threshold_fail")
|
| 124 |
+
|
| 125 |
+
strong = sum(
|
| 126 |
+
1 for cand in candidates[:10]
|
| 127 |
+
if (s := compute_signals(span, cand.get("text", ""), "Ayah"))["coverage"] >= 0.80 and s["lcs_ratio"] >= 0.75
|
| 128 |
+
)
|
| 129 |
+
if strong >= 2:
|
| 130 |
+
return self._result("Correct", 0.60 + min(0.20, strong * 0.05), score, best, n, "borderline_multi_cov")
|
| 131 |
+
return self._result("Incorrect", 0.58, score, best, n, "borderline_default")
|
| 132 |
+
|
| 133 |
+
def _verify_hadith(self, span: str) -> Verification:
|
| 134 |
+
cfg = self.cfg
|
| 135 |
+
candidates = self.kb.search_hadith(span, top_k=cfg.hadith_top_k)
|
| 136 |
+
if not candidates:
|
| 137 |
+
return self._result("Incorrect", 0.75, 0.0, None, 0, "no_candidates")
|
| 138 |
+
top_retrieval = candidates[0].get("retrieval_score", 0.0)
|
| 139 |
+
score, best = best_match_score(span, candidates, "Hadith")
|
| 140 |
+
signals = best.get("signals", {}) if best else {}
|
| 141 |
+
coverage, is_substring = signals.get("coverage", 0.0), signals.get("is_substring", 0)
|
| 142 |
+
n = len(candidates)
|
| 143 |
+
|
| 144 |
+
if is_substring and coverage >= cfg.hadith_min_coverage and top_retrieval >= cfg.hadith_retrieval_guard:
|
| 145 |
+
return self._result("Correct", min(0.97, 0.80 + score * 0.17), score, best, n, "substring_match", top_retrieval)
|
| 146 |
+
if score >= cfg.hadith_correct_threshold and coverage >= cfg.hadith_min_coverage:
|
| 147 |
+
return self._result("Correct", min(0.92, 0.65 + score * 0.27), score, best, n, "threshold_pass", top_retrieval)
|
| 148 |
+
if score <= cfg.hadith_uncertain_low:
|
| 149 |
+
return self._result("Incorrect", min(0.90, 0.65 + (1 - score) * 0.25), score, best, n, "threshold_fail", top_retrieval)
|
| 150 |
+
|
| 151 |
+
moderate = sum(
|
| 152 |
+
1 for cand in candidates[:8]
|
| 153 |
+
if (s := compute_signals(span, cand.get("text", ""), "Hadith"))["coverage"] >= 0.65 and s["lcs_ratio"] >= 0.55
|
| 154 |
+
)
|
| 155 |
+
if moderate >= 2 and top_retrieval >= 0.30:
|
| 156 |
+
return self._result("Correct", 0.58 + min(0.22, moderate * 0.06), score, best, n, "borderline_multi_cov", top_retrieval)
|
| 157 |
+
if top_retrieval < 0.25 or score < 0.45:
|
| 158 |
+
return self._result("Incorrect", 0.60, score, best, n, "borderline_low_retrieval", top_retrieval)
|
| 159 |
+
return self._result("Incorrect", 0.55, score, best, n, "borderline_default", top_retrieval)
|
| 160 |
+
|
| 161 |
+
@staticmethod
|
| 162 |
+
def _result(verdict, confidence, score, best, n_candidates, method, top_retrieval=0.0) -> Verification:
|
| 163 |
+
return Verification(verdict, round(confidence, 4), round(score, 4), method, best, n_candidates, round(top_retrieval, 4))
|
| 164 |
+
|
| 165 |
+
|
| 166 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 167 |
+
# Correction (Subtask 1C): locate the true ayah window / Hadith record and return its exact text
|
| 168 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 169 |
+
@dataclass
|
| 170 |
+
class CorrectionMatch:
|
| 171 |
+
kind: str # 'Ayah' | 'Hadith'
|
| 172 |
+
strength: float # coverage (Quran) / symmetric containment (Hadith)
|
| 173 |
+
full_ratio: float
|
| 174 |
+
text: str # proposed correction in the official 1C format (idgham + '(n)' ayah markers)
|
| 175 |
+
source: dict # reference metadata
|
| 176 |
+
display: str = "" # clean human-readable version
|
| 177 |
+
|
| 178 |
+
|
| 179 |
+
class Corrector:
|
| 180 |
+
def __init__(self, retriever: SourceRetriever, config: Optional[CorrectorConfig] = None) -> None:
|
| 181 |
+
self.kb = retriever
|
| 182 |
+
self.cfg = config or CorrectorConfig()
|
| 183 |
+
|
| 184 |
+
def match(self, span_text: str, content_type: str) -> Optional[CorrectionMatch]:
|
| 185 |
+
return self.match_quran(span_text) if content_type == "Ayah" else self.match_hadith(span_text)
|
| 186 |
+
|
| 187 |
+
def match_quran(self, query_text: str) -> Optional[CorrectionMatch]:
|
| 188 |
+
"""Best window of 1-8 consecutive ayahs (dynamic sliding window over BM25 seeds, exact-result pruning)."""
|
| 189 |
+
kb = self.kb
|
| 190 |
+
query_norm = normalize_for_matching(query_text)
|
| 191 |
+
query_words = content_words(query_norm.split())
|
| 192 |
+
if not query_words:
|
| 193 |
+
return None
|
| 194 |
+
query_len = len(query_norm)
|
| 195 |
+
memo: Dict[tuple, tuple] = {}
|
| 196 |
+
best = None # (key, coverage, ratio, surah, start, end)
|
| 197 |
+
for seed in kb.quran_seed_ayahs(query_words, top_k=25):
|
| 198 |
+
surah, ayah = kb.quran[seed]["surah_id"], kb.quran[seed]["ayah_id"]
|
| 199 |
+
ayahs = kb.quran_by_surah[surah]
|
| 200 |
+
min_ayah, max_ayah = min(ayahs), max(ayahs)
|
| 201 |
+
for offset in range(3):
|
| 202 |
+
start = ayah - offset
|
| 203 |
+
if start < min_ayah:
|
| 204 |
+
continue
|
| 205 |
+
window_len = -1
|
| 206 |
+
for length in range(1, self.cfg.max_window + 1):
|
| 207 |
+
end = start + length - 1
|
| 208 |
+
if end > max_ayah:
|
| 209 |
+
break
|
| 210 |
+
window_len += len(kb.q_norm_match[ayahs[end]]) + 1
|
| 211 |
+
len_diff = abs(window_len - query_len)
|
| 212 |
+
upper_bound = min(1.0, window_len / max(query_len, 1))
|
| 213 |
+
if best is not None: # exact-result pruning
|
| 214 |
+
best_cov, best_neg = best[0][0], best[0][1]
|
| 215 |
+
if upper_bound < best_cov or (upper_bound == best_cov and -len_diff < best_neg):
|
| 216 |
+
continue
|
| 217 |
+
key_pos = (surah, start, end)
|
| 218 |
+
if key_pos in memo:
|
| 219 |
+
continue
|
| 220 |
+
window = " ".join(kb.q_norm_match[ayahs[a]] for a in range(start, end + 1))
|
| 221 |
+
matcher = difflib.SequenceMatcher(None, query_norm, window, autojunk=False)
|
| 222 |
+
matched = sum(b.size for b in matcher.get_matching_blocks() if b.size >= 4)
|
| 223 |
+
coverage = matched / max(query_len, 1)
|
| 224 |
+
key = (coverage, -len_diff, matcher.ratio())
|
| 225 |
+
memo[key_pos] = key
|
| 226 |
+
if best is None or key > best[0]:
|
| 227 |
+
best = (key, coverage, key[2], surah, start, end)
|
| 228 |
+
if best is None:
|
| 229 |
+
return None
|
| 230 |
+
_, coverage, ratio, surah, start, end = best
|
| 231 |
+
display = " ".join(kb.quran[kb.quran_by_surah[surah][a]]["text"] for a in range(start, end + 1))
|
| 232 |
+
return CorrectionMatch(
|
| 233 |
+
"Ayah", coverage, ratio, self._ayah_text(surah, start, end),
|
| 234 |
+
{"type": "Quran", "surah_id": surah, "surah_name": kb.quran[kb.quran_by_surah[surah][start]]["surah_name"],
|
| 235 |
+
"ayah_start": start, "ayah_end": end},
|
| 236 |
+
display,
|
| 237 |
+
)
|
| 238 |
+
|
| 239 |
+
def _ayah_text(self, surah: int, start: int, end: int) -> str:
|
| 240 |
+
kb, multi = self.kb, end > start
|
| 241 |
+
parts = []
|
| 242 |
+
for a in range(start, end + 1):
|
| 243 |
+
text = kb.quran[kb.quran_by_surah[surah][a]]["text"]
|
| 244 |
+
parts.append(f"{text} ({a})" if multi else text)
|
| 245 |
+
return apply_idgham(" ".join(parts)).replace("\u0640", "")
|
| 246 |
+
|
| 247 |
+
def match_hadith(self, query_text: str) -> Optional[CorrectionMatch]:
|
| 248 |
+
kb = self.kb
|
| 249 |
+
query_norm = normalize_for_matching(query_text)
|
| 250 |
+
query_words = content_words(query_norm.split())
|
| 251 |
+
if not query_words:
|
| 252 |
+
return None
|
| 253 |
+
query_len = len(query_norm)
|
| 254 |
+
best = None # (key, idx, field, ratio)
|
| 255 |
+
for idx in kb.hadith_candidates(query_words, self.cfg.hadith_top_k):
|
| 256 |
+
for field_name in ("matn", "full"):
|
| 257 |
+
text = kb.hadith_norm(idx, field_name)
|
| 258 |
+
if not text:
|
| 259 |
+
continue
|
| 260 |
+
upper_bound = min(1.0, query_len / max(len(text), 1))
|
| 261 |
+
if best is not None and upper_bound < best[0][0]:
|
| 262 |
+
continue
|
| 263 |
+
matcher = difflib.SequenceMatcher(None, query_norm, text, autojunk=False)
|
| 264 |
+
matched = sum(b.size for b in matcher.get_matching_blocks() if b.size >= 4)
|
| 265 |
+
coverage, candidate_cov = matched / max(query_len, 1), matched / max(len(text), 1)
|
| 266 |
+
key = (min(coverage, candidate_cov), matcher.ratio())
|
| 267 |
+
if best is None or key > best[0]:
|
| 268 |
+
best = (key, idx, field_name, key[1])
|
| 269 |
+
if best is None:
|
| 270 |
+
return None
|
| 271 |
+
key, idx, field_name, ratio = best
|
| 272 |
+
record = kb.hadith[idx]
|
| 273 |
+
text = record[field_name].strip()
|
| 274 |
+
return CorrectionMatch(
|
| 275 |
+
"Hadith", key[0], ratio, text,
|
| 276 |
+
{"type": "Hadith", "hadithID": record["hadithID"], "book": record["book"], "title": record["title"],
|
| 277 |
+
"field": "matn" if field_name == "matn" else "full_text"},
|
| 278 |
+
text,
|
| 279 |
+
)
|
| 280 |
+
|
| 281 |
+
|
| 282 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 283 |
+
# End-to-end pipeline
|
| 284 |
+
# --------------------------------------------------------------------------------------------------------------
|
| 285 |
+
STATUS_INFO = {
|
| 286 |
+
"VERIFIED": {"ar": "موثّق", "group": "verified"},
|
| 287 |
+
"CORRECTED": {"ar": "غير مطابق — يوجد تصحيح من المصدر", "group": "mismatch"},
|
| 288 |
+
"UNSUPPORTED": {"ar": "غير مطابق — لا يوجد مصدر مطابق", "group": "mismatch"},
|
| 289 |
+
"HUMAN_REVIEW": {"ar": "يحتاج مراجعة بشرية", "group": "review"},
|
| 290 |
+
}
|
| 291 |
+
|
| 292 |
+
|
| 293 |
+
class IslamicContentVerifier:
|
| 294 |
+
"""Detect quotations, verify them against the corpora and decide: verified, corrected, unsupported or review."""
|
| 295 |
+
|
| 296 |
+
def __init__(self, retriever: Optional[SourceRetriever] = None, config: Optional[PipelineConfig] = None,
|
| 297 |
+
use_scanner: bool = True, decouple_triggers: bool = True) -> None:
|
| 298 |
+
self.cfg = config or PipelineConfig()
|
| 299 |
+
self.retriever = retriever or SourceRetriever()
|
| 300 |
+
self.verifier = Verifier(self.retriever, self.cfg.verifier)
|
| 301 |
+
self.corrector = Corrector(self.retriever, self.cfg.corrector)
|
| 302 |
+
self.detector = HybridDetector(self.retriever, use_scanner, rules_use_corpus=decouple_triggers,
|
| 303 |
+
decouple_triggers=decouple_triggers)
|
| 304 |
+
self.detector_name = type(self.detector).__name__
|
| 305 |
+
|
| 306 |
+
# ---- public API -----------------------------------------------------------------------------------------
|
| 307 |
+
def detect(self, text: str) -> List[DetectedSpan]:
|
| 308 |
+
return self.detector.detect(self._validate(text)) if text.strip() else []
|
| 309 |
+
|
| 310 |
+
def needs_hadith(self, text: str) -> bool:
|
| 311 |
+
"""True if the text contains a Hadith quotation (so the large Hadith index must be loaded)."""
|
| 312 |
+
return any(span.label == "Hadith" for span in self.detect(text))
|
| 313 |
+
|
| 314 |
+
def analyze(self, text: str) -> dict:
|
| 315 |
+
"""Run the full pipeline on a generated text."""
|
| 316 |
+
text = self._validate(text)
|
| 317 |
+
started = time.time()
|
| 318 |
+
spans = self.detector.detect(text) if text.strip() else []
|
| 319 |
+
detect_seconds = time.time() - started
|
| 320 |
+
result = self._analyze_spans(text, spans)
|
| 321 |
+
result["timings"] = {"detect_s": round(detect_seconds, 3), "total_s": round(time.time() - started, 3)}
|
| 322 |
+
return result
|
| 323 |
+
|
| 324 |
+
def analyze_spans(self, text: str, spans: List[dict]) -> dict:
|
| 325 |
+
"""Skip detection and use given spans ``[{label, start, end}]`` (evaluation / oracle mode)."""
|
| 326 |
+
given = [DetectedSpan(s["start"], s["end"], s["label"], None, "given", text[s["start"]:s["end"]]) for s in spans]
|
| 327 |
+
return self._analyze_spans(self._validate(text), given)
|
| 328 |
+
|
| 329 |
+
def analyze_detected(self, text: str, spans: List[DetectedSpan]) -> dict:
|
| 330 |
+
"""Verify spans produced elsewhere (e.g. detector + hosted model merged by ``camelbert_adapter.merge_spans``)."""
|
| 331 |
+
return self._analyze_spans(self._validate(text), spans)
|
| 332 |
+
|
| 333 |
+
# ---- internals ------------------------------------------------------------------------------------------
|
| 334 |
+
@staticmethod
|
| 335 |
+
def _validate(text: str) -> str:
|
| 336 |
+
if not isinstance(text, str):
|
| 337 |
+
raise TypeError("Input text must be a string")
|
| 338 |
+
if len(text) > MAX_INPUT_CHARS:
|
| 339 |
+
raise ValueError(f"Input is too long ({len(text)} characters); the limit is {MAX_INPUT_CHARS}")
|
| 340 |
+
return text
|
| 341 |
+
|
| 342 |
+
def _analyze_spans(self, text: str, spans: List[DetectedSpan]) -> dict:
|
| 343 |
+
reports = [self._process_span(i + 1, span) for i, span in enumerate(sorted(spans, key=lambda s: s.start))]
|
| 344 |
+
counts = {status: 0 for status in STATUS_INFO}
|
| 345 |
+
for report in reports:
|
| 346 |
+
counts[report["status"]] += 1
|
| 347 |
+
return {
|
| 348 |
+
"input_text": text,
|
| 349 |
+
"detector": self.detector_name,
|
| 350 |
+
"spans": reports,
|
| 351 |
+
"corrected_text": self._apply_corrections(text, reports),
|
| 352 |
+
"summary": {
|
| 353 |
+
"n_spans": len(reports),
|
| 354 |
+
"n_ayah": sum(r["type"] == "Ayah" for r in reports),
|
| 355 |
+
"n_hadith": sum(r["type"] == "Hadith" for r in reports),
|
| 356 |
+
**counts,
|
| 357 |
+
"needs_human_review": counts["HUMAN_REVIEW"] > 0,
|
| 358 |
+
},
|
| 359 |
+
}
|
| 360 |
+
|
| 361 |
+
def _process_span(self, index: int, span: DetectedSpan) -> dict:
|
| 362 |
+
try:
|
| 363 |
+
return self._decide(index, span)
|
| 364 |
+
except Exception: # a single failing quotation must not break the whole report
|
| 365 |
+
logger.exception("Failed to process span %d", index)
|
| 366 |
+
report = self._empty_report(index, span)
|
| 367 |
+
self._finalize(report, "HUMAN_REVIEW", {"code": "internal_error"})
|
| 368 |
+
return report
|
| 369 |
+
|
| 370 |
+
@staticmethod
|
| 371 |
+
def _empty_report(index: int, span: DetectedSpan) -> dict:
|
| 372 |
+
return {
|
| 373 |
+
"id": index, "type": span.label, "start": span.start, "end": span.end, "text": span.text,
|
| 374 |
+
"detection": {"backend": span.source, "confidence": None if span.confidence is None else round(span.confidence, 4)},
|
| 375 |
+
"verification": {"verdict": "Incorrect", "confidence": 0.0, "score": 0.0, "method": "error", "n_candidates": 0},
|
| 376 |
+
"evidence": None, "correction": None, "suggestion": None, "notes": [],
|
| 377 |
+
}
|
| 378 |
+
|
| 379 |
+
@staticmethod
|
| 380 |
+
def _finalize(report: dict, status: str, reason: dict) -> None:
|
| 381 |
+
info = STATUS_INFO[status]
|
| 382 |
+
report.update(status=status, status_ar=info["ar"], group=info["group"], reason=reason)
|
| 383 |
+
|
| 384 |
+
@staticmethod
|
| 385 |
+
def source_label(source: dict) -> str:
|
| 386 |
+
"""Plain reference used in reasons, e.g. ``سورة البقرة 153`` or ``حديث رقم 5``."""
|
| 387 |
+
if source["type"] == "Quran":
|
| 388 |
+
start, end = source["ayah_start"], source["ayah_end"]
|
| 389 |
+
return f"سورة {source['surah_name']} {start}" + (f"–{end}" if end != start else "")
|
| 390 |
+
return f"حديث رقم {source['hadithID']}"
|
| 391 |
+
|
| 392 |
+
def _decide(self, index: int, span: DetectedSpan) -> dict:
|
| 393 |
+
report = self._empty_report(index, span)
|
| 394 |
+
verification = self.verifier.verify(span.text, span.label)
|
| 395 |
+
report["verification"] = {
|
| 396 |
+
"verdict": verification.verdict, "confidence": verification.confidence, "score": verification.best_score,
|
| 397 |
+
"method": verification.method, "n_candidates": verification.n_candidates,
|
| 398 |
+
}
|
| 399 |
+
whole = self._whole_ayah(span.text) if span.label == "Ayah" else None
|
| 400 |
+
if whole is not None:
|
| 401 |
+
self._verified_whole_ayah(report, span, whole) # a complete ayah, however short or common its words
|
| 402 |
+
elif span.label == "Ayah":
|
| 403 |
+
self._decide_quran(report, span, verification)
|
| 404 |
+
else:
|
| 405 |
+
self._decide_hadith(report, span, verification)
|
| 406 |
+
if whole is None and span.source == "rules" and len(normalize_for_matching(span.text).split()) < SHORT_QUOTE_WORDS:
|
| 407 |
+
# two words cannot identify a source: never claim "verified" or "wrong", and never correct
|
| 408 |
+
self._finalize(report, "HUMAN_REVIEW", {"code": "too_short"})
|
| 409 |
+
report["correction"] = None
|
| 410 |
+
report["suggestion"] = None
|
| 411 |
+
elif report["status"] in ("UNSUPPORTED", "HUMAN_REVIEW"):
|
| 412 |
+
self._cross_check(report, span)
|
| 413 |
+
if report["status"] == "VERIFIED" and span.hint and span.hint != span.label:
|
| 414 |
+
report["notes"].append({"code": "is_ayah" if span.label == "Ayah" else "is_hadith", "source": self.source_label(report["evidence"]["source"]), "misattributed": True})
|
| 415 |
+
return report
|
| 416 |
+
|
| 417 |
+
# ---- helpers for the decision ---------------------------------------------------------------------------
|
| 418 |
+
def _whole_ayah(self, text: str) -> Optional[int]:
|
| 419 |
+
"""Index of the ayah whose complete text equals ``text`` (two words or more), else None."""
|
| 420 |
+
if getattr(self, "_whole_map", None) is None:
|
| 421 |
+
mapping: Dict[str, int] = {}
|
| 422 |
+
for i, norm in enumerate(self.retriever.q_norm_match):
|
| 423 |
+
if len(norm.split()) >= 2:
|
| 424 |
+
mapping.setdefault(norm, i)
|
| 425 |
+
self._whole_map = mapping
|
| 426 |
+
norm = normalize_for_matching(text)
|
| 427 |
+
return self._whole_map.get(norm) if len(norm.split()) >= 2 else None
|
| 428 |
+
|
| 429 |
+
def _verified_whole_ayah(self, report: dict, span: DetectedSpan, idx: int) -> None:
|
| 430 |
+
record = self.retriever.quran[idx]
|
| 431 |
+
source_ref = {"type": "Quran", "surah_id": record["surah_id"], "surah_name": record["surah_name"],
|
| 432 |
+
"ayah_start": record["ayah_id"], "ayah_end": record["ayah_id"]}
|
| 433 |
+
alignment = align(span.text, record["text"], self.retriever.quran_vocabulary)
|
| 434 |
+
report["evidence"] = {"source": source_ref, "signals": compute_signals(span.text, record["text"], "Ayah"),
|
| 435 |
+
"comparison": alignment}
|
| 436 |
+
report["verification"].update(verdict="Correct", confidence=0.98, score=1.0, method="whole_ayah")
|
| 437 |
+
self._finalize(report, "VERIFIED", {"code": "exact_match", "source": self.source_label(source_ref)})
|
| 438 |
+
if alignment["diacritic_notes"]:
|
| 439 |
+
report["notes"].append({"code": "diacritic_conflict", "n": len(alignment["diacritic_notes"])})
|
| 440 |
+
|
| 441 |
+
def _cross_check(self, report: dict, span: DetectedSpan) -> None:
|
| 442 |
+
"""A quotation that fails against its claimed corpus but is found verbatim in the other one is probably
|
| 443 |
+
mis-attributed (a Hadith presented as an ayah, or the reverse): say so."""
|
| 444 |
+
if len(normalize_for_matching(span.text).split()) < 3:
|
| 445 |
+
return
|
| 446 |
+
try:
|
| 447 |
+
if span.label == "Ayah":
|
| 448 |
+
other = self.verifier._verify_hadith(span.text)
|
| 449 |
+
if other.verdict == "Correct" and other.method == "substring_match" and other.source:
|
| 450 |
+
report["notes"].append({"code": "is_hadith", "source": f"حديث رقم {other.source['hadithID']}"})
|
| 451 |
+
else:
|
| 452 |
+
other = self.verifier._verify_quran(span.text)
|
| 453 |
+
if other.verdict == "Correct" and other.method == "substring_match" and other.source:
|
| 454 |
+
c = other.source
|
| 455 |
+
report["notes"].append({"code": "is_ayah", "source": f"سورة {c['surah_name']} {c['ayah_id']}"})
|
| 456 |
+
except Exception: # a diagnostic note must never break the decision
|
| 457 |
+
logger.exception("cross-check failed")
|
| 458 |
+
|
| 459 |
+
@staticmethod
|
| 460 |
+
def _proposal(span: DetectedSpan, match: Optional[CorrectionMatch]) -> Optional[dict]:
|
| 461 |
+
if match is None:
|
| 462 |
+
return None
|
| 463 |
+
return {"text": match.text, "display_text": match.display, "source": match.source,
|
| 464 |
+
"match_strength": round(match.strength, 4), "full_ratio": round(match.full_ratio, 4)}
|
| 465 |
+
|
| 466 |
+
def _decide_quran(self, report: dict, span: DetectedSpan, verification: Verification) -> None:
|
| 467 |
+
cfg, corr_cfg = self.cfg, self.cfg.corrector
|
| 468 |
+
match = self.corrector.match_quran(span.text)
|
| 469 |
+
if match is not None:
|
| 470 |
+
source_text, source_ref = match.display, match.source
|
| 471 |
+
elif verification.source:
|
| 472 |
+
candidate = verification.source
|
| 473 |
+
source_text = candidate["text"]
|
| 474 |
+
source_ref = {"type": "Quran", "surah_id": candidate["surah_id"], "surah_name": candidate["surah_name"],
|
| 475 |
+
"ayah_start": candidate["ayah_id"], "ayah_end": candidate["ayah_id"]}
|
| 476 |
+
else:
|
| 477 |
+
source_text, source_ref = "", None
|
| 478 |
+
alignment = align(span.text, source_text, self.retriever.quran_vocabulary) if source_text else None
|
| 479 |
+
if source_ref:
|
| 480 |
+
report["evidence"] = {"source": source_ref, "signals": compute_signals(span.text, source_text, "Ayah"),
|
| 481 |
+
"comparison": alignment}
|
| 482 |
+
proposal = self._proposal(span, match)
|
| 483 |
+
n_tokens = len(content_words(normalize_for_matching(span.text).split())) if span.text else 0
|
| 484 |
+
has_tokens = alignment is not None and alignment["exact"] and len(alignment["word_diff"]) >= 1
|
| 485 |
+
|
| 486 |
+
if has_tokens and sum(len(op["span"].split()) for op in alignment["word_diff"]) >= MIN_EXACT_TOKENS:
|
| 487 |
+
self._finalize(report, "VERIFIED", {"code": "exact_match", "source": self.source_label(source_ref)})
|
| 488 |
+
if alignment["diacritic_notes"]:
|
| 489 |
+
report["notes"].append({"code": "diacritic_conflict", "n": len(alignment["diacritic_notes"])})
|
| 490 |
+
if alignment["orthographic_variants"]:
|
| 491 |
+
report["notes"].append({"code": "orthographic_variant", "n": alignment["orthographic_variants"]})
|
| 492 |
+
return
|
| 493 |
+
|
| 494 |
+
local = bool(alignment and alignment.get("near") and alignment.get("word_similarity", 0) >= 0.75 and alignment.get("source_excerpt"))
|
| 495 |
+
strong = match is not None and match.strength >= corr_cfg.quran_strong and (match.full_ratio >= corr_cfg.min_full_ratio or local)
|
| 496 |
+
if strong:
|
| 497 |
+
self._finalize(report, "CORRECTED", {"code": "altered_passage", "source": self.source_label(source_ref),
|
| 498 |
+
"n": alignment["mismatches"] if alignment else 0,
|
| 499 |
+
"reordered": bool(alignment and alignment["reordered"])})
|
| 500 |
+
# replace only the passage the author quoted, not the whole ayah that contains it
|
| 501 |
+
excerpt = alignment["source_excerpt"] if alignment and alignment["source_excerpt"] else proposal["display_text"]
|
| 502 |
+
report["correction"] = {**proposal, "display_text": excerpt, "applied": True}
|
| 503 |
+
return
|
| 504 |
+
|
| 505 |
+
if verification.verdict == "Correct":
|
| 506 |
+
self._finalize(report, "HUMAN_REVIEW", {"code": "weak_match"})
|
| 507 |
+
report["suggestion"] = proposal
|
| 508 |
+
elif match is None or match.strength < corr_cfg.quran_low:
|
| 509 |
+
confident = verification.confidence >= cfg.unsupported_min_conf and not verification.method.startswith("borderline")
|
| 510 |
+
if match is None or match.strength < cfg.unsupported_strength or confident:
|
| 511 |
+
self._finalize(report, "UNSUPPORTED", {"code": "no_source"})
|
| 512 |
+
else:
|
| 513 |
+
self._finalize(report, "HUMAN_REVIEW", {"code": "insufficient_evidence"})
|
| 514 |
+
report["suggestion"] = proposal
|
| 515 |
+
else:
|
| 516 |
+
self._finalize(report, "HUMAN_REVIEW", {"code": "candidate_not_strong", "source": self.source_label(source_ref),
|
| 517 |
+
"strength": round(match.strength, 2)})
|
| 518 |
+
report["suggestion"] = proposal
|
| 519 |
+
|
| 520 |
+
def _decide_hadith(self, report: dict, span: DetectedSpan, verification: Verification) -> None:
|
| 521 |
+
cfg, corr_cfg = self.cfg, self.cfg.corrector
|
| 522 |
+
best = verification.source
|
| 523 |
+
alignment = None
|
| 524 |
+
if best:
|
| 525 |
+
alignment = align(span.text, best["text"])
|
| 526 |
+
report["evidence"] = {
|
| 527 |
+
"source": {"type": "Hadith", "hadithID": best["hadithID"], "book": best["book"], "title": best["title"]},
|
| 528 |
+
"signals": best.get("signals"), "comparison": alignment,
|
| 529 |
+
}
|
| 530 |
+
exact_tokens = alignment is not None and alignment["exact"] and sum(len(op["span"].split()) for op in alignment["word_diff"]) >= 4
|
| 531 |
+
|
| 532 |
+
if verification.verdict == "Correct" or exact_tokens:
|
| 533 |
+
altered = (not exact_tokens and alignment is not None and not alignment["exact"] and alignment["mismatches"] >= 1)
|
| 534 |
+
if altered: # any added, dropped or changed word can alter the meaning (خيرًا / شرًّا, or a missing "لا"): only an exact match is verified
|
| 535 |
+
match = self.corrector.match_hadith(span.text)
|
| 536 |
+
self._finalize(report, "HUMAN_REVIEW", {"code": "hadith_altered", "n": alignment["mismatches"],
|
| 537 |
+
"source": f"حديث رقم {best['hadithID']}"})
|
| 538 |
+
report["suggestion"] = self._proposal(span, match)
|
| 539 |
+
report["verification"]["verdict"] = "Incorrect"
|
| 540 |
+
elif exact_tokens or (not verification.method.startswith("borderline") and verification.confidence >= cfg.verified_min_conf):
|
| 541 |
+
self._finalize(report, "VERIFIED", {"code": "exact_match" if exact_tokens else "close_match",
|
| 542 |
+
"source": f"حديث رقم {best['hadithID']}"})
|
| 543 |
+
if alignment and not alignment["exact"]:
|
| 544 |
+
report["notes"].append({"code": "hadith_minor_diffs", "n": alignment["mismatches"]})
|
| 545 |
+
if alignment and alignment["diacritic_notes"]:
|
| 546 |
+
report["notes"].append({"code": "diacritic_conflict", "n": len(alignment["diacritic_notes"])})
|
| 547 |
+
else:
|
| 548 |
+
self._finalize(report, "HUMAN_REVIEW", {"code": "weak_match"})
|
| 549 |
+
return
|
| 550 |
+
|
| 551 |
+
match = self.corrector.match_hadith(span.text)
|
| 552 |
+
proposal = self._proposal(span, match)
|
| 553 |
+
if match is None or match.strength < corr_cfg.hadith_low:
|
| 554 |
+
confident = verification.confidence >= cfg.unsupported_min_conf and not verification.method.startswith("borderline")
|
| 555 |
+
if match is None or match.strength < cfg.unsupported_strength or confident:
|
| 556 |
+
self._finalize(report, "UNSUPPORTED", {"code": "no_source"})
|
| 557 |
+
else:
|
| 558 |
+
self._finalize(report, "HUMAN_REVIEW", {"code": "insufficient_evidence"})
|
| 559 |
+
report["suggestion"] = proposal
|
| 560 |
+
else: # a similar narration exists, but Hadith is never corrected automatically
|
| 561 |
+
self._finalize(report, "HUMAN_REVIEW", {"code": "hadith_candidate", "source": self.source_label(match.source),
|
| 562 |
+
"strength": round(match.strength, 2)})
|
| 563 |
+
report["suggestion"] = proposal
|
| 564 |
+
|
| 565 |
+
@staticmethod
|
| 566 |
+
def _apply_corrections(text: str, reports: List[dict]) -> str:
|
| 567 |
+
out = text
|
| 568 |
+
for report in sorted(reports, key=lambda r: r["start"], reverse=True):
|
| 569 |
+
if report["status"] == "CORRECTED" and report["correction"] and report["correction"].get("applied"):
|
| 570 |
+
out = out[: report["start"]] + report["correction"]["display_text"] + out[report["end"]:]
|
| 571 |
+
return out
|