Ghada-99-Ragab commited on
Commit
0dff1a5
·
verified ·
1 Parent(s): 8989767

Upload 26 files

Browse files
Files changed (27) hide show
  1. .gitattributes +1 -0
  2. README.md +12 -5
  3. README_AR.md +5 -0
  4. _headers +7 -0
  5. alignment.py +180 -0
  6. app.py +181 -0
  7. benchmark_format.py +115 -0
  8. camelbert_adapter.py +103 -0
  9. config.json +7 -0
  10. detector.py +178 -0
  11. examples.json +227 -0
  12. hadith.idx.gz +3 -0
  13. hadith.json +3 -0
  14. idgham.py +71 -0
  15. index.html +516 -17
  16. index_builder.py +163 -0
  17. islamic_unified_dataset.jsonl +0 -0
  18. islamiceval_dev_subset.jsonl +0 -0
  19. llm_client.py +96 -0
  20. normalization.py +117 -0
  21. quran.idx.gz +3 -0
  22. quran.json +0 -0
  23. retrieval.py +187 -0
  24. scanner.py +349 -0
  25. similarity.py +121 -0
  26. ui.py +534 -0
  27. verifier.py +571 -0
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ hadith.json filter=lfs diff=lfs merge=lfs -text
README.md CHANGED
@@ -1,10 +1,17 @@
1
  ---
2
- title: Lastfinal2
3
- emoji: 🐨
4
- colorFrom: indigo
5
- colorTo: gray
6
  sdk: static
 
7
  pinned: false
 
 
8
  ---
9
 
10
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
 
 
1
  ---
2
+ title: Quran & Hadith Verification
3
+ emoji: 🕌
4
+ colorFrom: green
5
+ colorTo: yellow
6
  sdk: static
7
+ app_file: index.html
8
  pinned: false
9
+ license: mit
10
+ short_description: Verify Quran and Hadith quotations
11
  ---
12
 
13
+ # Quran & Hadith Verification
14
+
15
+ Finds Quran verses and Hadith in text and verifies them against their sources.
16
+
17
+ Arabic version: [README_AR.md](README_AR.md)
README_AR.md ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ # التحقق من هلوسة القرآن والحديث وتصحيحها
2
+
3
+ يكتشف الآيات والأحاديث في أي نص، ويعرض الدليل، ويقترح التصحيح من نص المصدر.
4
+ يعمل الكود كاملًا داخل متصفحك (بايثون عبر Pyodide)، ولا يُرسل النص الملصوق إلى أي خادم.
5
+ النسخة الإنجليزية: [README.md](README.md).
_headers ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ /*
2
+ X-Content-Type-Options: nosniff
3
+ Referrer-Policy: no-referrer
4
+ /index/*
5
+ Cache-Control: public, max-age=86400
6
+ /config.json
7
+ Cache-Control: no-store
alignment.py ADDED
@@ -0,0 +1,180 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Word-level sequence alignment between a quotation and a source text.
2
+
3
+ * Phonetic skeletons (``normalization.aligned_words``) make Uthmani / modern spelling and diacritics irrelevant for
4
+ matching, while every substituted, missing, extra or re-ordered word still shows up as an edit operation.
5
+ * A dynamic sliding window locates the best-matching region inside long sources (Hadith with chains of narrators).
6
+ * The gap threshold scales with quotation length: a short quotation tolerates no stray words, a long one a few.
7
+ * Diacritic conflicts are reported separately as notes, because spelling of harakat differs legitimately between prints.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ from difflib import SequenceMatcher
12
+ from math import ceil
13
+ from typing import Container, Dict, List, Optional, Sequence, Tuple
14
+
15
+ from idgham import apply_idgham
16
+ from normalization import aligned_words, vowel_signature
17
+
18
+
19
+ def lcs_length(a: Sequence[str], b: Sequence[str]) -> int:
20
+ m, n = len(a), len(b)
21
+ if m == 0 or n == 0:
22
+ return 0
23
+ if m < n:
24
+ a, b, m, n = b, a, n, m
25
+ prev = [0] * (n + 1)
26
+ for i in range(m):
27
+ curr = [0] * (n + 1)
28
+ for j in range(n):
29
+ curr[j + 1] = prev[j] + 1 if a[i] == b[j] else max(curr[j], prev[j + 1])
30
+ prev = curr
31
+ return prev[n]
32
+
33
+
34
+ def allowed_gap(n_tokens: int) -> int:
35
+ """Dynamic gap threshold: how many differing words still count as 'the same passage, altered'."""
36
+ return 0 if n_tokens <= 3 else max(1, ceil(0.2 * n_tokens))
37
+
38
+
39
+ def find_window(quote: List[str], source: List[str], gap: int) -> Tuple[int, int]:
40
+ """Best source region for the quotation: windows of ``len(quote)+gap`` ranked by shared words, refined by LCS."""
41
+ n, window = len(quote), len(quote) + gap
42
+ if len(source) <= window + 2 * gap + 8:
43
+ return 0, len(source)
44
+ quote_set = set(quote)
45
+ prefix = [0]
46
+ for word in source:
47
+ prefix.append(prefix[-1] + (word in quote_set))
48
+ starts = sorted(range(len(source) - window + 1), key=lambda i: prefix[i + window] - prefix[i], reverse=True)[:8]
49
+ best_start = max(starts, key=lambda i: lcs_length(quote, source[i : i + window]))
50
+ return max(0, best_start - gap), min(len(source), best_start + window + gap)
51
+
52
+
53
+ def contains_sequence(source: Sequence[str], quote: Sequence[str]) -> bool:
54
+ """True if ``quote`` appears as a contiguous run of whole words in ``source``."""
55
+ n = len(quote)
56
+ return n > 0 and any(source[i : i + n] == list(quote) for i in range(len(source) - n + 1))
57
+
58
+
59
+ def locate_region(q_norm: List[str], s_norm: List[str], gap: int, slack: Optional[int] = None):
60
+ """Region of the source that corresponds to the quotation (matched part plus slack for edge differences).
61
+
62
+ Returns ``(lo, hi, matcher)`` where the matcher compares the quotation with ``s_norm[lo:hi]``."""
63
+ slack = gap if slack is None else slack
64
+ lo, hi = find_window(q_norm, s_norm, gap)
65
+ matcher = SequenceMatcher(None, q_norm, s_norm[lo:hi], autojunk=False)
66
+ blocks = [b for b in matcher.get_matching_blocks() if b.size]
67
+ if blocks:
68
+ first, last = blocks[0], blocks[-1]
69
+ base = lo
70
+ lo = base + max(0, first.b - first.a - slack)
71
+ hi = min(hi, base + last.b + last.size + (len(q_norm) - last.a - last.size) + slack)
72
+ matcher = SequenceMatcher(None, q_norm, s_norm[lo:hi], autojunk=False)
73
+ return lo, hi, matcher
74
+
75
+
76
+ def best_region(quote: str, source: str) -> str:
77
+ """The part of ``source`` that a (possibly partial) quotation refers to, as original words.
78
+
79
+ Scoring a snippet against the whole verse / Hadith would punish it for being shorter than its source; comparing it
80
+ with this region keeps the score about what was actually quoted."""
81
+ q_pairs, s_pairs = aligned_words(quote), aligned_words(source)
82
+ if not q_pairs or len(s_pairs) <= len(q_pairs) + allowed_gap(len(q_pairs)) + 2:
83
+ return source
84
+ q_norm, s_norm = [p[1] for p in q_pairs], [p[1] for p in s_pairs]
85
+ lo, hi, _ = locate_region(q_norm, s_norm, allowed_gap(len(q_norm)), slack=0)
86
+ return " ".join(p[0] for p in s_pairs[lo:hi]) if hi > lo else source
87
+
88
+
89
+ def _diacritic_notes(quote_words: List[str], source_words: List[str], idgham_words: List[str]) -> List[dict]:
90
+ """Words whose written harakat contradict the source (missing harakat are fine; idgham spelling is accepted)."""
91
+ notes = []
92
+ for q_word, s_word, g_word in zip(quote_words, source_words, idgham_words):
93
+ q_sig = vowel_signature(q_word)
94
+ if not any(marks for _, marks in q_sig):
95
+ continue
96
+ conflict = True
97
+ for variant in (s_word, g_word):
98
+ v_sig = vowel_signature(variant)
99
+ if len(v_sig) == len(q_sig) and all(set(qm) <= set(vm) for (_, qm), (_, vm) in zip(q_sig, v_sig)):
100
+ conflict = False
101
+ break
102
+ if conflict:
103
+ notes.append({"word": q_word, "source_word": s_word})
104
+ return notes
105
+
106
+
107
+ def _orthographic_variant(q_tokens: Sequence[str], s_tokens: Sequence[str], vocabulary: Optional[Container[str]]) -> bool:
108
+ """Same letters written differently: spacing (ياأيها / يا أيها) or a medial alef (إسحق / إسحاق).
109
+
110
+ The alef variant is tolerated only when the quoted spelling is not itself a word of the corpus (so قتل / قاتل stay a
111
+ real difference)."""
112
+ if "".join(q_tokens) == "".join(s_tokens):
113
+ return True
114
+ if vocabulary is not None and len(q_tokens) == 1 and len(s_tokens) == 1:
115
+ q, s = q_tokens[0], s_tokens[0]
116
+ return q.replace("ا", "") == s.replace("ا", "") and q not in vocabulary
117
+ return False
118
+
119
+
120
+ def align(quote: str, source: str, vocabulary: Optional[Container[str]] = None) -> Dict[str, object]:
121
+ """Align ``quote`` to the best matching region of ``source`` and describe every difference.
122
+
123
+ ``vocabulary``: phonetic-skeleton words of the corpus; enables the medial-alef spelling tolerance."""
124
+ q_pairs, s_pairs = aligned_words(quote), aligned_words(source)
125
+ idgham_pairs = aligned_words(apply_idgham(source)) if len(source) < 20000 else s_pairs
126
+ if len(idgham_pairs) != len(s_pairs):
127
+ idgham_pairs = s_pairs
128
+ q_orig, q_norm = [p[0] for p in q_pairs], [p[1] for p in q_pairs]
129
+ s_orig, s_norm = [p[0] for p in s_pairs], [p[1] for p in s_pairs]
130
+ g_orig = [p[0] for p in idgham_pairs]
131
+
132
+ gap = allowed_gap(len(q_norm))
133
+ lo, hi, matcher = locate_region(q_norm, s_norm, gap)
134
+
135
+ raw_ops = [(tag, i1, i2, j1 + lo, j2 + lo) for tag, i1, i2, j1, j2 in matcher.get_opcodes()]
136
+ # Source words before / after a partial quotation are not errors unless the quotation itself contains them elsewhere
137
+ # (a moved word); drop such edge insertions.
138
+ deleted = {w for tag, i1, i2, _, _ in raw_ops if tag in ("delete", "replace") for w in q_norm[i1:i2]}
139
+ for position in (0, -1):
140
+ if len(raw_ops) > 1 and raw_ops[position][0] == "insert" and not (set(s_norm[raw_ops[position][3]:raw_ops[position][4]]) & deleted):
141
+ raw_ops.pop(position)
142
+
143
+ ops, quote_side, source_side, missing, extra, notes = [], [], [], [], [], []
144
+ mismatches = 0
145
+ orthographic = 0
146
+ for position, (tag, i1, i2, j1, j2) in enumerate(raw_ops):
147
+ source_idx = list(range(j1, j2))
148
+ if tag == "replace" and _orthographic_variant(q_norm[i1:i2], s_norm[j1:j2], vocabulary):
149
+ tag, orthographic = "equal", orthographic + 1 # spelling-only difference, not a different word
150
+ if tag == "replace" and position in (0, len(raw_ops) - 1) and len(raw_ops) > 1 and (j2 - j1) > (i2 - i1):
151
+ # at the edges the slack may have pulled in unrelated source words: keep only words the quotation also uses
152
+ keep = [j for j in source_idx if s_norm[j] in set(q_norm[i1:i2]) | deleted]
153
+ source_idx = keep
154
+ tag = "replace" if keep else "delete"
155
+ ops.append({"op": tag, "span": " ".join(q_orig[i1:i2]), "source": " ".join(s_orig[j] for j in source_idx)})
156
+ if tag == "equal":
157
+ notes += _diacritic_notes(q_orig[i1:i2], s_orig[j1:j2], g_orig[j1:j2])
158
+ else:
159
+ mismatches += max(i2 - i1, len(source_idx))
160
+ quote_side += q_norm[i1:i2]
161
+ source_side += [s_norm[j] for j in source_idx]
162
+ extra += q_orig[i1:i2] if tag in ("delete", "replace") else []
163
+ missing += [s_orig[j] for j in source_idx] if tag in ("insert", "replace") else []
164
+
165
+ matched = sum(i2 - i1 for tag, i1, i2, _, _ in raw_ops if tag == "equal")
166
+ exact = bool(q_norm) and mismatches == 0
167
+ return {
168
+ "word_similarity": round(2 * matched / max(len(q_norm) + (raw_ops[-1][4] - raw_ops[0][3] if raw_ops else 0), 1), 3),
169
+ "word_diff": ops,
170
+ "missing_from_span": missing,
171
+ "extra_in_span": extra,
172
+ "source_excerpt": " ".join(op["source"] for op in ops if op["source"]),
173
+ "exact": exact,
174
+ "mismatches": mismatches,
175
+ "allowed_gap": gap,
176
+ "near": (not exact) and matched > 0 and mismatches <= gap,
177
+ "reordered": bool(quote_side) and sorted(quote_side) == sorted(source_side),
178
+ "diacritic_notes": notes,
179
+ "orthographic_variants": orthographic,
180
+ }
app.py ADDED
@@ -0,0 +1,181 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """التحقق من هلوسة القرآن والحديث وتصحيحها: Gradio interface.
2
+
3
+ python app.py # http://127.0.0.1:7860
4
+
5
+ Mode A verifies pasted text (rule + corpus detection and, when configured, CAMeLBERT, merged silently). Mode B ("ask then verify") sends the question to a single
6
+ pre-configured OpenAI client and verifies the answer; the key comes from the OPENAI_API_KEY environment variable only
7
+ (never from the form, never from the repository). The in-browser page (build_static_space.py) shares these handlers.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ import logging
13
+ import os
14
+ import threading
15
+ from pathlib import Path
16
+ from typing import List, Optional
17
+
18
+ import ui
19
+ from camelbert_adapter import analyze_hybrid, entities_to_spans, query_hosted_model
20
+ from llm_client import LLMError, generate, sanitize_answer
21
+ from verifier import MAX_INPUT_CHARS, IslamicContentVerifier
22
+
23
+ logger = logging.getLogger(__name__)
24
+
25
+ EXAMPLES_PATH = Path(__file__).resolve().parent / "demo" / "examples.json"
26
+
27
+ _pipeline: Optional[IslamicContentVerifier] = None
28
+ _lock = threading.Lock()
29
+
30
+
31
+ def get_pipeline() -> IslamicContentVerifier:
32
+ """Created once. The Quran index loads immediately; the Hadith index loads lazily (see ``warm_in_background``)."""
33
+ global _pipeline
34
+ with _lock:
35
+ if _pipeline is None:
36
+ _pipeline = IslamicContentVerifier()
37
+ return _pipeline
38
+
39
+
40
+ def warm_in_background() -> None:
41
+ threading.Thread(target=lambda: get_pipeline().retriever.warm(), daemon=True).start()
42
+
43
+
44
+ def load_examples(path: Path = EXAMPLES_PATH) -> List[dict]:
45
+ try:
46
+ with open(path, encoding="utf-8") as handle:
47
+ return json.load(handle)
48
+ except (OSError, json.JSONDecodeError):
49
+ logger.exception("Could not load demo examples from %s", path)
50
+ return []
51
+
52
+
53
+ def _analyze(text: str, entities_json: str = "") -> dict:
54
+ """One analysis path: the bundled rule + corpus detector and (when configured) the fine-tuned CAMeLBERT model run
55
+ together and are merged, with no user-facing choice. In the browser the page calls the hosted model and passes its
56
+ entities; locally ``ICV_HF_MODEL`` does the same server-side. Any model failure is silent: the bundled detector alone
57
+ gives the answer."""
58
+ pipeline = get_pipeline()
59
+ model_spans: list = []
60
+ try:
61
+ if entities_json:
62
+ model_spans = entities_to_spans(text, json.loads(entities_json))
63
+ elif os.environ.get("ICV_HF_MODEL", "").strip():
64
+ model_spans = query_hosted_model(text, os.environ["ICV_HF_MODEL"].strip(), os.environ.get("HF_TOKEN", ""))
65
+ except (RuntimeError, ValueError, TypeError):
66
+ logger.warning("Hosted model unavailable; using the bundled detector only")
67
+ model_spans = []
68
+ return analyze_hybrid(pipeline, text, model_spans)
69
+
70
+
71
+ def verify_text(text: str, entities_json: str = "") -> str:
72
+ """Mode A. Never raises: problems become Arabic notices."""
73
+ if not text or not text.strip():
74
+ return ui.render_message("الرجاء إدخال نص للتحقق منه.", "warn")
75
+ try:
76
+ return ui.render_results(_analyze(text, entities_json))
77
+ except ValueError:
78
+ return ui.render_message(f"النص طويل جدًا (الحد الأقصى {MAX_INPUT_CHARS} حرف).", "warn")
79
+ except Exception:
80
+ logger.exception("Verification failed")
81
+ return ui.render_message("حدث خطأ غير متوقع أثناء التحقق.", "bad")
82
+
83
+
84
+ def verify_generated_answer(answer: str, entities_json: str = "") -> str:
85
+ """Verify a model answer and show it above the report (also used by the in-browser page)."""
86
+ try:
87
+ answer = sanitize_answer(answer)
88
+ return ui.render_results(_analyze(answer, entities_json), generated_answer=answer)
89
+ except ValueError:
90
+ return ui.render_message(f"الإجابة طويلة جدًا (الحد الأقصى {MAX_INPUT_CHARS} حرف).", "warn")
91
+ except Exception:
92
+ logger.exception("Verification of the generated answer failed")
93
+ return ui.render_message("تعذّر التحقق من إجابة النموذج.", "bad")
94
+
95
+
96
+ def ask_then_verify(prompt: str) -> str:
97
+ """Mode B: ask the pre-configured model, then verify every quotation in its answer."""
98
+ try:
99
+ answer = generate(None, prompt)
100
+ except LLMError as exc:
101
+ return ui.render_message(str(exc), "warn")
102
+ return verify_generated_answer(answer)
103
+
104
+
105
+ def build_interface():
106
+ import gradio as gr
107
+
108
+ examples = load_examples()
109
+
110
+ def next_example(index: int):
111
+ if not examples:
112
+ return "", 0
113
+ return examples[index % len(examples)]["text"], (index + 1) % len(examples)
114
+
115
+ samples = [e for e in examples if e.get("question")]
116
+
117
+ def next_ask_example(index: int):
118
+ if not samples:
119
+ return "", 0
120
+ return samples[index % len(samples)]["question"], (index + 1) % len(samples)
121
+
122
+ def saved_answer_check(question: str):
123
+ """Verify the saved answer that belongs to the example question (no live model call)."""
124
+ for sample in samples:
125
+ if sample["question"] == question:
126
+ return verify_generated_answer(sample["text"])
127
+ return ui.render_message("اختر مثالًا أولًا.", "warn")
128
+
129
+ theme = gr.themes.Base(primary_hue="emerald", neutral_hue="stone")
130
+ with gr.Blocks(title=ui.APP_TITLE, css=ui.CSS, theme=theme, head=f"<script>{ui.COPY_JS}</script>") as demo:
131
+ gr.HTML(ui.HERO)
132
+ with gr.Tabs():
133
+ with gr.Tab("تحقّق مباشر"):
134
+ example_index = gr.State(0)
135
+ text_input = gr.Textbox(label="النص المراد التحقق منه", lines=9, max_lines=24, placeholder=ui.PLACEHOLDER,
136
+ rtl=True, elem_classes="input-area")
137
+ with gr.Row():
138
+ verify_button = gr.Button("تحقّق من النص", variant="primary", scale=3)
139
+ example_button = gr.Button("جرّب مثالًا", variant="secondary", scale=2)
140
+ results = gr.HTML(elem_classes="results")
141
+ verify_button.click(verify_text, inputs=text_input, outputs=results)
142
+ example_button.click(next_example, inputs=example_index, outputs=[text_input, example_index]).then(
143
+ verify_text, inputs=text_input, outputs=results)
144
+ with gr.Tab("اسأل ثم تحقّق"):
145
+ gr.HTML('<div class="icv"><div class="notice">اكتب سؤالًا، وسيجيب عنه النظام مباشرةً، ثم يفحص كل آية '
146
+ 'وحديث ورد في الإجابة ويعرض الأخطاء والتصحيحات.</div></div>')
147
+ prompt = gr.Textbox(label="سؤالك", lines=3, placeholder=ui.PROMPT_PLACEHOLDER, rtl=True, elem_classes="input-area")
148
+ with gr.Row():
149
+ ask_button = gr.Button("اسأل ثم تحقّق", variant="primary", scale=3)
150
+ ask_example_button = gr.Button("جرّب مثالًا", variant="secondary", scale=2)
151
+ answer_results = gr.HTML(elem_classes="results")
152
+ ask_button.click(ask_then_verify, inputs=prompt, outputs=answer_results)
153
+ ask_example_state = gr.State(0)
154
+ ask_example_button.click(next_ask_example, inputs=ask_example_state, outputs=[prompt, ask_example_state]).then(
155
+ saved_answer_check, inputs=prompt, outputs=answer_results)
156
+ gr.HTML(ui.DISCLAIMER)
157
+ return demo
158
+
159
+
160
+ def _load_dotenv(path: Path = Path(__file__).resolve().parent / ".env") -> None:
161
+ """Minimal ``.env`` reader for local runs (no extra dependency); existing environment variables win."""
162
+ try:
163
+ lines = path.read_text(encoding="utf-8").splitlines()
164
+ except OSError:
165
+ return
166
+ for line in lines:
167
+ name, sep, value = line.strip().partition("=")
168
+ if sep and name and not name.startswith("#") and value.strip():
169
+ os.environ.setdefault(name.strip(), value.strip().strip('"').strip("'"))
170
+
171
+
172
+ def main() -> None:
173
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
174
+ _load_dotenv()
175
+ get_pipeline()
176
+ warm_in_background()
177
+ build_interface().queue().launch(share=os.environ.get("ICV_SHARE") == "1")
178
+
179
+
180
+ if __name__ == "__main__":
181
+ main()
benchmark_format.py ADDED
@@ -0,0 +1,115 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """IslamicEval 2025 (Subtask 1) input / output formats.
2
+
3
+ The pipeline result is converted to the three official submission files:
4
+
5
+ 1A Question_ID <TAB> Span_Start <TAB> Span_End <TAB> Span_Type Span_Type = Ayah | Hadith | No_Spans (end exclusive)
6
+ 1B Sequence_ID <TAB> Label Sequence_ID = <QID>_<annotation number>; Correct | Incorrect
7
+ 1C Sequence_ID <TAB> Correction Sequence_ID = <QID>_<start>_<end>; only spans judged Incorrect;
8
+ 'خطأ' when there is no source / no safe correction
9
+
10
+ The layout follows the competition-format routine of the project's research notebook, which was written against the
11
+ organisers' scoring code. It is **not** re-validated against the live CodaBench scorer here (the organisers' repository
12
+ could not be fetched when this module was written), so compare with the official sample files before a real submission.
13
+
14
+ Command line: python benchmark_format.py responses.xml outputs/ (XML: <Question><ID/><Response/></Question> ...)
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import csv
19
+ import io
20
+ import json
21
+ import re
22
+ import sys
23
+ from pathlib import Path
24
+ from typing import Dict, Iterable, List, Sequence, Tuple
25
+
26
+ NO_SOURCE = "خطأ"
27
+ FILE_NAMES = ("task1A_predictions.tsv", "task1B_predictions.tsv", "task1C_predictions.tsv")
28
+ Row = Tuple
29
+
30
+
31
+ def _clean(value) -> str:
32
+ return str(value).replace("\t", " ").replace("\r", " ").replace("\n", " ").strip()
33
+
34
+
35
+ def competition_rows(result: dict, qid: str) -> Tuple[List[Row], List[Row], List[Row]]:
36
+ """``(rows_1A, rows_1B, rows_1C)`` for one analysed response."""
37
+ a, b, c = [], [], []
38
+ spans = result["spans"]
39
+ if not spans:
40
+ a.append((qid, 0, 0, "No_Spans"))
41
+ for k, span in enumerate(spans, 1):
42
+ a.append((qid, span["start"], span["end"], span["type"]))
43
+ verdict = span["verification"]["verdict"]
44
+ b.append((f"{qid}_{k}", verdict))
45
+ if verdict == "Incorrect":
46
+ correction = span["correction"]["text"] if (span["status"] == "CORRECTED" and span["correction"]) else NO_SOURCE
47
+ c.append((f"{qid}_{span['start']}_{span['end']}", _clean(correction)))
48
+ return a, b, c
49
+
50
+
51
+ def to_tsv(rows: Iterable[Row]) -> str:
52
+ return "".join("\t".join(str(x) for x in row) + "\n" for row in rows)
53
+
54
+
55
+ def official_payload(result: dict, qid: str = "Q1") -> Dict[str, str]:
56
+ """The three files as TSV strings (used by the browser page for downloads)."""
57
+ return {name: to_tsv(rows) for name, rows in zip(FILE_NAMES, competition_rows(result, qid))}
58
+
59
+
60
+ def write_competition_files(all_rows: Sequence[Sequence[Row]], out_dir: str = "outputs") -> List[Path]:
61
+ out = Path(out_dir)
62
+ out.mkdir(parents=True, exist_ok=True)
63
+ paths = []
64
+ for name, rows in zip(FILE_NAMES, all_rows):
65
+ path = out / name
66
+ path.write_text(to_tsv(rows), encoding="utf-8", newline="")
67
+ paths.append(path)
68
+ return paths
69
+
70
+
71
+ def load_responses(xml_path: str) -> Dict[str, str]:
72
+ """``{question id: response}`` from the competition XML."""
73
+ data = Path(xml_path).read_text(encoding="utf-8")
74
+ out = {}
75
+ for block in re.findall(r"<Question>(.*?)</Question>", data, re.S):
76
+ ident, response = re.search(r"<ID>(.*?)</ID>", block, re.S), re.search(r"<Response>(.*?)</Response>", block, re.S)
77
+ if ident and response:
78
+ out[ident.group(1).strip()] = response.group(1)
79
+ return out
80
+
81
+
82
+ def read_tsv(path: str) -> List[dict]:
83
+ """Rows of a TSV with a header line (gold files)."""
84
+ with open(path, encoding="utf-8", newline="") as handle:
85
+ rows = list(csv.reader(handle, delimiter="\t"))
86
+ return [dict(zip(rows[0], row)) for row in rows[1:]] if rows else []
87
+
88
+
89
+ def run_on_texts(pipeline, texts: Dict[str, str], out_dir: str = "outputs") -> Dict[str, dict]:
90
+ """Analyse ``{qid: text}``, write the three official TSVs plus ``full_report.json`` into ``out_dir``."""
91
+ acc: Tuple[List[Row], List[Row], List[Row]] = ([], [], [])
92
+ results = {}
93
+ for qid, text in texts.items():
94
+ result = pipeline.analyze(text)
95
+ results[qid] = result
96
+ for target, part in zip(acc, competition_rows(result, qid)):
97
+ target.extend(part)
98
+ write_competition_files(acc, out_dir)
99
+ (Path(out_dir) / "full_report.json").write_text(json.dumps(results, ensure_ascii=False, indent=2), encoding="utf-8")
100
+ return results
101
+
102
+
103
+ def main(argv: Sequence[str]) -> int:
104
+ if len(argv) < 2:
105
+ print(__doc__)
106
+ return 2
107
+ from verifier import IslamicContentVerifier
108
+
109
+ results = run_on_texts(IslamicContentVerifier(), load_responses(argv[1]), argv[2] if len(argv) > 2 else "outputs")
110
+ print(f"{len(results)} responses written to {argv[2] if len(argv) > 2 else 'outputs'}/")
111
+ return 0
112
+
113
+
114
+ if __name__ == "__main__":
115
+ sys.exit(main(sys.argv))
camelbert_adapter.py ADDED
@@ -0,0 +1,103 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Adapter for a fine-tuned CAMeLBERT-MSA span detector (Subtask 1A) hosted on Hugging Face, plus a simulation mode.
2
+
3
+ Two ways to get detection spans into the verification pipeline:
4
+
5
+ * ``entities_to_spans`` turns the output of a Hugging Face *token-classification* call (aggregated entities, or raw
6
+ ``B-Ayah`` / ``I-Ayah`` / ``B-Hadith`` / ``I-Hadith`` tokens) into ``[{label, start, end, score}]``.
7
+ The browser page calls the hosted model and hands the entities to ``verify_with_entities``.
8
+ * ``simulate_spans`` a stand-in used when no hosted model is configured or reachable: the bundled hybrid detector
9
+ produces the spans, and they are labelled as simulated so nobody mistakes them for model output.
10
+
11
+ No model weights live in this repository. Train with ``research/train_detector.py`` (``--model CAMeL-Lab/bert-base-arabic-camelbert-msa``),
12
+ push the result to the Hub and set its id in the page configuration (see docs/DEPLOYMENT.md).
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import json
17
+ import re
18
+ import urllib.error
19
+ import urllib.request
20
+ from typing import Dict, Iterable, List, Optional
21
+
22
+ HF_ENDPOINT = "https://router.huggingface.co/hf-inference/models/{model}" # configurable; not verified against the live service
23
+
24
+ LABEL_ALIASES = {"ayah": "Ayah", "quran": "Ayah", "label_1": "Ayah", "label_2": "Ayah",
25
+ "hadith": "Hadith", "label_3": "Hadith", "label_4": "Hadith"}
26
+ DEFAULT_MIN_SCORE = 0.5
27
+
28
+
29
+ def _label_of(raw: str) -> Optional[str]:
30
+ name = re.sub(r"^[BI]-", "", str(raw or ""), flags=re.I).strip().lower()
31
+ return LABEL_ALIASES.get(name)
32
+
33
+
34
+ def entities_to_spans(text: str, entities: Iterable[dict], min_score: float = DEFAULT_MIN_SCORE,
35
+ max_gap: int = 1) -> List[dict]:
36
+ """Merge Hugging Face token-classification output into character spans.
37
+
38
+ Works with ``aggregation_strategy="simple"`` output (``entity_group``) and with raw output (``entity`` = ``B-Ayah`` ...).
39
+ Pieces of the same label separated by at most ``max_gap`` characters are joined; low-score pieces are dropped."""
40
+ pieces = []
41
+ for item in entities or []:
42
+ label = _label_of(item.get("entity_group") or item.get("entity"))
43
+ start, end = item.get("start"), item.get("end")
44
+ if label is None or start is None or end is None or end <= start:
45
+ continue
46
+ pieces.append({"label": label, "start": int(start), "end": int(end), "score": float(item.get("score", 1.0)),
47
+ "begin": str(item.get("entity", "")).upper().startswith("B-")})
48
+ pieces.sort(key=lambda p: (p["start"], p["end"]))
49
+ merged: List[dict] = []
50
+ for piece in pieces:
51
+ last = merged[-1] if merged else None
52
+ if last and last["label"] == piece["label"] and not piece["begin"] and piece["start"] - last["end"] <= max_gap:
53
+ last["end"] = max(last["end"], piece["end"])
54
+ last["scores"].append(piece["score"])
55
+ else:
56
+ merged.append({"label": piece["label"], "start": piece["start"], "end": piece["end"], "scores": [piece["score"]]})
57
+ spans = []
58
+ for item in merged:
59
+ start, end = item["start"], item["end"]
60
+ while start < end and text[start].isspace():
61
+ start += 1
62
+ while end > start and text[end - 1].isspace():
63
+ end -= 1
64
+ score = sum(item["scores"]) / len(item["scores"])
65
+ if end > start and score >= min_score:
66
+ spans.append({"label": item["label"], "start": start, "end": end, "score": round(score, 4)})
67
+ return spans
68
+
69
+
70
+ def simulate_spans(pipeline, text: str) -> List[dict]:
71
+ """Spans from the bundled hybrid detector, flagged as simulated (``score`` is ``None``)."""
72
+ return [{"label": s.label, "start": s.start, "end": s.end, "score": None} for s in pipeline.detect(text)]
73
+
74
+
75
+ def analyze_with_spans(pipeline, text: str, spans: List[dict], engine: str = "camelbert") -> dict:
76
+ """Run verification + correction on externally detected spans and record which engine produced them."""
77
+ result = pipeline.analyze_spans(text, [{"label": s["label"], "start": s["start"], "end": s["end"]} for s in spans])
78
+ scores: Dict[tuple, Optional[float]] = {(s["start"], s["end"]): s.get("score") for s in spans}
79
+ for report in result["spans"]:
80
+ score = scores.get((report["start"], report["end"]))
81
+ report["detection"] = {"backend": engine, "confidence": None if score is None else round(score, 4)}
82
+ result["detector"] = engine
83
+ return result
84
+
85
+
86
+ def query_hosted_model(text: str, model: str, token: str = "", endpoint: str = HF_ENDPOINT, timeout: float = 30.0) -> List[dict]:
87
+ """Call a hosted token-classification model (server-side use: the local Gradio app and tests with a mock server).
88
+
89
+ Returns ``entities_to_spans`` output. Raises ``RuntimeError`` with a short message on any failure so the caller can fall
90
+ back to the simulation."""
91
+ headers = {"Content-Type": "application/json"}
92
+ if token:
93
+ headers["Authorization"] = f"Bearer {token}"
94
+ body = json.dumps({"inputs": text, "parameters": {"aggregation_strategy": "simple"}}).encode("utf-8")
95
+ request = urllib.request.Request(endpoint.format(model=model), data=body, headers=headers, method="POST")
96
+ try:
97
+ with urllib.request.urlopen(request, timeout=timeout) as response:
98
+ payload = json.loads(response.read().decode("utf-8"))
99
+ except (urllib.error.URLError, TimeoutError, json.JSONDecodeError) as exc:
100
+ raise RuntimeError("hosted model unavailable") from exc
101
+ if not isinstance(payload, list):
102
+ raise RuntimeError("unexpected response from the hosted model")
103
+ return entities_to_spans(text, payload)
config.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "askEndpoint": "https://icv-ask-proxy.ghada-islamic-verifier-2026.workers.dev",
3
+ "hf": {
4
+ "model": "",
5
+ "endpoint": "https://router.huggingface.co/hf-inference/models/{model}"
6
+ }
7
+ }
detector.py ADDED
@@ -0,0 +1,178 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Quotation detection (Subtask 1A): finds Quran / Hadith quotations in a text and their boundaries.
2
+
3
+ The detector is rule-based: quotation marks and brackets mark candidate segments. The *type* of a segment (Ayah /
4
+ Hadith) is decided by the reference corpora themselves (word coverage against the Quran and the six Hadith books).
5
+ Introductory phrases such as "قال الله تعالى" or "قال رسول الله ﷺ" are only a tie-breaker / fallback hint, never a
6
+ requirement: a quotation is found and typed even when no such phrase precedes it. No model weights are loaded."""
7
+ from __future__ import annotations
8
+
9
+ import re
10
+ from dataclasses import dataclass
11
+ from typing import List, Optional
12
+
13
+ from normalization import normalize_for_matching, normalize_lenient, normalize_strict, tokenize
14
+
15
+
16
+ # --------------------------------------------------------------------------------------------------------------
17
+ @dataclass
18
+ class DetectedSpan:
19
+ start: int
20
+ end: int # exclusive
21
+ label: str # 'Ayah' | 'Hadith'
22
+ confidence: Optional[float] # None for the rule backend
23
+ source: str # 'rules' | 'given'
24
+ text: str = ""
25
+ hint: Optional[str] = None # type suggested by an introductory phrase, if any (may disagree with ``label``)
26
+
27
+
28
+ QUOTE_CHARS = " \t\r\n\"“”«»﴿﴾{}()[]"
29
+
30
+
31
+ def trim_span(text: str, start: int, end: int):
32
+ """Drop whitespace and quotation marks at both edges (gold spans exclude the quote marks)."""
33
+ while start < end and text[start] in QUOTE_CHARS:
34
+ start += 1
35
+ while end > start and text[end - 1] in QUOTE_CHARS + ".،,؛:":
36
+ end -= 1
37
+ return start, end
38
+
39
+
40
+ AYAH_TRIGGERS = [
41
+ "قال الله", "قوله تعالى", "قال تعالى", "يقول الله", "يقول تعالى", "قال سبحانه", "قوله سبحانه", "في كتابه",
42
+ "سورة", "الآية", "الاية", "الآيات", "القرآن", "القران", "كتاب الله", "عز وجل", "جل جلاله", "فقال تعالى",
43
+ "ذكر الله", "﴿",
44
+ ]
45
+ HADITH_TRIGGERS = [
46
+ "رسول الله", "النبي", "صلى الله عليه وسلم", "ﷺ", "عليه الصلاة والسلام", "حديث", "رواه", "روى", "متفق عليه",
47
+ "الحديث", "فقال", "قال ص", "صلى الله عليه", "وسلم",
48
+ ]
49
+ _FORMULA_WORDS = {
50
+ normalize_for_matching(w)
51
+ for w in "قال قالت رسول الله صلى عليه وسلم النبي تعالى سبحانه عز وجل فقال يقول الكريم الشريف الحديث الآية روى رواه عن أن أنه البخاري ومسلم".split()
52
+ }
53
+ _BRACKET_PAIRS = [("“", "”"), ("«", "»"), ("﴿", "﴾"), ("{", "}"), ("(", ")"), ("[", "]")]
54
+
55
+
56
+ class RuleDetector:
57
+ """Quotation-mark and trigger-phrase detector with an optional corpus lookup (no training, no GPU)."""
58
+
59
+ def __init__(self, retriever=None, min_words: int = 3, context_chars: int = 110,
60
+ min_corpus_cov: float = 0.6, decouple_triggers: bool = True, quoted_min_cov: float = 0.5,
61
+ quoted_min_cov_hadith: float = 0.75) -> None:
62
+ """``decouple_triggers``: type a delimited segment from the corpora first and use introductory phrases only as a
63
+ hint (needs a retriever). ``quoted_min_cov``: lower coverage bar for text the author explicitly delimited."""
64
+ self.kb, self.min_words, self.context_chars, self.min_corpus_cov = retriever, min_words, context_chars, min_corpus_cov
65
+ self.decouple_triggers = decouple_triggers and retriever is not None
66
+ self.quoted_min_cov, self.quoted_min_cov_hadith = quoted_min_cov, quoted_min_cov_hadith
67
+
68
+ @staticmethod
69
+ def _segments(text: str):
70
+ segments = []
71
+ quote_positions = [m.start() for m in re.finditer('"', text)]
72
+ if len(quote_positions) % 2 == 0:
73
+ pairs = zip(quote_positions[0::2], quote_positions[1::2]) # opening/closing pairs
74
+ else: # a stray quote: fall back to every consecutive pair
75
+ pairs = zip(quote_positions, quote_positions[1:])
76
+ for a, b in pairs:
77
+ segments.append((a + 1, b))
78
+ for opener, closer in _BRACKET_PAIRS:
79
+ for m in re.finditer(re.escape(opener) + r"(.*?)" + re.escape(closer), text, re.S):
80
+ segments.append((m.start(1), m.end(1)))
81
+ return segments
82
+
83
+ @staticmethod
84
+ def _trigger_type(context: str) -> Optional[str]:
85
+ best_end, best_label = -1, None
86
+ for label, triggers in (("Ayah", AYAH_TRIGGERS), ("Hadith", HADITH_TRIGGERS)):
87
+ for trigger in triggers:
88
+ pos = context.rfind(trigger)
89
+ if pos >= 0 and pos + len(trigger) > best_end:
90
+ best_end, best_label = pos + len(trigger), label
91
+ return best_label
92
+
93
+ def _corpus_coverage(self, span: str):
94
+ """Highest word coverage of the span by any top Quran ayah / Hadith candidate."""
95
+ if self.kb is None:
96
+ return 0.0, 0.0
97
+ quran_words = set(tokenize(normalize_strict(span)))
98
+ if not quran_words:
99
+ return 0.0, 0.0
100
+ quran_cov = self._quran_window_coverage(quran_words, span)
101
+ hadith_words = set(tokenize(normalize_lenient(span)))
102
+ hadith_cov = max(
103
+ (len(hadith_words & set(tokenize(normalize_lenient(c["text"])))) / len(hadith_words)
104
+ for c in self.kb.search_hadith(span, top_k=5)),
105
+ default=0.0,
106
+ ) if hadith_words else 0.0
107
+ return quran_cov, hadith_cov
108
+
109
+ def _quran_window_coverage(self, quran_words: set, span: str) -> float:
110
+ """Best word coverage of the span by a single ayah or by 2-3 consecutive ayahs around a top candidate (a quotation
111
+ often runs across an ayah boundary, and no single ayah then covers it)."""
112
+ kb, best = self.kb, 0.0
113
+ for cand in kb.search_quran_ayahs(span, top_k=5):
114
+ surah = kb.quran_by_surah.get(cand["surah_id"], {})
115
+ for first in range(cand["ayah_id"] - 2, cand["ayah_id"] + 1):
116
+ words: set = set()
117
+ for length in (1, 2, 3):
118
+ idx = surah.get(first + length - 1)
119
+ if idx is None:
120
+ break
121
+ ayah_words = set(tokenize(normalize_strict(kb.quran[idx]["text"])))
122
+ if length > 1 and not any(len(w) >= 4 for w in quran_words & ayah_words):
123
+ break # every ayah of a multi-ayah window must contribute a real (not particle-like) word of the quotation
124
+ words |= ayah_words
125
+ if first <= cand["ayah_id"] <= first + length - 1:
126
+ best = max(best, len(quran_words & words) / len(quran_words))
127
+ return best
128
+
129
+ def _label_from_corpus(self, n_words: int, trigger: Optional[str], quran_cov: float, hadith_cov: float) -> Optional[str]:
130
+ """Corpus-first typing. The introductory phrase only breaks near-ties or types an altered quotation whose
131
+ coverage is too low for the corpus to decide on its own."""
132
+ quran_ok = quran_cov >= self.quoted_min_cov
133
+ hadith_ok = hadith_cov >= self.quoted_min_cov_hadith # Hadith records are long: stricter, ordinary prose overlaps them
134
+ if (quran_ok or hadith_ok) and n_words >= 4:
135
+ if trigger and ((quran_ok if trigger == "Ayah" else hadith_ok)) and abs(quran_cov - hadith_cov) < 0.25:
136
+ return trigger
137
+ if quran_ok and hadith_ok:
138
+ return "Ayah" if quran_cov >= hadith_cov else "Hadith"
139
+ return "Ayah" if quran_ok else "Hadith"
140
+ return trigger # may be None: a delimited segment that matches nothing and has no hint is not reported
141
+
142
+ def detect(self, text: str) -> List[DetectedSpan]:
143
+ candidates = []
144
+ for start, end in self._segments(text):
145
+ start, end = trim_span(text, start, end)
146
+ if end <= start:
147
+ continue
148
+ inner = text[start:end]
149
+ words = [w for w in normalize_for_matching(inner).split() if w]
150
+ if len(words) < 2 or len(inner) > 3000:
151
+ continue
152
+ trigger = self._trigger_type(text[max(0, start - self.context_chars):start])
153
+ if len(words) < self.min_words and not trigger: # a very short quote is only taken after an introductory phrase
154
+ continue
155
+ if sum(w in _FORMULA_WORDS for w in words) / len(words) >= 0.6:
156
+ continue
157
+ quran_cov, hadith_cov = self._corpus_coverage(inner)
158
+ label = None
159
+ if self.decouple_triggers:
160
+ label = self._label_from_corpus(len(words), trigger, quran_cov, hadith_cov)
161
+ elif trigger:
162
+ label = trigger
163
+ other, mine = (hadith_cov, quran_cov) if trigger == "Ayah" else (quran_cov, hadith_cov)
164
+ if other >= 0.8 and mine < 0.5:
165
+ label = "Hadith" if trigger == "Ayah" else "Ayah"
166
+ elif max(quran_cov, hadith_cov) >= self.min_corpus_cov and len(words) >= 4:
167
+ label = "Ayah" if quran_cov >= hadith_cov else "Hadith"
168
+ if label is None:
169
+ continue
170
+ candidates.append((bool(trigger), max(quran_cov, hadith_cov), end - start, start, end, label, trigger))
171
+
172
+ candidates.sort(key=lambda c: (c[0], c[1], c[2]), reverse=True) # trigger first, then corpus match, then length
173
+ taken = []
174
+ for _, _, _, start, end, label, hint in candidates:
175
+ if all(end <= t_start or start >= t_end for t_start, t_end, _, _ in taken):
176
+ taken.append((start, end, label, hint))
177
+ taken.sort()
178
+ return [DetectedSpan(s, e, label, None, "rules", text[s:e], hint) for s, e, label, hint in taken]
examples.json ADDED
@@ -0,0 +1,227 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "id": "scenario_all_correct_ayahs",
4
+ "title": "كل الآيات صحيحة",
5
+ "question": "ما فضل الصبر واليسر بعد العسر وغاية خلق الإنسان في القرآن؟",
6
+ "scenario": "سيناريو للتجربة: كل الآيات صحيحة",
7
+ "text": "ذكر القرآن حكمة خلق الإنسان وبشّر بعد الشدة بالفرج. قال الله تعالى: \"وَمَا خَلَقْتُ الْجِنَّ وَالْإِنسَ إِلَّا لِيَعْبُدُونِ\" [الذاريات: 56]. وقال سبحانه: \"فَإِنَّ مَعَ الْعُسْرِ يُسْرًا ۝ إِنَّ مَعَ الْعُسْرِ يُسْرًا\" [الشرح: 5-6]. وقال عز وجل: \"وَاسْتَعِينُوا بِالصَّبْرِ وَالصَّلَاةِ ۚ وَإِنَّهَا لَكَبِيرَةٌ إِلَّا عَلَى الْخَاشِعِينَ\" [البقرة: 45].",
8
+ "built_from": "كُتب من نصوص القرآن والحديث المضمّنة؛ الأخطاء مصطنعة لغرض التجربة",
9
+ "observed_statuses": [
10
+ "VERIFIED",
11
+ "VERIFIED",
12
+ "VERIFIED"
13
+ ]
14
+ },
15
+ {
16
+ "id": "scenario_all_wrong_ayahs",
17
+ "title": "كل الآيات خاطئة",
18
+ "question": "ما الآيات الواردة في العبادة والفرج والصبر؟",
19
+ "scenario": "سيناريو للتجربة: كل الآيات خاطئة",
20
+ "text": "وردت في القرآن آيات في هذا الباب. قال الله تعالى: \"وَمَا خَلَقْتُ الْجِنَّ وَالْإِنسَ إِلَّا لِيَأْكُلُونِ\". وقال تعالى: \"فَإِنَّ مَعَ الْعُسْرِ يُسْرًا ۝ إِنَّ بَعْدَ الْعُسْرِ سُرُورًا\". وقال سبحانه: \"وَاسْتَعِينُوا بِالصَّبْرِ وَالصَّلَاةِ ۚ وَإِنَّهَا لَسَهْلَةٌ عَلَى الْخَاشِعِينَ\".",
21
+ "built_from": "كُتب من نصوص القرآن والحديث المضمّنة؛ الأخطاء مصطنعة لغرض التجربة",
22
+ "observed_statuses": [
23
+ "CORRECTED",
24
+ "CORRECTED",
25
+ "CORRECTED"
26
+ ]
27
+ },
28
+ {
29
+ "id": "scenario_all_correct_hadiths",
30
+ "title": "كل الأحاديث صحيحة",
31
+ "question": "ما الأحاديث الصحيحة في النية والطهارة وحفظ اللسان؟",
32
+ "scenario": "سيناريو للتجربة: كل الأحاديث صحيحة",
33
+ "text": "من الأحاديث الصحيحة في هذا الباب: قال رسول الله ﷺ: \"إِنَّمَا الْأَعْمَالُ بِالنِّيَّاتِ ، وَإِنَّمَا لِكُلِّ امْرِئٍ مَا نَوَى\" رواه البخاري. وقال ﷺ: \"الطُّهُورُ شَطْرُ الْإِيمَانِ\" رواه مسلم. وقال ﷺ: \"مَنْ كَانَ يُؤْمِنُ بِاللَّهِ وَالْيَوْمِ الْآخِرِ فَلْيَقُلْ خَيْرًا أَوْ لِيَصْمُتْ\" متفق عليه.",
34
+ "built_from": "كُتب من نصوص القرآن والحديث المضمّنة؛ الأخطاء مصطنعة لغرض التجربة",
35
+ "observed_statuses": [
36
+ "VERIFIED",
37
+ "VERIFIED",
38
+ "VERIFIED"
39
+ ]
40
+ },
41
+ {
42
+ "id": "scenario_all_wrong_hadiths",
43
+ "title": "كل الأحاديث خاطئة",
44
+ "question": "هل ورد حديث في فضل الذكر والصدقة وبر الوالدين؟",
45
+ "scenario": "سيناريو للتجربة: كل الأحاديث خاطئة",
46
+ "text": "يُروى في هذا الباب: قال رسول الله ﷺ: \"مَنْ لَبِسَ الْأَخْضَرَ يَوْمَ الْجُمُعَةِ زَادَ اللَّهُ فِي رِزْقِهِ عَشْرَةَ أَضْعَافٍ\". وقال ﷺ: \"مَنْ أَكَلَ الْعَسَلَ مَعَ الثُّومِ سَبْعَةَ أَيَّامٍ شُفِيَ مِنْ كُلِّ دَاءٍ بِإِذْنِ اللَّهِ\". وقال ﷺ: \"مَنْ دَخَلَ بَيْتَهُ بِرِجْلِهِ الْيُسْرَى انْتَقَصَ عُمُرُهُ سَنَةً كَامِلَةً\".",
47
+ "built_from": "كُتب من نصوص القرآن والحديث المضمّنة؛ الأخطاء مصطنعة لغرض التجربة",
48
+ "observed_statuses": [
49
+ "UNSUPPORTED",
50
+ "UNSUPPORTED",
51
+ "UNSUPPORTED"
52
+ ]
53
+ },
54
+ {
55
+ "id": "scenario_wrong_ayahs_correct_hadiths",
56
+ "title": "آيات خاطئة وأحاديث صحيحة",
57
+ "question": "ما الدليل على فضل الطهارة والنية من القرآن والسنة؟",
58
+ "scenario": "سيناريو للتجربة: آيات خاطئة وأحاديث صحيحة",
59
+ "text": "قال الله تعالى: \"إِنَّ اللَّهَ يُحِبُّ التَّوَّابِينَ وَيُحِبُّ الْمُسْرِفِينَ\". وقال تعالى: \"وَثِيَابَكَ فَطَهِّرْ وَالرُّجْزَ فَاعْبُدْ\". وأما السنة فقال رسول الله ﷺ: \"الطُّهُورُ شَطْرُ الْإِيمَانِ\" رواه مسلم. وقال ﷺ: \"إِنَّمَا الْأَعْمَالُ بِالنِّيَّاتِ ، وَإِنَّمَا لِكُلِّ امْرِئٍ مَا نَوَى\" رواه البخاري.",
60
+ "built_from": "كُتب من نصوص القرآن والحديث المضمّنة؛ الأخطاء مصطنعة لغرض التجربة",
61
+ "observed_statuses": [
62
+ "CORRECTED",
63
+ "CORRECTED",
64
+ "VERIFIED",
65
+ "VERIFIED"
66
+ ]
67
+ },
68
+ {
69
+ "id": "scenario_correct_ayahs_wrong_hadiths",
70
+ "title": "آيات صحيحة وأحاديث خاطئة",
71
+ "question": "ما الدليل على فضل الصدق وبر الوالدين والصلاة؟",
72
+ "scenario": "سيناريو للتجربة: آيات صحيحة وأحاديث خاطئة",
73
+ "text": "قال الله تعالى: \"يَا أَيُّهَا الَّذِينَ آمَنُوا اتَّقُوا اللَّهَ وَكُونُوا مَعَ الصَّادِقِينَ\" [التوبة: 119]. وقال تعالى: \"وَقَضَىٰ رَبُّكَ أَلَّا تَعْبُدُوا إِلَّا إِيَّاهُ وَبِالْوَالِدَيْنِ إِحْسَانًا\" [الإسراء: 23]. وقال رسول الله ﷺ: \"مَنْ سَافَرَ يَوْمَ الْأَرْبِعَاءِ لَمْ يَرْجِعْ إِلَّا بِالْخَسَارَةِ وَالْهَمِّ\". وقال ﷺ: \"مَنْ دَخَلَ بَيْتَهُ بِرِجْلِهِ الْيُسْرَى انْتَقَصَ عُمُرُهُ سَنَةً كَامِلَةً\".",
74
+ "built_from": "كُتب من نصوص القرآن والحديث المضمّنة؛ الأخطاء مصطنعة لغرض التجربة",
75
+ "observed_statuses": [
76
+ "VERIFIED",
77
+ "VERIFIED",
78
+ "UNSUPPORTED",
79
+ "UNSUPPORTED"
80
+ ]
81
+ },
82
+ {
83
+ "id": "correct_ayah",
84
+ "title": "آية صحيحة",
85
+ "scenario": "correct",
86
+ "text": "قال الله تعالى: \"يَا أَيُّهَا الَّذِينَ آمَنُوا اسْتَعِينُوا بِالصَّبْرِ وَالصَّلَاةِ ۚ إِنَّ اللَّهَ مَعَ الصَّابِرِينَ\". وفي الآية بشارة لمن صبر.",
87
+ "built_from": [
88
+ "سورة البقرة 153"
89
+ ],
90
+ "observed_statuses": [
91
+ "VERIFIED"
92
+ ]
93
+ },
94
+ {
95
+ "id": "correct_hadith",
96
+ "title": "حديث صحيح",
97
+ "scenario": "correct",
98
+ "text": "وقال رسول الله ﷺ: \"إِنَّمَا الْأَعْمَالُ بِالنِّيَّاتِ ، وَإِنَّمَا لِكُلِّ امْرِئٍ مَا نَوَى ، فَمَنْ كَانَتْ هِجْرَتُهُ إِلَى دُنْيَا يُصِيبُهَا ، أَوْ إِلَى امْرَأَةٍ يَنْكِحُهَا ، فَهِجْرَتُهُ إِلَى مَا هَاجَرَ إِلَيْهِ\". ويستفاد منه أهمية النية.",
99
+ "built_from": [
100
+ "حديث رقم 5"
101
+ ],
102
+ "observed_statuses": [
103
+ "VERIFIED"
104
+ ]
105
+ },
106
+ {
107
+ "id": "substituted_word",
108
+ "title": "كلمة مستبدلة: «ثم استقم» بدل «فاستقم»",
109
+ "scenario": "altered",
110
+ "text": "قال الله تعالى: \"ثُمَّ اسْتَقِمْ كَمَا أُمِرْتَ وَمَنْ تَابَ مَعَكَ\".",
111
+ "built_from": [
112
+ "سورة هود 112 (فَاسْتَقِمْ ← ثُمَّ اسْتَقِمْ)"
113
+ ],
114
+ "observed_statuses": [
115
+ "CORRECTED"
116
+ ]
117
+ },
118
+ {
119
+ "id": "swapped_order",
120
+ "title": "تبديل ترتيب الكلمات",
121
+ "scenario": "altered",
122
+ "text": "قال الله تعالى: \"كَمَا أُمِرْتَ فَاسْتَقِمْ وَمَنْ تَابَ مَعَكَ\".",
123
+ "built_from": [
124
+ "سورة هود 112 (تقديم وتأخير)"
125
+ ],
126
+ "observed_statuses": [
127
+ "CORRECTED"
128
+ ]
129
+ },
130
+ {
131
+ "id": "wrong_diacritic",
132
+ "title": "تشكيل خاطئ لكلمة",
133
+ "scenario": "altered",
134
+ "text": "قال الله تعالى: \"فَاسْتَقَمْ كَمَا أُمِرْتَ وَمَنْ تَابَ مَعَكَ\".",
135
+ "built_from": [
136
+ "سورة هود 112 (فَاسْتَقِمْ ← فَاسْتَقَمْ)"
137
+ ],
138
+ "observed_statuses": [
139
+ "VERIFIED"
140
+ ]
141
+ },
142
+ {
143
+ "id": "missing_diacritics",
144
+ "title": "بدون تشكيل (مقبول)",
145
+ "scenario": "correct",
146
+ "text": "قال الله تعالى: \"فاستقم كما أمرت ومن تاب معك\".",
147
+ "built_from": [
148
+ "سورة هود 112 بلا تشكيل"
149
+ ],
150
+ "observed_statuses": [
151
+ "VERIFIED"
152
+ ]
153
+ },
154
+ {
155
+ "id": "partial_citation",
156
+ "title": "اقتباس جزئي من آية",
157
+ "scenario": "partial",
158
+ "text": "قال الله تعالى: \"۞ وَقَضَىٰ رَبُّكَ أَلَّا تَعْبُدُوا إِلَّا\".",
159
+ "built_from": [
160
+ "سورة الإسراء 23 (أول ست كلمات)"
161
+ ],
162
+ "observed_statuses": [
163
+ "VERIFIED"
164
+ ]
165
+ },
166
+ {
167
+ "id": "missing_word",
168
+ "title": "كلمة ناقصة",
169
+ "scenario": "altered",
170
+ "text": "قال الله تعالى: \"يَا أَيُّهَا آمَنُوا اسْتَعِينُوا بِالصَّبْرِ وَالصَّلَاةِ ۚ إِنَّ ال��َّهَ مَعَ الصَّابِرِينَ\".",
171
+ "built_from": [
172
+ "سورة البقرة 153 (حُذفت الكلمة الثالثة)"
173
+ ],
174
+ "observed_statuses": [
175
+ "CORRECTED"
176
+ ]
177
+ },
178
+ {
179
+ "id": "unannounced_quote",
180
+ "title": "اقتباس غير معلَن بلا علامات تنصيص",
181
+ "scenario": "unannounced",
182
+ "text": "وهنا نتذكر أن العبد عليه أن يثبت وألا يتراجع، فاستقم كما أمرت ومن تاب معك ولا تطغوا، وهذا من أعظم ما يعين على الطاعة.",
183
+ "built_from": [
184
+ "سورة هود 112 داخل الجملة"
185
+ ],
186
+ "observed_statuses": [
187
+ "VERIFIED"
188
+ ]
189
+ },
190
+ {
191
+ "id": "uncertain_hadith",
192
+ "title": "حديث مركّب من حديثين",
193
+ "scenario": "uncertain",
194
+ "text": "روى البخاري أن النبي ﷺ قال: \"إِنَّمَا الْأَعْمَالُ بِالنِّيَّاتِ ، وَإِنَّمَا لِكُلِّ امْرِئٍ مَا نَوَى ، فَمَنْ كَانَتْ هِجْرَتُهُ أَهْلَ الْأَرْضِ يَرْحَمْكُمْ مَنْ فِي السَّمَاءِ\".",
195
+ "built_from": [
196
+ "حديث رقم 5 (النصف الأول)",
197
+ "حديث رقم 95766 (النصف الثاني)"
198
+ ],
199
+ "observed_statuses": [
200
+ "HUMAN_REVIEW"
201
+ ]
202
+ },
203
+ {
204
+ "id": "no_quotation",
205
+ "title": "نص بلا اقتباسات",
206
+ "scenario": "none",
207
+ "text": "الصبر من أهم الصفات التي ينبغي أن يتحلى بها الإنسان في حياته، ويساعده على مواجهة المصاعب بثبات وهدوء.",
208
+ "built_from": [],
209
+ "observed_statuses": []
210
+ },
211
+ {
212
+ "id": "mixed_response",
213
+ "title": "رد مختلط",
214
+ "scenario": "mixed",
215
+ "text": "بخصوص بر الوالدين، قال الله تعالى: \"۞ وَقَضَىٰ رَبُّكَ أَلَّا تَعْبُدُوا إِلَّا إِيَّاهُ وَبِالْوَالِدَيْنِ إِحْسَانًا ۚ إِمَّا يَبْلُغَنَّ عِنْدَكَ الْكِبَرَ أَحَدُهُمَا أَوْ كِلَاهُمَا فَلَا تَقُلْ لَهُمَا أُفٍّ وَلَا تَنْهَرْهُمَا وَقُلْ لَهُمَا قَوْلًا كَرِيمًا\". وقال رسول الله ﷺ: \"الرَّاحِمُونَ يَرْحَمُهُمُ الرَّحْمَنُ ، ارْحَمُوا أَهْلَ الْأَرْضِ يَرْحَمْكُمْ مَنْ فِي السَّمَاءِ\". وقال سبحانه: \"يَا أَيُّهَا آمَنُوا اسْتَعِينُوا بِالصَّبْرِ وَالصَّلَاةِ ۚ إِنَّ اللَّهَ مَعَ الصَّابِرِينَ\". وهذا من مكارم الأخلاق.",
216
+ "built_from": [
217
+ "سورة الإسراء 23",
218
+ "حديث رقم 95766",
219
+ "سورة البقرة 153 (كلمة ناقصة)"
220
+ ],
221
+ "observed_statuses": [
222
+ "VERIFIED",
223
+ "VERIFIED",
224
+ "CORRECTED"
225
+ ]
226
+ }
227
+ ]
hadith.idx.gz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6f5792317d7577220e0bd4b1aab84be1ed665c17f974ab7ba1cd8920f34c9955
3
+ size 13597012
hadith.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f1f514d7a2de2d4179d9588a861cedf07e81630886444f37d0c08c2cc777075a
3
+ size 48498788
idgham.py ADDED
@@ -0,0 +1,71 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Idgham rendering of the Quran text in the published mushaf convention (used by the Subtask 1C gold corrections).
2
+
3
+ The flat Quran text writes assimilated letters without a shadda; the mushaf marks the assimilation. Covers noon
4
+ sakinah / tanween before ن م ر ل, meem sakinah before م and lam sakinah before ل ر, also across ayah-number markers
5
+ such as ``(20)``. The و / ي cases are excluded on purpose because the mushaf convention is inconsistent there."""
6
+ from __future__ import annotations
7
+
8
+ import re
9
+
10
+ _MARKER = re.compile(r"^\(\d+\)$")
11
+
12
+ _SUKUN, _SHADDA, _FATHATAN = "\u0652", "\u0651", "\u064B"
13
+ _TANWEEN = set("\u064B\u064C\u064D")
14
+ _IDGHAM_AFTER_NOON = set("نمرل")
15
+ _IDGHAM_AFTER_LAM = set("لر")
16
+ _DIACRITIC_CHARS = set(
17
+ "\u0610\u0611\u0612\u0613\u0614\u0615\u0616\u0617\u0618\u0619\u061A"
18
+ "\u064B\u064C\u064D\u064E\u064F\u0650\u0651\u0652\u0670"
19
+ "\u06D6\u06D7\u06D8\u06D9\u06DA\u06DB\u06DC\u06DF\u06E0\u06E1\u06E2\u06E3\u06E4"
20
+ "\u06E7\u06E8\u06EA\u06EB\u06EC\u06ED"
21
+ )
22
+
23
+
24
+ def _base_letters(word: str) -> str:
25
+ return "".join(c for c in word if c not in _DIACRITIC_CHARS)
26
+
27
+
28
+ def _insert_shadda(word: str) -> str:
29
+ return word if not word else word[0] + _SHADDA + word[1:]
30
+
31
+
32
+ def _ends_with_tanween(word: str) -> bool:
33
+ if not word:
34
+ return False
35
+ if word[-1] in _TANWEEN:
36
+ return True
37
+ return len(word) >= 2 and word[-1] in "اى" and word[-2] == _FATHATAN
38
+
39
+
40
+ def apply_idgham(text: str, extended: bool = True) -> str:
41
+ """Convert the flat Quran text into the mushaf rendering that marks assimilation with a shadda.
42
+
43
+ Covers noon sakinah / tanween before ن م ر ل, meem sakinah before م and lam sakinah before ل ر, also across
44
+ ayah-number markers such as ``(20)``. The و / ي cases are excluded on purpose because the mushaf convention
45
+ is inconsistent there."""
46
+ words = text.split(" ")
47
+ noon_set = _IDGHAM_AFTER_NOON if extended else set("نم")
48
+ i = 0
49
+ while i < len(words):
50
+ word = words[i]
51
+ if _MARKER.match(word) or not word:
52
+ i += 1
53
+ continue
54
+ j = i + 1
55
+ while j < len(words) and _MARKER.match(words[j]):
56
+ j += 1
57
+ if j < len(words):
58
+ base = _base_letters(words[j])
59
+ first = base[0] if base else ""
60
+ if word.endswith("\u0646" + _SUKUN) and first in noon_set:
61
+ words[i], words[j] = word[:-1], _insert_shadda(words[j])
62
+ elif _ends_with_tanween(word) and first in noon_set:
63
+ words[j] = _insert_shadda(words[j])
64
+ elif word.endswith("\u0645" + _SUKUN) and first == "\u0645":
65
+ words[i], words[j] = word[:-1], _insert_shadda(words[j])
66
+ elif extended and word.endswith("\u0644" + _SUKUN) and first in _IDGHAM_AFTER_LAM:
67
+ words[i], words[j] = word[:-1], _insert_shadda(words[j])
68
+ i += 1
69
+ return " ".join(words)
70
+
71
+
index.html CHANGED
@@ -1,19 +1,518 @@
1
  <!doctype html>
2
- <html>
3
- <head>
4
- <meta charset="utf-8" />
5
- <meta name="viewport" content="width=device-width" />
6
- <title>My static Space</title>
7
- <link rel="stylesheet" href="style.css" />
8
- </head>
9
- <body>
10
- <div class="card">
11
- <h1>Welcome to your static Space!</h1>
12
- <p>You can modify this app directly by editing <i>index.html</i> in the Files and versions tab.</p>
13
- <p>
14
- Also don't forget to check the
15
- <a href="https://huggingface.co/docs/hub/spaces" target="_blank">Spaces documentation</a>.
16
- </p>
17
- </div>
18
- </body>
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
19
  </html>
 
1
  <!doctype html>
2
+ <html lang="ar" dir="rtl">
3
+ <head>
4
+ <meta charset="utf-8">
5
+ <meta name="viewport" content="width=device-width, initial-scale=1">
6
+ <title>التحقق من هلوسة القرآن والحديث وتصحيحها</title>
7
+ <meta name="description" content="تحقّق من آيات القرآن والأحاديث النبوية بالدليل">
8
+ <link rel="preconnect" href="https://cdn.jsdelivr.net" crossorigin>
9
+ <link rel="preconnect" href="https://fonts.googleapis.com">
10
+ <link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
11
+ <link rel="preload" href="https://cdn.jsdelivr.net/pyodide/v0.26.4/full/pyodide.js" as="script" crossorigin>
12
+ <link rel="preload" href="index/quran.idx.gz" as="fetch" crossorigin>
13
+ <link rel="prefetch" href="index/hadith.idx.gz" as="fetch" crossorigin>
14
+ <style>
15
+ @import url('https://fonts.googleapis.com/css2?family=Cairo:wght@400;600;700&family=Amiri:wght@400;700&display=swap');
16
+ :root { --green:#0F4C3A; --green-2:#17694F; --gold:#B8912F; --gold-2:#E2C06E; --cream:#FBF6EA; --paper:#FFFFFF;
17
+ --ink:#1F2933; --muted:#55626D; --line:#E6DCC3; --ok:#1B7A4B; --ok-bg:#EAF6EF; --ok-line:#BFE3CD;
18
+ --bad:#B3261E; --bad-bg:#FDECEA; --bad-line:#F4B8B3; --warn:#7A5B0C; --warn-bg:#FFF6DA; --warn-line:#EBD28A; }
19
+ .gradio-container, body.icv-page { background:var(--cream) url("data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' width='88' height='88' viewBox='0 0 88 88'%3E%3Cg fill='none' stroke='%230F4C3A' stroke-opacity='0.07' stroke-width='1'%3E%3Crect x='22' y='22' width='44' height='44'/%3E%3Crect x='22' y='22' width='44' height='44' transform='rotate(45 44 44)'/%3E%3C/g%3E%3C/svg%3E") !important; font-family:'Cairo','Segoe UI',Tahoma,sans-serif; color:var(--ink); }
20
+ .gradio-container { max-width:1060px !important; --body-text-color:#1F2933; --block-background-fill:#FFFFFF; --block-border-color:#E6DCC3;
21
+ --input-background-fill:#FFFFFF; --block-label-text-color:#55626D; --button-primary-background-fill:#0F4C3A;
22
+ --button-primary-background-fill-hover:#17694F; --button-primary-text-color:#FFFFFF; --button-secondary-background-fill:#FFFFFF;
23
+ --button-secondary-text-color:#0F4C3A; --button-secondary-border-color:#B8912F; --color-accent:#B8912F; }
24
+ .icv { direction:rtl; text-align:right; color:var(--ink); line-height:1.9; }
25
+ .icv .hero { background:linear-gradient(135deg,#0F4C3A,#17694F); border-radius:20px; padding:30px 22px 26px; margin:10px 0 18px; text-align:center;
26
+ box-shadow:0 6px 22px rgba(15,76,58,.18); border-bottom:4px solid var(--gold); }
27
+ .icv .hero .mark { width:54px; height:54px; display:block; margin:0 auto 4px; }
28
+ .icv .hero h1 { font-size:2.3rem; margin:.1rem 0; color:#FFFFFF; font-weight:700; }
29
+ .icv .hero .tagline { color:var(--gold-2); font-size:1.2rem; font-weight:600; margin:.1rem 0 .6rem; }
30
+ .icv .hero .sub { max-width:720px; margin:0 auto; color:#E9F2EE; font-size:1rem; }
31
+ .icv .flow { display:flex; flex-wrap:wrap; justify-content:center; align-items:center; gap:8px; margin-top:16px; }
32
+ .icv .flow span { background:rgba(255,255,255,.12); border:1px solid rgba(226,192,110,.55); color:#FFF; border-radius:999px; padding:3px 16px; font-size:.9rem; }
33
+ .icv .flow i { color:var(--gold-2); font-style:normal; font-size:1.2rem; }
34
+ .icv .disclaimer { font-size:.85rem; color:var(--muted); text-align:center; padding:14px 8px; }
35
+ .icv .en-free { direction:rtl; }
36
+ .input-area textarea { direction:rtl; text-align:right; font-family:'Amiri','Cairo',serif !important; font-size:1.25rem !important; line-height:2.1 !important; background:#FFFFFF !important; color:#1F2933 !important; }
37
+ .llm-row label, .gradio-container label span { font-family:'Cairo',sans-serif; }
38
+ .results { direction:rtl; }
39
+ .icv .summary { display:grid; grid-template-columns:repeat(auto-fit,minmax(130px,1fr)); gap:10px; margin:8px 0 14px; }
40
+ .icv .tile { background:var(--paper); border:1px solid var(--line); border-radius:14px; padding:12px 8px; text-align:center; }
41
+ .icv .tile b { display:block; font-size:1.8rem; color:var(--green); line-height:1.3; } .icv .tile span { font-size:.92rem; color:var(--muted); }
42
+ .icv .tile.verified { background:var(--ok-bg); border-color:var(--ok-line); } .icv .tile.verified b { color:var(--ok); }
43
+ .icv .tile.mismatch { background:var(--bad-bg); border-color:var(--bad-line); } .icv .tile.mismatch b { color:var(--bad); }
44
+ .icv .tile.review { background:var(--warn-bg); border-color:var(--warn-line); } .icv .tile.review b { color:var(--warn); }
45
+ .icv .tile.zero { background:var(--paper); border-color:var(--line); opacity:.6; } .icv .tile.zero b { color:var(--muted); }
46
+ .icv .highlight, .icv .generated, .icv .final { background:var(--paper); border:1px solid var(--line); border-radius:14px; padding:14px 18px; margin-bottom:14px; }
47
+ .icv label { display:block; color:var(--muted); font-size:.82rem; font-weight:600; margin-bottom:4px; }
48
+ .icv .highlight p, .icv .generated p, .icv .final-text { font-family:'Amiri','Cairo',serif; font-size:1.2rem; line-height:2.2; margin:0; white-space:pre-wrap; }
49
+ .icv mark { color:var(--ink); border-radius:6px; padding:1px 5px; }
50
+ .icv mark.verified { background:#D5EEDF; } .icv mark.mismatch { background:#F8CFCB; } .icv mark.review { background:#F7E3A6; }
51
+ .icv .hl-legend { display:flex; flex-wrap:wrap; gap:8px; margin-top:10px; font-size:.78rem; }
52
+ .icv .qcard { background:var(--paper); border:1px solid var(--line); border-inline-start:6px solid var(--line); border-radius:16px; padding:16px 18px; margin:12px 0; }
53
+ .icv .qcard.verified { border-inline-start-color:var(--ok); } .icv .qcard.mismatch { border-inline-start-color:var(--bad); } .icv .qcard.review { border-inline-start-color:var(--gold); }
54
+ .icv .qhead { display:flex; flex-wrap:wrap; gap:8px; align-items:center; margin-bottom:8px; }
55
+ .icv .idx { background:var(--green); color:#FFF; border-radius:50%; width:28px; height:28px; display:inline-flex; align-items:center; justify-content:center; font-weight:700; font-size:.9rem; }
56
+ .icv .badge { padding:2px 14px; border-radius:999px; font-size:.84rem; background:#F3EEDD; border:1px solid var(--line); color:var(--ink); }
57
+ .icv .badge.st.verified { background:var(--ok-bg); border-color:var(--ok-line); color:var(--ok); font-weight:600; }
58
+ .icv .badge.st.mismatch { background:var(--bad-bg); border-color:var(--bad-line); color:var(--bad); font-weight:600; }
59
+ .icv .badge.st.review { background:var(--warn-bg); border-color:var(--warn-line); color:var(--warn); font-weight:600; }
60
+ .icv .badge.scan { background:#EEF3FB; border-color:#C9D8EE; color:#2F4F7F; }
61
+ .icv .conf { margin-inline-start:auto; color:var(--muted); font-size:.88rem; } .icv .conf b { color:var(--green); }
62
+ .icv .field { margin:8px 0; } .icv .field p { margin:0; }
63
+ .icv .quote { font-family:'Amiri','Cairo',serif; font-size:1.25rem; line-height:2.2; color:var(--ink); } .icv .quote.small { font-size:1.05rem; color:#3a4651; }
64
+ .icv .diff { font-family:'Amiri','Cairo',serif; font-size:1.2rem; line-height:2.2; }
65
+ .icv .w-extra { background:var(--bad-bg); color:var(--bad); border-radius:4px; padding:0 4px; text-decoration:line-through; }
66
+ .icv .w-missing { background:var(--ok-bg); color:var(--ok); border-radius:4px; padding:0 4px; font-weight:700; }
67
+ .icv .legend { color:var(--muted); font-size:.82rem; margin:4px 0 0; } .icv .legend span { text-decoration:none; font-size:.8rem; }
68
+ .icv .reason { color:var(--muted); font-size:.92rem; margin:6px 0 0; }
69
+ .icv .note { color:var(--warn); background:var(--warn-bg); border:1px solid var(--warn-line); border-radius:10px; padding:5px 12px; font-size:.88rem; margin:8px 0 0; }
70
+ .icv .action { border-radius:12px; padding:10px 14px; margin-top:10px; }
71
+ .icv .action.ok { background:var(--ok-bg); border:1px solid var(--ok-line); } .icv .action.ok b { color:var(--ok); }
72
+ .icv .action.warn { background:var(--warn-bg); border:1px solid var(--warn-line); } .icv .action.warn b { color:var(--warn); }
73
+ .icv .action.bad { background:var(--bad-bg); border:1px solid var(--bad-line); } .icv .action.bad b { color:var(--bad); }
74
+ .icv .evidence { margin-top:12px; border-top:1px dashed var(--line); padding-top:8px; }
75
+ .icv .evidence summary { cursor:pointer; color:var(--green); font-weight:700; }
76
+ .icv .ev-row { margin:10px 0; } .icv .ev-row b { color:var(--gold); font-size:.88rem; } .icv .ev-row p { margin:2px 0; }
77
+ .icv .indicators { display:grid; gap:6px; margin-top:6px; }
78
+ .icv .ind { display:grid; grid-template-columns:150px 1fr 48px; gap:10px; align-items:center; font-size:.88rem; }
79
+ .icv .bar { background:#EFE8D3; border-radius:999px; height:9px; overflow:hidden; direction:rtl; } .icv .bar span { display:block; height:100%; background:linear-gradient(270deg,var(--green),var(--gold)); }
80
+ .icv .ind-val { text-align:left; direction:ltr; color:var(--ink); } .icv .ind-val.wide { grid-column:2 / span 2; text-align:right; direction:rtl; }
81
+ .icv .final { border-color:var(--gold); background:#FFFDF6; }
82
+ .icv .final-head { display:flex; justify-content:space-between; align-items:center; } .icv .final-head b { color:var(--green); font-size:1.05rem; }
83
+ .icv .copy { background:var(--green); color:#FFF; border:0; border-radius:10px; padding:5px 16px; font-family:inherit; cursor:pointer; }
84
+ .icv .notice { background:var(--paper); border:1px solid var(--line); border-radius:14px; padding:16px; }
85
+ .icv .notice.warn { background:var(--warn-bg); border-color:var(--warn-line); color:var(--warn); } .icv .notice.bad { background:var(--bad-bg); border-color:var(--bad-line); color:var(--bad); }
86
+ @media (max-width:640px){ .icv .hero h1{font-size:1.7rem;} .icv .ind{grid-template-columns:104px 1fr 40px;} }
87
+
88
+ :root { --glass:rgba(255,255,255,.62); --glass-strong:rgba(255,255,255,.82); --glass-line:rgba(255,255,255,.75);
89
+ --shadow-1:0 1px 2px rgba(15,76,58,.06), 0 8px 24px rgba(15,76,58,.08); --shadow-2:0 2px 4px rgba(15,76,58,.08), 0 18px 44px rgba(15,76,58,.16); }
90
+ .gradio-container, body.icv-page { background:
91
+ radial-gradient(900px 520px at 88% -8%, rgba(226,192,110,.34), transparent 60%),
92
+ radial-gradient(760px 520px at 6% 4%, rgba(23,105,79,.20), transparent 62%),
93
+ linear-gradient(180deg,#FBF6EA 0%,#F3EBD3 100%) !important; background-attachment:fixed !important; }
94
+ /* RTL safety: long words, URLs and mixed Latin/digit runs wrap inside their box instead of spilling out */
95
+ .icv, .icv * { box-sizing:border-box; min-width:0; }
96
+ .icv p, .icv span, .icv b, .icv label, .icv summary, .icv button, .icv .badge, .icv .tile { overflow-wrap:anywhere; }
97
+ .icv .quote, .icv .diff, .icv .final-text, .icv .highlight p, .icv .generated p { unicode-bidi:plaintext; text-align:start; }
98
+ .icv .hero { position:relative; overflow:hidden; border-bottom:0; border:1px solid rgba(226,192,110,.45);
99
+ background:linear-gradient(120deg,#0B3B2D,#17694F 45%,#0F4C3A 70%,#1d7a5c); background-size:240% 240%; animation:icv-flow 16s ease-in-out infinite;
100
+ box-shadow:var(--shadow-2); }
101
+ .icv .hero::before { content:""; position:absolute; inset:-40% -10% auto auto; width:60%; aspect-ratio:1; border-radius:50%;
102
+ background:radial-gradient(circle, rgba(226,192,110,.38), transparent 65%); pointer-events:none; }
103
+ .icv .hero::after { content:""; position:absolute; inset:auto auto 0 0; width:100%; height:4px; background:linear-gradient(90deg,transparent,var(--gold-2),transparent); }
104
+ .icv .hero h1 { font-size:clamp(1.45rem,4.2vw,2.3rem); line-height:1.5; text-wrap:balance; position:relative; }
105
+ .icv .hero .tagline, .icv .hero .sub, .icv .flow { position:relative; }
106
+ .icv .flow span { backdrop-filter:blur(6px); -webkit-backdrop-filter:blur(6px); }
107
+ @keyframes icv-flow { 0%,100%{background-position:0% 50%} 50%{background-position:100% 50%} }
108
+ .icv .tile, .icv .highlight, .icv .generated, .icv .final, .icv .qcard, .icv .notice, .icv .export {
109
+ background:var(--glass); border:1px solid var(--glass-line); box-shadow:var(--shadow-1);
110
+ backdrop-filter:blur(14px) saturate(150%); -webkit-backdrop-filter:blur(14px) saturate(150%); transition:transform .25s ease, box-shadow .25s ease; }
111
+ .icv .qcard { border-inline-start:6px solid var(--line); }
112
+ .icv .qcard:hover, .icv .tile:hover { transform:translateY(-2px); box-shadow:var(--shadow-2); }
113
+ .icv .tile.verified { background:linear-gradient(160deg,rgba(234,246,239,.92),rgba(255,255,255,.6)); }
114
+ .icv .tile.mismatch { background:linear-gradient(160deg,rgba(253,236,234,.92),rgba(255,255,255,.6)); }
115
+ .icv .tile.review { background:linear-gradient(160deg,rgba(255,246,218,.95),rgba(255,255,255,.6)); }
116
+ .icv .final { background:linear-gradient(160deg,rgba(255,253,246,.92),rgba(250,240,208,.55)); border-color:rgba(184,145,47,.55); }
117
+ .icv .final-head { flex-wrap:wrap; gap:10px; }
118
+ .icv .qhead .badge { max-width:100%; white-space:normal; }
119
+ .icv .ind { grid-template-columns:minmax(96px,150px) minmax(0,1fr) 48px; }
120
+ .icv .bar span { background:linear-gradient(270deg,var(--green),var(--gold-2)); transition:width .6s ease; }
121
+ .icv .copy, .icv .dl { background:linear-gradient(135deg,var(--green),var(--green-2)); color:#FFF; border:0; border-radius:12px; padding:7px 18px;
122
+ font-family:inherit; font-weight:600; cursor:pointer; box-shadow:0 4px 12px rgba(15,76,58,.25); transition:transform .15s ease, box-shadow .15s ease, filter .15s ease; }
123
+ .icv .copy:hover, .icv .dl:hover { transform:translateY(-1px); filter:brightness(1.08); box-shadow:0 8px 18px rgba(15,76,58,.3); }
124
+ .icv .copy:active, .icv .dl:active { transform:translateY(0); }
125
+ .icv .copy:focus-visible, .icv .dl:focus-visible { outline:3px solid rgba(184,145,47,.55); outline-offset:2px; }
126
+ .icv .copy.done { background:linear-gradient(135deg,#1B7A4B,#2a9d66); }
127
+ .icv .export { border-radius:16px; padding:12px 18px; margin:12px 0; }
128
+ .icv .export summary { cursor:pointer; color:var(--green); font-weight:700; }
129
+ .icv .dls { display:flex; flex-wrap:wrap; gap:10px; margin-top:10px; }
130
+ /* toast */
131
+ #icv-toast { position:fixed; inset-inline:0; bottom:28px; margin-inline:auto; width:max-content; max-width:calc(100vw - 32px); z-index:99999;
132
+ direction:rtl; text-align:center; font:600 1rem 'Cairo','Segoe UI',Tahoma,sans-serif; color:#FFF; padding:12px 24px; border-radius:999px;
133
+ background:linear-gradient(135deg,rgba(15,76,58,.94),rgba(23,105,79,.94)); border:1px solid rgba(226,192,110,.6);
134
+ box-shadow:0 14px 40px rgba(15,76,58,.35); backdrop-filter:blur(10px); -webkit-backdrop-filter:blur(10px);
135
+ opacity:0; transform:translateY(18px) scale(.97); pointer-events:none; transition:opacity .28s ease, transform .28s cubic-bezier(.2,.9,.3,1.2); }
136
+ #icv-toast.show { opacity:1; transform:none; } #icv-toast.bad { background:linear-gradient(135deg,rgba(179,38,30,.95),rgba(214,69,58,.95)); }
137
+ /* skeleton loaders */
138
+ .skel { position:relative; overflow:hidden; background:rgba(15,76,58,.08); border-radius:10px; }
139
+ .skel::after { content:""; position:absolute; inset:0; transform:translateX(100%); animation:icv-shimmer 1.4s infinite;
140
+ background:linear-gradient(90deg,transparent,rgba(255,255,255,.75),transparent); }
141
+ @keyframes icv-shimmer { 100% { transform:translateX(-100%); } }
142
+ .skel-card { background:var(--glass); border:1px solid var(--glass-line); border-radius:16px; padding:16px 18px; margin:12px 0; box-shadow:var(--shadow-1); }
143
+ .skel-line { height:14px; margin:10px 0; } .skel-line.w60 { width:60%; } .skel-line.w85 { width:85%; } .skel-line.w40 { width:40%; }
144
+ .skel-tiles { display:grid; grid-template-columns:repeat(auto-fit,minmax(120px,1fr)); gap:10px; margin:8px 0 14px; } .skel-tile { height:78px; border-radius:14px; }
145
+ .icv-spinner { width:18px; height:18px; border-radius:50%; border:3px solid rgba(15,76,58,.18); border-top-color:var(--gold); display:inline-block;
146
+ vertical-align:middle; margin-inline-end:10px; animation:icv-spin .8s linear infinite; }
147
+ @keyframes icv-spin { to { transform:rotate(360deg); } }
148
+ @media (prefers-reduced-motion: reduce) { .icv .hero, .skel::after, .icv-spinner { animation:none; } .icv .qcard, .icv .tile, #icv-toast { transition:none; } }
149
+ @media (max-width:640px){ .icv .ind{grid-template-columns:minmax(84px,104px) minmax(0,1fr) 40px;} .icv .final-head{flex-direction:column; align-items:stretch;} .icv .copy{width:100%;} }
150
+ /* layout polish: centred, compact, professional */
151
+ .icv .notice, .icv .disclaimer, .icv .export { text-align:center; }
152
+ .icv .export .legend { max-width:640px; margin:8px auto 0; font-size:.9rem; line-height:1.9; color:var(--muted); }
153
+ .icv .dls { justify-content:center; }
154
+ .icv .dl { display:inline-flex; flex-direction:column; align-items:center; gap:2px; min-width:170px; }
155
+ .icv .dl .t { font-weight:700; } .icv .dl .s { font-size:.78rem; font-weight:500; opacity:.85; }
156
+ .icv .hero .flow { display:flex; justify-content:center; flex-wrap:wrap; gap:8px; }
157
+
158
+ body.icv-page{margin:0;min-height:100vh;background-attachment:fixed;} .page{max-width:980px;margin:0 auto;padding:10px 16px 28px;box-sizing:border-box;}
159
+ .glass{background:var(--glass);border:1px solid var(--glass-line);border-radius:20px;box-shadow:var(--shadow-1);
160
+ backdrop-filter:blur(16px) saturate(150%);-webkit-backdrop-filter:blur(16px) saturate(150%);}
161
+ .tabs{display:flex;gap:6px;margin:6px auto 14px;direction:rtl;padding:6px;border-radius:18px;max-width:560px;}
162
+ .tab{flex:1;border:0;background:transparent;color:var(--green);border-radius:13px;padding:11px 14px;font:700 1rem 'Cairo',sans-serif;cursor:pointer;transition:background .2s,color .2s,box-shadow .2s;}
163
+ .tab:hover{background:rgba(15,76,58,.07);}
164
+ .tab.active{background:linear-gradient(135deg,var(--green),var(--green-2));color:#FFF;box-shadow:0 6px 16px rgba(15,76,58,.28);}
165
+ .tab:focus-visible,.btns button:focus-visible{outline:3px solid rgba(184,145,47,.55);outline-offset:2px;}
166
+ section[hidden]{display:none;}
167
+ .form{padding:18px 20px 14px;margin-bottom:14px;} .form .icv{margin-bottom:6px;}
168
+ .form label{display:block;color:var(--muted);font-size:.88rem;font-weight:600;margin:10px 0 5px;direction:rtl;text-align:right;}
169
+ .form textarea,.form select{width:100%;box-sizing:border-box;background:rgba(255,255,255,.88);color:var(--ink);border:1px solid var(--line);border-radius:14px;padding:12px 14px;font:1rem 'Cairo',sans-serif;direction:rtl;text-align:right;outline:none;transition:border-color .2s,box-shadow .2s;}
170
+ .form textarea{min-height:200px;resize:vertical;font-family:'Amiri','Cairo',serif;font-size:1.25rem;line-height:2.1;}
171
+ .form textarea.short{min-height:96px;}
172
+ .form textarea:focus,.form select:focus{border-color:var(--gold);box-shadow:0 0 0 4px rgba(184,145,47,.18);}
173
+ .hint{font-size:.84rem;color:var(--muted);direction:rtl;margin:2px 0 0;}
174
+ .btns{display:flex;gap:12px;margin:14px 0 6px;flex-wrap:wrap;justify-content:center;}
175
+ .btns button{flex:1 1 200px;border:0;border-radius:14px;padding:13px 18px;font:700 1.02rem 'Cairo',sans-serif;cursor:pointer;transition:transform .15s,box-shadow .15s,filter .15s;}
176
+ .btn-main{background:linear-gradient(135deg,var(--green),var(--green-2));color:#FFF;box-shadow:0 8px 20px rgba(15,76,58,.28);}
177
+ .btn-main:hover{transform:translateY(-1px);filter:brightness(1.07);}
178
+ .btn-alt{background:rgba(255,255,255,.85);color:var(--green);border:1.5px solid var(--gold)!important;} .btn-alt:hover{background:#FFFBEF;transform:translateY(-1px);}
179
+ .btns button:disabled{opacity:.5;cursor:not-allowed;transform:none;}
180
+ #status{direction:rtl;text-align:center;color:var(--muted);font-size:.93rem;margin:8px 0;} #status:empty{display:none;}
181
+ .boot{padding:14px 18px;margin:0 0 14px;direction:rtl;text-align:center;}
182
+ .boot .step{text-align:right;}
183
+ .boot h2{margin:0 0 8px;font-size:.98rem;color:var(--green);}
184
+ .boot .step{display:grid;grid-template-columns:22px minmax(0,1fr) 54px;gap:10px;align-items:center;margin:8px 0;font-size:.92rem;color:var(--muted);}
185
+ .boot .dot{width:18px;height:18px;border-radius:50%;border:2px solid rgba(15,76,58,.22);display:inline-block;}
186
+ .boot .step.active .dot{border-color:rgba(15,76,58,.18);border-top-color:var(--gold);animation:icv-spin .8s linear infinite;}
187
+ .boot .step.done .dot{background:var(--ok);border-color:var(--ok);position:relative;} .boot .step.done .dot::after{content:"";position:absolute;inset:3px 5px 5px 5px;border:solid #FFF;border-width:0 2px 2px 0;transform:rotate(40deg);}
188
+ .boot .step.done{color:var(--ok);} .boot .step.active{color:var(--ink);font-weight:600;}
189
+ .boot .meter{height:8px;border-radius:999px;background:rgba(15,76,58,.1);overflow:hidden;} .boot .meter i{display:block;height:100%;width:0;background:linear-gradient(270deg,var(--green),var(--gold-2));transition:width .3s ease;}
190
+ .boot .step.pending .meter{position:relative;} .boot .pct{font-size:.8rem;direction:ltr;text-align:left;}
191
+ .boot.ready{padding:8px 16px;} .boot.ready h2{margin:0;} .boot.ready .step{display:none;}
192
+ .banner{margin:0 0 12px;}
193
+ </style>
194
+ </head>
195
+ <body class="icv-page">
196
+ <div class="page">
197
+
198
+ <div class="icv"><div class="hero">
199
+ <svg class="mark" viewBox="0 0 64 64" fill="none" stroke="#E2C06E" stroke-width="1.8" aria-hidden="true"><rect x="14" y="14" width="36" height="36"/><rect x="14" y="14" width="36" height="36" transform="rotate(45 32 32)"/><circle cx="32" cy="32" r="7" fill="#E2C06E" stroke="none"/></svg>
200
+ <h1>التحقق من هلوسة القرآن والحديث وتصحيحها</h1>
201
+ <p class="tagline">تحقّق من آيات القرآن والأحاديث النبوية بالدليل</p>
202
+ <p class="sub">يكتشف النظام الاقتباسات داخل أي نص، ويسترجع نصوصها من المصادر، ويقارنها كلمةً بكلمة، ثم يعرض الدليل.
203
+ وإذا لم تكفِ الأدلة فلن يختلق تصحيحًا، بل يحيل الحالة إلى المراجعة البشرية.</p>
204
+ <div class="flow"><span>كشف</span><i>‹</i><span>استرجاع</span><i>‹</i><span>محاذاة</span><i>‹</i><span>دليل</span><i>‹</i><span>قرار</span></div>
205
+ </div></div>
206
+
207
+ <div class="boot glass" id="boot" aria-live="polite">
208
+ <h2 id="boot-title">جارٍ تجهيز النظام في متصفحك…</h2>
209
+ <div class="step pending" data-step="python"><i class="dot"></i><span>بيئة بايثون</span><span class="pct"></span><div class="meter" style="grid-column:2 / span 2"><i></i></div></div>
210
+ <div class="step pending" data-step="code"><i class="dot"></i><span>الكود وفهرس القرآن</span><span class="pct"></span><div class="meter" style="grid-column:2 / span 2"><i></i></div></div>
211
+ <div class="step pending" data-step="hadith"><i class="dot"></i><span>كتب الحديث (تُحمَّل بالتوازي)</span><span class="pct"></span><div class="meter" style="grid-column:2 / span 2"><i></i></div></div>
212
+ </div>
213
+ <div class="tabs glass" role="tablist"><button class="tab active" role="tab" aria-selected="true" data-tab="direct">تحقّق مباشر</button><button class="tab" role="tab" aria-selected="false" data-tab="ask">اسأل ثم تحقّق</button></div>
214
+ <section id="tab-direct" class="form glass">
215
+ <label for="text">النص المراد التحقق منه</label>
216
+ <textarea id="text" placeholder="الصق هنا النص الذي ولّده نموذج لغوي. يكتشف النظام الآيات والأحاديث الواردة فيه، بعلامات تنصيص أو بدونها…"></textarea>
217
+ <div class="btns"><button id="verify" class="btn-main">تحقّق من النص</button><button id="example" class="btn-alt">جرّب مثالًا</button></div>
218
+ </section>
219
+ <section id="tab-ask" class="form glass" hidden>
220
+ <div class="icv"><div class="notice">اكتب سؤالًا، وسيجيب عنه النظام مباشرةً، ثم يفحص كل آية وحديث في الإجابة ويعرض الأخطاء والتصحيحات.</div></div>
221
+ <label for="prompt">سؤالك</label>
222
+ <textarea id="prompt" class="short" maxlength="1500" placeholder="اكتب سؤالك، مثل: اشرح لي فضل الصبر في القرآن والسنة مع ذكر الأدلة."></textarea>
223
+ <div class="btns"><button id="ask" class="btn-main">اسأل ثم تحقّق</button><button id="ask-example" class="btn-alt">جرّب مثالًا</button></div>
224
+ </section>
225
+ <div id="status" role="status"></div>
226
+ <div id="results" class="results"></div>
227
+ <div class="icv"><div class="disclaimer">أداة مساعدة للتدقيق النصي وليست فتوى ولا بديلًا عن المراجعة المتخصصة.
228
+ النتائج مبنية على مراجع القرآن الكريم والكتب الستة المضمّنة فقط.</div></div>
229
+ </div>
230
+ <script>
231
+ (function () {
232
+ if (window.icvCopy) return;
233
+ function toast(message, bad) {
234
+ var t = document.getElementById('icv-toast');
235
+ if (!t) { t = document.createElement('div'); t.id = 'icv-toast'; t.setAttribute('role', 'status'); t.setAttribute('aria-live', 'polite'); document.body.appendChild(t); }
236
+ t.textContent = message; t.className = 'show' + (bad ? ' bad' : '');
237
+ clearTimeout(window.__icvToast); window.__icvToast = setTimeout(function () { t.className = bad ? 'bad' : ''; }, 2400);
238
+ }
239
+ function legacyCopy(text) {
240
+ var area = document.createElement('textarea'); area.value = text; area.setAttribute('readonly', '');
241
+ area.style.cssText = 'position:fixed;top:0;left:0;opacity:0;pointer-events:none'; document.body.appendChild(area);
242
+ area.select(); area.setSelectionRange(0, text.length); var ok = false;
243
+ try { ok = document.execCommand('copy'); } catch (e) { ok = false; }
244
+ document.body.removeChild(area); return ok;
245
+ }
246
+ window.icvToast = toast;
247
+ window.icvCopy = function (button) {
248
+ var text = button.getAttribute('data-text') || '';
249
+ function finish(ok) {
250
+ toast(ok ? 'تم نسخ النص بنجاح' : 'تعذّر النسخ، حدّد النص وانسخه يدويًا', !ok);
251
+ if (ok) { button.classList.add('done'); setTimeout(function () { button.classList.remove('done'); }, 1600); }
252
+ }
253
+ if (navigator.clipboard && window.isSecureContext) {
254
+ navigator.clipboard.writeText(text).then(function () { finish(true); }, function () { finish(legacyCopy(text)); });
255
+ } else { finish(legacyCopy(text)); }
256
+ };
257
+ window.icvDownload = function (button) {
258
+ var blob = new Blob([button.getAttribute('data-text') || ''], { type: 'text/tab-separated-values;charset=utf-8' });
259
+ var link = document.createElement('a'); link.href = URL.createObjectURL(blob); link.download = button.getAttribute('data-name') || 'results.tsv';
260
+ document.body.appendChild(link); link.click(); document.body.removeChild(link); setTimeout(function () { URL.revokeObjectURL(link.href); }, 1000);
261
+ };
262
+ })();
263
+ </script>
264
+ <script type="text/plain" id="worker-src">
265
+ // Python (Pyodide) and the whole verification pipeline run here, off the main thread.
266
+ let py = null, booted = null, hadithReady = null;
267
+ const post = (message) => self.postMessage(message);
268
+ const step = (name, state, extra) => post(Object.assign({ type: "step", name, state }, extra || {}));
269
+
270
+ async function openCache(buildId) {
271
+ try {
272
+ const names = await caches.keys();
273
+ await Promise.all(names.filter((n) => n.startsWith("icv-") && n !== "icv-" + buildId).map((n) => caches.delete(n))); // purge old builds
274
+ return await caches.open("icv-" + buildId);
275
+ } catch (e) { return null; }
276
+ }
277
+
278
+ async function fetchBytes(url, cache, onProgress) {
279
+ if (cache) { try { const hit = await cache.match(url); if (hit) return { bytes: new Uint8Array(await hit.arrayBuffer()), cached: true }; } catch (e) { /* fall through */ } }
280
+ const response = await fetch(url);
281
+ if (!response.ok) throw new Error("تعذّر تحميل " + url + " (" + response.status + ")");
282
+ if (cache) { try { await cache.put(url, response.clone()); } catch (e) { /* the cache is an optimisation only */ } }
283
+ const total = Number(response.headers.get("Content-Length")) || 0;
284
+ if (!response.body || !onProgress) return { bytes: new Uint8Array(await response.arrayBuffer()), cached: false };
285
+ const reader = response.body.getReader(), chunks = []; let received = 0;
286
+ for (;;) { const { done, value } = await reader.read(); if (done) break; chunks.push(value); received += value.length; onProgress(received, total); }
287
+ const bytes = new Uint8Array(received); let offset = 0;
288
+ for (const chunk of chunks) { bytes.set(chunk, offset); offset += chunk.length; }
289
+ return { bytes, cached: false };
290
+ }
291
+
292
+ // Accept both layouts: files inside folders (index/, demo/) or uploaded flat next to index.html.
293
+ async function fetchAny(file, base, cache, onProgress) {
294
+ let lastError = null;
295
+ for (const path of [file, file.split("/").pop()]) {
296
+ try { return await fetchBytes(new URL(path, base).href, cache, onProgress); } catch (error) { lastError = error; }
297
+ }
298
+ throw lastError;
299
+ }
300
+
301
+ // A server that sets Content-Encoding: gzip on .gz files makes the browser inflate them; Python expects gzip, so re-pack.
302
+ async function ensureGzip(file, bytes) {
303
+ if (!file.endsWith(".gz") || (bytes[0] === 0x1f && bytes[1] === 0x8b)) return bytes;
304
+ const stream = new Blob([bytes]).stream().pipeThrough(new CompressionStream("gzip"));
305
+ return new Uint8Array(await new Response(stream).arrayBuffer());
306
+ }
307
+
308
+ async function boot(message) {
309
+ const cache = await openCache(message.buildId);
310
+ step("python", "active");
311
+ importScripts("https://cdn.jsdelivr.net/pyodide/v" + message.pyodideVersion + "/full/pyodide.js");
312
+ const hadithFile = message.lazy[0];
313
+ // the large Hadith index starts downloading now, in parallel with the Python runtime
314
+ const hadithDownload = fetchAny(hadithFile, message.base, cache, (got, total) => step("hadith", "active", { pct: total ? Math.round((got / total) * 100) : null, mb: (got / 1048576).toFixed(1) }));
315
+ hadithDownload.catch(() => {});
316
+ py = await loadPyodide();
317
+ step("python", "done");
318
+ step("code", "active", { pct: 0 });
319
+ for (const dir of ["/app", "/app/index", "/app/demo", "/app/data"]) py.FS.mkdir(dir);
320
+ let cached = true, n = 0;
321
+ for (const file of message.files) {
322
+ const result = await fetchAny(file, message.base, cache);
323
+ cached = cached && result.cached;
324
+ py.FS.writeFile("/app/" + file, await ensureGzip(file, result.bytes));
325
+ step("code", "active", { pct: Math.round((++n / message.files.length) * 100) });
326
+ }
327
+ const examplesText = new TextDecoder().decode(py.FS.readFile("/app/demo/examples.json"));
328
+ py.runPython("import sys; sys.path.insert(0, '/app')");
329
+ py.runPython("from app import get_pipeline, verify_text, verify_generated_answer; get_pipeline()");
330
+ step("code", "done", { cached });
331
+ post({ type: "ready", examples: JSON.parse(examplesText) });
332
+ hadithReady = (async () => {
333
+ const result = await hadithDownload;
334
+ step("hadith", "active", { pct: 100, prepare: true });
335
+ py.FS.writeFile("/app/" + hadithFile, await ensureGzip(hadithFile, result.bytes));
336
+ py.runPython("get_pipeline().retriever.warm()");
337
+ step("hadith", "done", { cached: result.cached });
338
+ post({ type: "hadith" });
339
+ })();
340
+ await hadithReady;
341
+ }
342
+
343
+ self.onmessage = async (event) => {
344
+ const message = event.data;
345
+ if (message.type === "boot") {
346
+ booted = boot(message);
347
+ try { await booted; } catch (error) { post({ type: "error", text: String((error && error.message) || error) }); }
348
+ } else if (message.type === "verify") {
349
+ try {
350
+ await booted;
351
+ const html = py.globals.get(message.generated ? "verify_generated_answer" : "verify_text")(message.text, message.entities || "");
352
+ post({ type: "result", id: message.id, html });
353
+ } catch (error) { post({ type: "result", id: message.id, html: null }); }
354
+ }
355
+ };
356
+ </script>
357
+ <script>
358
+ const DEFAULT_CONFIG = {"askEndpoint": "https://icv-ask-proxy.ghada-islamic-verifier-2026.workers.dev", "hf": {"model": "", "endpoint": "https://router.huggingface.co/hf-inference/models/{model}"}};
359
+ const BUILD_ID = "eae236d51019";
360
+ const SKELETON = "<div class=\"icv\" aria-busy=\"true\" aria-label=\"جارٍ التحقق\"><div class=\"skel-tiles\"><div class=\"skel skel-tile\"></div><div class=\"skel skel-tile\"></div><div class=\"skel skel-tile\"></div><div class=\"skel skel-tile\"></div></div><div class=\"skel-card\"><div class=\"skel skel-line w40\"></div><div class=\"skel skel-line w85\"></div><div class=\"skel skel-line\"></div><div class=\"skel skel-line w60\"></div></div><div class=\"skel-card\"><div class=\"skel skel-line w40\"></div><div class=\"skel skel-line w85\"></div><div class=\"skel skel-line\"></div><div class=\"skel skel-line w60\"></div></div></div>";
361
+ const MAX_PROMPT = 1500;
362
+ const $ = (id) => document.getElementById(id);
363
+ let cfg = DEFAULT_CONFIG, examples = [], exampleIndex = 0, askExampleIndex = 0, nextId = 0, worker = null, ready = false;
364
+ const pending = new Map();
365
+
366
+ const store = {
367
+ get(key) { try { return localStorage.getItem(key); } catch (e) { return null; } },
368
+ set(key, value) { try { localStorage.setItem(key, value); } catch (e) { /* private mode / quota: caching is optional */ } },
369
+ };
370
+ function purgeOldLocalCache() {
371
+ try { for (const key of Object.keys(localStorage)) if (key.startsWith("icv:examples:") && key !== "icv:examples:" + BUILD_ID) localStorage.removeItem(key); } catch (e) { /* ignore */ }
372
+ }
373
+
374
+ const esc = (s) => String(s).replace(/[&<>"]/g, (c) => ({ "&": "&amp;", "<": "&lt;", ">": "&gt;", '"': "&quot;" }[c]));
375
+ function setStatus(text, busy = true) { $("status").innerHTML = (busy && text ? '<span class="icv-spinner"></span>' : "") + esc(text); }
376
+ function setBusy(flag) { for (const id of ["verify", "example", "ask", "ask-example"]) $(id).disabled = flag; }
377
+ function notice(text, kind) { return '<div class="icv banner"><div class="notice ' + kind + '">' + esc(text) + "</div></div>"; }
378
+ function show(html, banner) { $("results").innerHTML = (banner || "") + html; }
379
+ function showSkeleton() { $("results").innerHTML = SKELETON; }
380
+
381
+ function verifyRemote(payload) {
382
+ return new Promise((resolve) => { const id = ++nextId; pending.set(id, resolve); worker.postMessage(Object.assign({ type: "verify", id }, payload)); });
383
+ }
384
+
385
+ // ---- detection: the bundled rule + corpus detector always runs; a fine-tuned CAMeLBERT-MSA hosted on Hugging Face (when configured)
386
+ // runs alongside it and both results are merged inside the pipeline. Nothing here is visible to the visitor and any failure is silent.
387
+ async function hostedEntities(text) {
388
+ const model = (cfg.hf && cfg.hf.model || "").trim();
389
+ if (!model) return "";
390
+ const controller = new AbortController(), timer = setTimeout(() => controller.abort(), 6000);
391
+ try {
392
+ const response = await fetch(cfg.hf.endpoint.replace("{model}", encodeURIComponent(model)), {
393
+ method: "POST", headers: { "Content-Type": "application/json" }, signal: controller.signal,
394
+ body: JSON.stringify({ inputs: text, parameters: { aggregation_strategy: "simple" } }),
395
+ });
396
+ if (!response.ok) return "";
397
+ const data = await response.json();
398
+ return Array.isArray(data) ? JSON.stringify(data) : "";
399
+ } catch (e) { return ""; } finally { clearTimeout(timer); }
400
+ }
401
+
402
+ async function verifyAuto(text, generated) {
403
+ const entities = await hostedEntities(text);
404
+ return verifyRemote({ text, generated, entities });
405
+ }
406
+
407
+ // Small models sometimes drift into Chinese etc.; strip scripts that never belong in an Arabic answer (the server does it too).
408
+ const FOREIGN = /[\u0400-\u04ff\u0900-\u097f\u0e00-\u0e7f\u1100-\u11ff\u3000-\u303f\u3040-\u30ff\u3130-\u318f\u3400-\u4dbf\u4e00-\u9fff\uac00-\ud7af\uf900-\ufaff\uff00-\uffef]+/g;
409
+ const cleanAnswer = (t) => String(t).replace(FOREIGN, " ").replace(/([،,؛.])\s*[،,؛]+/g, "$1").replace(/[ \t]{2,}/g, " ").replace(/ +([،؛.:])/g, "$1").trim();
410
+
411
+ async function runDirect() {
412
+ if (!$("text").value.trim()) return show(notice("الرجاء إدخال نص للتحقق منه.", "warn"));
413
+ setBusy(true); showSkeleton(); setStatus(ready ? "جارٍ التحقق…" : "جارٍ تجهيز النظام ثم التحقق…");
414
+ const html = await verifyAuto($("text").value, false);
415
+ show(html !== null ? html : notice("حدث خطأ غير متوقع أثناء التحقق.", "bad"));
416
+ setStatus("", false); setBusy(false);
417
+ }
418
+
419
+ // ---- ask then verify: same-origin proxy that holds the key; saved sample answer when it is not reachable ----------------
420
+ async function askServer(prompt) {
421
+ const response = await fetch(cfg.askEndpoint, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ prompt }) });
422
+ if (!response.ok) {
423
+ const error = new Error("http_" + response.status); error.status = response.status;
424
+ try { error.code = (await response.json()).error; } catch (e) { error.code = ""; }
425
+ throw error;
426
+ }
427
+ const data = await response.json();
428
+ if (!data.answer) throw new Error("empty");
429
+ return cleanAnswer(data.answer);
430
+ }
431
+
432
+ async function runAsk() {
433
+ const prompt = $("prompt").value.trim();
434
+ if (!prompt) return show(notice("اكتب سؤالًا أولًا.", "warn"));
435
+ if (prompt.length > MAX_PROMPT) return show(notice("السؤال طويل جدًا (الحد الأقصى " + MAX_PROMPT + " حرف).", "warn"));
436
+ setBusy(true); showSkeleton(); setStatus("جارٍ إعداد الإجابة…");
437
+ let answer = null, banner = "";
438
+ try { answer = await askServer(prompt); }
439
+ catch (error) {
440
+ const stop = (message) => { show(notice(message, "warn")); setStatus("", false); setBusy(false); };
441
+ if (error.status === 402) return stop("رصيد حساب OpenAI المرتبط بالخدمة غير كافٍ؛ على صاحب الحساب مراجعة الفوترة ثم المحاولة مجددًا.");
442
+ if (error.status === 429) return stop("الطلبات كثيرة الآن؛ انتظر قليلًا ثم أعد المحاولة.");
443
+ if (error.status === 413) return stop("السؤال طويل جدًا.");
444
+ if (error.code === "upstream_auth") return stop("مفتاح الخدمة على الخادم غير صالح؛ يلزم تحديثه من صاحب الحساب.");
445
+ if (examples.length) {
446
+ answer = examples[exampleIndex++ % examples.length].text;
447
+ banner = notice("خدمة الإجابة المباشرة غير متاحة الآن؛ عُرضت إجابة تجريبية محفوظة لتوضيح خطوات التحقق.", "warn");
448
+ }
449
+ }
450
+ if (answer === null) { show(notice("تعذّر الحصول على إجابة الآن؛ حاول لاحقًا.", "warn")); setStatus("", false); return setBusy(false); }
451
+ setStatus("جارٍ التحقق من الإجابة…");
452
+ const html = await verifyAuto(answer, true);
453
+ show(html !== null ? html : notice("تعذّر التحقق من الإجابة.", "bad"), banner);
454
+ setStatus("", false); setBusy(false);
455
+ }
456
+
457
+ // Saved scenario: fills the question and verifies a stored model-style answer (no live call), so every case can be tried at once.
458
+ async function runAskExample() {
459
+ const samples = examples.filter((e) => e.question);
460
+ if (!samples.length) return show(notice("الأمثلة لم تُحمَّل بعد؛ انتظر لحظة ثم أعد المحاولة.", "warn"));
461
+ const sample = samples[askExampleIndex++ % samples.length];
462
+ $("prompt").value = sample.question;
463
+ setBusy(true); showSkeleton(); setStatus("جارٍ التحقق من الإجابة…");
464
+ const html = await verifyAuto(sample.text, true);
465
+ show(html !== null ? html : notice("تعذّر التحقق من الإجابة.", "bad"),
466
+ notice("إجابة تجريبية محفوظة لحالة «" + sample.title + "»، لم تُرسل إلى أي خدمة. لتجربة إجابة حيّة اكتب سؤالك واضغط «اسأل ثم تحقّق».", ""));
467
+ setStatus("", false); setBusy(false);
468
+ }
469
+
470
+ // ---- boot panel ------------------------------------------------------------------------------------------------------
471
+ function applyStep(m) {
472
+ const row = document.querySelector('.boot .step[data-step="' + m.name + '"]');
473
+ if (!row) return;
474
+ row.className = "step " + m.state;
475
+ const pct = m.state === "done" ? 100 : (m.pct == null ? 0 : m.pct);
476
+ row.querySelector(".meter i").style.width = pct + "%";
477
+ row.querySelector(".pct").textContent = m.state === "done" ? (m.cached ? "محفوظ" : "تم") : (m.prepare ? "…" : (m.pct == null ? (m.mb ? m.mb + " MB" : "") : pct + "%"));
478
+ if (document.querySelectorAll(".boot .step.done").length === 3) { $("boot").classList.add("ready"); $("boot-title").textContent = "النظام جاهز · يعمل كله داخل متصفحك ولا يُرسل نصك إلى أي خادم"; }
479
+ }
480
+
481
+ function start() {
482
+ try { worker = new Worker(URL.createObjectURL(new Blob([$("worker-src").textContent], { type: "text/javascript" }))); }
483
+ catch (error) { return setStatus("تعذّر تشغيل النظام: " + error.message, false); }
484
+ worker.onerror = (event) => setStatus("تعذّر تشغيل النظام: " + (event.message || ""), false);
485
+ worker.onmessage = (event) => {
486
+ const m = event.data;
487
+ if (m.type === "step") applyStep(m);
488
+ else if (m.type === "error") setStatus("تعذّر تشغيل النظام: " + m.text + ". جرّب متصفح كمبيوتر حديثًا ثم أعد تحميل الصفحة.", false);
489
+ else if (m.type === "result") { const resolve = pending.get(m.id); pending.delete(m.id); resolve(m.html); }
490
+ else if (m.type === "ready") { examples = m.examples; ready = true; store.set("icv:examples:" + BUILD_ID, JSON.stringify(m.examples)); }
491
+ };
492
+ worker.postMessage({ type: "boot", base: location.href, buildId: BUILD_ID, pyodideVersion: "0.26.4", files: ["app.py", "verifier.py", "retrieval.py", "index_builder.py", "normalization.py", "alignment.py", "similarity.py", "detector.py", "scanner.py", "idgham.py", "ui.py", "llm_client.py", "benchmark_format.py", "camelbert_adapter.py", "index/quran.idx.gz", "demo/examples.json"], lazy: ["index/hadith.idx.gz"] });
493
+ }
494
+
495
+ async function loadConfig() {
496
+ try {
497
+ const response = await fetch("config.json", { cache: "no-store" });
498
+ if (response.ok) { cfg = Object.assign({}, DEFAULT_CONFIG, await response.json()); store.set("icv:config", JSON.stringify(cfg)); return; }
499
+ } catch (e) { /* offline or missing: fall back to the last known configuration */ }
500
+ try { const saved = store.get("icv:config"); if (saved) cfg = Object.assign({}, DEFAULT_CONFIG, JSON.parse(saved)); } catch (e) { /* ignore */ }
501
+ }
502
+
503
+ document.querySelectorAll(".tab").forEach((tab) => tab.addEventListener("click", () => {
504
+ document.querySelectorAll(".tab").forEach((t) => { const on = t === tab; t.classList.toggle("active", on); t.setAttribute("aria-selected", on); });
505
+ for (const name of ["direct", "ask"]) $("tab-" + name).hidden = tab.dataset.tab !== name;
506
+ }));
507
+ $("verify").addEventListener("click", runDirect);
508
+ $("ask").addEventListener("click", runAsk);
509
+ $("ask-example").addEventListener("click", runAskExample);
510
+ $("example").addEventListener("click", () => { if (!examples.length) return; $("text").value = examples[exampleIndex++ % examples.length].text; runDirect(); });
511
+
512
+ purgeOldLocalCache();
513
+ try { const saved = store.get("icv:examples:" + BUILD_ID); if (saved) examples = JSON.parse(saved); } catch (e) { examples = []; }
514
+ loadConfig();
515
+ start();
516
+ </script>
517
+ </body>
518
  </html>
index_builder.py ADDED
@@ -0,0 +1,163 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Build the pre-tokenised search indexes shipped in ``index/``.
2
+
3
+ python index_builder.py # writes index/quran.idx.gz and index/hadith.idx.gz
4
+
5
+ An index is a gzip-compressed pickle holding the raw records, their phonetic skeletons and BM25 postings, so the
6
+ application never re-tokenises the corpora at start-up. The pickles are produced locally by this script and read
7
+ back only by this project; never load an index file from an untrusted source.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import gzip
12
+ import json
13
+ import math
14
+ import pickle
15
+ import sys
16
+ import time
17
+ import zlib
18
+ from array import array
19
+ from collections import Counter, defaultdict
20
+ from pathlib import Path
21
+ from typing import Dict, List, Sequence, Tuple
22
+
23
+ from normalization import content_words, normalize_for_matching, normalize_strict, phonetic_key, tokenize
24
+
25
+ INDEX_VERSION = 3
26
+ BM25_K1, BM25_B = 1.2, 0.75
27
+
28
+
29
+ class BM25Index:
30
+ """Okapi BM25 over a postings table ``term -> (doc ids, term frequencies)``."""
31
+
32
+ def __init__(self, postings: Dict[str, Tuple[array, array]], doc_len: array) -> None:
33
+ self.postings, self.doc_len = postings, doc_len
34
+ self.n_docs = len(doc_len)
35
+ self.avg_len = (sum(doc_len) / self.n_docs) if self.n_docs else 1.0
36
+ self.idf = {t: math.log(1 + (self.n_docs - len(d) + 0.5) / (len(d) + 0.5)) for t, (d, _) in postings.items()}
37
+
38
+ def search(self, terms: Sequence[str], top_k: int) -> List[Tuple[int, float]]:
39
+ scores: Dict[int, float] = defaultdict(float)
40
+ for term in set(terms):
41
+ entry = self.postings.get(term)
42
+ if entry is None:
43
+ continue
44
+ idf = self.idf[term]
45
+ for doc, tf in zip(*entry):
46
+ norm = 1 - BM25_B + BM25_B * self.doc_len[doc] / self.avg_len
47
+ scores[doc] += idf * tf * (BM25_K1 + 1) / (tf + BM25_K1 * norm)
48
+ return sorted(scores.items(), key=lambda item: item[1], reverse=True)[:top_k]
49
+
50
+
51
+ def _postings(documents: List[List[str]]) -> Tuple[Dict[str, Tuple[array, array]], array]:
52
+ docs: Dict[str, array] = defaultdict(lambda: array("I"))
53
+ tfs: Dict[str, array] = defaultdict(lambda: array("H"))
54
+ doc_len = array("I")
55
+ for doc_id, tokens in enumerate(documents):
56
+ doc_len.append(len(tokens))
57
+ for term, tf in Counter(tokens).items():
58
+ docs[term].append(doc_id)
59
+ tfs[term].append(min(tf, 65535))
60
+ return {term: (docs[term], tfs[term]) for term in docs}, doc_len
61
+
62
+
63
+ def _read_json(path: Path) -> list:
64
+ opener = gzip.open if path.suffix == ".gz" else open
65
+ with opener(path, "rt", encoding="utf-8") as handle:
66
+ data = json.load(handle)
67
+ if not isinstance(data, list):
68
+ raise ValueError(f"Unexpected corpus format in {path}: expected a JSON list")
69
+ return data
70
+
71
+
72
+
73
+ def anchor_keys(tokens: Sequence[str]) -> List[Tuple[int, int]]:
74
+ """Seed keys for unannounced-quotation search: ``(crc32, position)`` for every phonetic trigram plus two gapped
75
+ trigrams that survive a single substituted or inserted word."""
76
+ keys = [phonetic_key(t) for t in tokens]
77
+ out = []
78
+ for i in range(len(keys) - 2):
79
+ out.append((zlib.crc32(f"{keys[i]} {keys[i + 1]} {keys[i + 2]}".encode()), i))
80
+ if i + 3 < len(keys):
81
+ out.append((zlib.crc32(f"{keys[i]} {keys[i + 1]} _ {keys[i + 3]}".encode()), i))
82
+ out.append((zlib.crc32(f"{keys[i]} _ {keys[i + 2]} {keys[i + 3]}".encode()), i))
83
+ return out
84
+
85
+
86
+ def build_quran_index(path: Path) -> dict:
87
+ records, norm, word_count = [], [], []
88
+ all_index: Dict[str, List[int]] = defaultdict(list)
89
+ by_surah: Dict[int, Dict[int, int]] = defaultdict(dict)
90
+ documents: List[List[str]] = []
91
+ anchors: Dict[int, List[Tuple[int, int]]] = defaultdict(list)
92
+ for entry in _read_json(Path(path)):
93
+ text = (entry.get("ayah_text") or "").strip()
94
+ if not text:
95
+ continue
96
+ idx = len(records)
97
+ records.append({"surah_id": entry.get("surah_id"), "surah_name": entry.get("surah_name", ""),
98
+ "ayah_id": entry.get("ayah_id"), "text": text})
99
+ by_surah[entry.get("surah_id")][entry.get("ayah_id")] = idx
100
+ norm.append(normalize_for_matching(text))
101
+ strict_tokens = tokenize(normalize_strict(text))
102
+ word_count.append(len(strict_tokens))
103
+ for word in set(strict_tokens):
104
+ all_index[word].append(idx)
105
+ documents.append(content_words(norm[-1].split()))
106
+ for key, pos in anchor_keys(norm[-1].split()):
107
+ anchors[key].append((idx, pos))
108
+ postings, doc_len = _postings(documents)
109
+ return {"version": INDEX_VERSION, "records": records, "norm": norm, "word_count": word_count,
110
+ "all_index": dict(all_index), "by_surah": dict(by_surah), "postings": postings, "doc_len": doc_len,
111
+ "anchors": dict(anchors)}
112
+
113
+
114
+ def build_hadith_index(path: Path) -> dict:
115
+ records, documents = [], []
116
+ for entry in _read_json(Path(path)):
117
+ if not entry:
118
+ continue
119
+ matn = (entry.get("Matn") or "").strip() or None
120
+ full = (entry.get("hadithTxt") or "").strip() or None
121
+ if not (matn or full):
122
+ continue
123
+ records.append({"hadithID": entry.get("hadithID"), "book": entry.get("BookID"), "title": entry.get("title"),
124
+ "matn": matn, "full": full})
125
+ words = content_words(normalize_for_matching(full or matn).split())
126
+ if matn and full: # words of the matn that the full text may lack
127
+ extra = set(content_words(normalize_for_matching(matn).split())) - set(words)
128
+ words += sorted(extra)
129
+ documents.append(words)
130
+ postings, doc_len = _postings(documents)
131
+ return {"version": INDEX_VERSION, "records": records, "postings": postings, "doc_len": doc_len}
132
+
133
+
134
+ def save_index(index: dict, path: Path) -> None:
135
+ path = Path(path)
136
+ path.parent.mkdir(parents=True, exist_ok=True)
137
+ with gzip.open(path, "wb", compresslevel=6) as handle:
138
+ pickle.dump(index, handle, protocol=pickle.HIGHEST_PROTOCOL)
139
+
140
+
141
+ def load_index(path: Path) -> dict:
142
+ with gzip.open(path, "rb") as handle:
143
+ index = pickle.load(handle) # produced by save_index() from this project's own data
144
+ if index.get("version") != INDEX_VERSION:
145
+ raise ValueError("Index version mismatch; rebuild with index_builder.py")
146
+ return index
147
+
148
+
149
+ def main() -> None:
150
+ base = Path(__file__).resolve().parent
151
+ out = base / "index"
152
+ for name, builder, source in (("quran", build_quran_index, base / "data" / "quran.json"),
153
+ ("hadith", build_hadith_index, base / "data" / "hadith.json")):
154
+ started = time.time()
155
+ source = source if source.is_file() else source.with_name(source.name + ".gz")
156
+ index = builder(source)
157
+ save_index(index, out / f"{name}.idx.gz")
158
+ size = (out / f"{name}.idx.gz").stat().st_size / 1e6
159
+ print(f"{name}: {len(index['records'])} records -> index/{name}.idx.gz ({size:.1f} MB, {time.time() - started:.1f}s)")
160
+
161
+
162
+ if __name__ == "__main__":
163
+ sys.exit(main())
islamic_unified_dataset.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
islamiceval_dev_subset.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
llm_client.py ADDED
@@ -0,0 +1,96 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Ask-then-verify backend: one pre-configured OpenAI (ChatGPT) client.
2
+
3
+ There is no provider or model choice for the user. The API key is **never** part of the source code, the page or the
4
+ repository (the competition rules forbid secrets in the repository, and a key shipped to a browser is public): it is read
5
+ from the ``OPENAI_API_KEY`` environment variable on the server side. The browser page talks to a same-origin proxy
6
+ (``functions/api/ask.js``, a Cloudflare Pages Function) that holds the key as a platform secret; this module serves the
7
+ local Gradio app and the tests with the same request shapes.
8
+
9
+ Environment: OPENAI_API_KEY (required) OPENAI_MODEL (optional, default below) OPENAI_BASE_URL (optional, tests / gateways)
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ import os
15
+ import urllib.error
16
+ import urllib.request
17
+ from dataclasses import dataclass
18
+ from typing import Dict, Optional, Tuple
19
+
20
+ SYSTEM_PROMPT = (
21
+ "أنت مساعد معرفي في العلوم الإسلامية. أجب بالعربية بإيجاز ودقة. عند الاستشهاد بآية قرآنية أو حديث نبوي اكتب نصه كاملًا "
22
+ "بين علامتي تنصيص مزدوجتين \"...\" بعد عبارة تمهيدية مثل: قال الله تعالى: أو قال رسول الله ﷺ:. "
23
+ "لا تضع بين علامات التنصيص إلا نص الآية أو الحديث، واذكر السورة ورقم الآية أو مصدر الحديث بعد الاقتباس."
24
+ )
25
+
26
+ DEFAULT_MODEL = "gpt-4o-mini"
27
+ DEFAULT_BASE_URL = "https://api.openai.com/v1"
28
+ MAX_PROMPT_CHARS = 1500
29
+
30
+
31
+ class LLMError(RuntimeError):
32
+ """Raised with a user-presentable Arabic message."""
33
+
34
+
35
+ @dataclass
36
+ class LLMSettings:
37
+ api_key: str = ""
38
+ model: str = ""
39
+ base_url: str = "" # override for tests or a compatible gateway
40
+ timeout: float = 60.0
41
+
42
+ @classmethod
43
+ def from_env(cls) -> "LLMSettings":
44
+ return cls(api_key=os.environ.get("OPENAI_API_KEY", ""), model=os.environ.get("OPENAI_MODEL", ""),
45
+ base_url=os.environ.get("OPENAI_BASE_URL", ""))
46
+
47
+
48
+ def build_request(settings: LLMSettings, prompt: str) -> Tuple[str, Dict[str, str], dict]:
49
+ """``(url, headers, json_body)`` for OpenAI chat completions."""
50
+ base = (settings.base_url or DEFAULT_BASE_URL).rstrip("/")
51
+ body = {"model": settings.model.strip() or DEFAULT_MODEL,
52
+ "messages": [{"role": "system", "content": SYSTEM_PROMPT}, {"role": "user", "content": prompt}]}
53
+ headers = {"Content-Type": "application/json", "Authorization": f"Bearer {settings.api_key}"}
54
+ return f"{base}/chat/completions", headers, body
55
+
56
+
57
+ def parse_response(payload: dict) -> str:
58
+ try:
59
+ text = payload["choices"][0]["message"]["content"]
60
+ except (KeyError, IndexError, TypeError) as exc:
61
+ raise LLMError("وصلت استجابة غير متوقعة من النموذج.") from exc
62
+ if not text or not text.strip():
63
+ raise LLMError("لم يُرجع النموذج أي نص.")
64
+ return text.strip()
65
+
66
+
67
+ _HTTP_MESSAGES = {
68
+ 400: "رفض المزوّد الطلب.",
69
+ 401: "خدمة الإجابة غير مهيّأة بعد (المفتاح غير صالح).",
70
+ 403: "الخدمة غير مصرّح لها باستخدام هذا النموذج.",
71
+ 404: "النموذج غير متاح لدى المزوّد.",
72
+ 429: "تجاوزت حد الاستخدام المسموح؛ حاول لاحقًا.",
73
+ }
74
+
75
+
76
+ def generate(settings: Optional[LLMSettings], prompt: str) -> str:
77
+ """Return the model's answer. Raises ``LLMError`` with an Arabic message on any failure."""
78
+ settings = settings or LLMSettings.from_env()
79
+ if not prompt or not prompt.strip():
80
+ raise LLMError("اكتب سؤالًا أولًا.")
81
+ if len(prompt) > MAX_PROMPT_CHARS:
82
+ raise LLMError(f"السؤال طويل جدًا (الحد الأقصى {MAX_PROMPT_CHARS} حرف).")
83
+ if not settings.api_key or not settings.api_key.strip():
84
+ raise LLMError("خدمة الإجابة غير مهيّأة: لم يُضبط مفتاح الخادم.")
85
+ url, headers, body = build_request(settings, prompt.strip())
86
+ request = urllib.request.Request(url, data=json.dumps(body).encode("utf-8"), headers=headers, method="POST")
87
+ try:
88
+ with urllib.request.urlopen(request, timeout=settings.timeout) as response:
89
+ payload = json.loads(response.read().decode("utf-8"))
90
+ except urllib.error.HTTPError as exc:
91
+ raise LLMError(_HTTP_MESSAGES.get(exc.code, f"فشل الطلب (الرمز {exc.code}).")) from exc
92
+ except (urllib.error.URLError, TimeoutError) as exc:
93
+ raise LLMError("تعذّر الاتصال بالمزوّد؛ تحقق من الإنترنت.") from exc
94
+ except json.JSONDecodeError as exc:
95
+ raise LLMError("وصلت استجابة غير مقروءة من المزوّد.") from exc
96
+ return parse_response(payload)
normalization.py ADDED
@@ -0,0 +1,117 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Phonetic-aware Arabic normalisation shared by indexing, retrieval, alignment and verification.
2
+
3
+ Three levels, from gentle to aggressive:
4
+
5
+ * ``normalize_strict`` diacritics / tatweel / Quranic marks removed, alef forms unified (hamza on waw / ya kept).
6
+ * ``normalize_lenient`` strict + ta marbuta -> ha and alef maqsura -> ya (used for Hadith, whose spelling varies).
7
+ * ``normalize_for_matching`` the phonetic skeleton used for retrieval and word alignment: additionally folds hamza
8
+ carriers and keeps Arabic letters only.
9
+
10
+ All levels also reconcile the Uthmani mushaf script with Modern Standard Arabic: alef wasla (ٱ) becomes alef, the
11
+ dagger alef / "alif khanjariyah" (ـٰ) and tatweel are removed, Persian ya / kaf and the ligature ﷲ are mapped to Arabic
12
+ letters, and zero-width / bidi control characters are dropped.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import re
17
+ import unicodedata
18
+ from typing import List, Tuple
19
+
20
+ _ZERO_WIDTH = re.compile("[\u200b-\u200f\u202a-\u202e\u2066-\u2069\ufeff]")
21
+ _MARKS = re.compile(r"[\u0610-\u061A\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E4\u06E7\u06E8\u06EA-\u06ED\u0640]")
22
+ _MATCH_MARKS = re.compile(r"[\u0610-\u061A\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E8\u06EA-\u06ED\u0640]")
23
+ _PUNCTUATION = re.compile(r"[،؛؟!،.,:;'\"()\[\]{}<>«»\-_/\\|@#$%^&*+=~`﴿﴾]")
24
+ _SPACES = re.compile(r"\s+")
25
+ _ALEF_FORMS = re.compile(r"[أإآٱ]")
26
+ _PERSIAN = str.maketrans({"ی": "ي", "ې": "ي", "ک": "ك", "ە": "ه", "ہ": "ه", "ۀ": "ه"})
27
+
28
+ STOPWORDS = frozenset(
29
+ """من في على ان أن إن الى إلى عن مع ما لا لم لن قد و ثم أو او هو هي هم انت أنتم كان كانت يكون تكون قال قالت
30
+ هذا هذه ذلك تلك الذي التي الذين اللاتي اللائي كل بعض غير عند بين حتى إذا اذا لو لكن بل يا أيها ايها""".split()
31
+ )
32
+
33
+
34
+ def _prefold(text: str) -> str:
35
+ """Script-level clean-up common to every normalisation level."""
36
+ text = unicodedata.normalize("NFC", text).replace("ﷲ", "الله")
37
+ return _ZERO_WIDTH.sub("", text).translate(_PERSIAN)
38
+
39
+
40
+ def normalize_strict(text) -> str:
41
+ if not text:
42
+ return ""
43
+ text = _MARKS.sub("", _prefold(text))
44
+ text = _ALEF_FORMS.sub("ا", text)
45
+ return _SPACES.sub(" ", _PUNCTUATION.sub(" ", text)).strip()
46
+
47
+
48
+ def normalize_lenient(text) -> str:
49
+ return normalize_strict(text).replace("ة", "ه").replace("ى", "ي")
50
+
51
+
52
+ def normalize_for_matching(text) -> str:
53
+ """Phonetic skeleton: Arabic letters only, hamza carriers / ta marbuta / alef maqsura folded."""
54
+ if not text:
55
+ return ""
56
+ text = _MATCH_MARKS.sub("", _prefold(text))
57
+ for source, target in (("أ", "ا"), ("إ", "ا"), ("آ", "ا"), ("ٱ", "ا"), ("ؤ", "و"), ("ئ", "ي"), ("ة", "ه"), ("ى", "ي")):
58
+ text = text.replace(source, target)
59
+ text = re.sub(r"[^\u0621-\u064A\s]", " ", text) # Arabic letters only: drops Arabic punctuation, digits, Latin
60
+ return _SPACES.sub(" ", text).strip()
61
+
62
+
63
+ def tokenize(text: str) -> List[str]:
64
+ return [token for token in text.split() if token]
65
+
66
+
67
+ def content_words(tokens) -> List[str]:
68
+ """Drop stop-words and single-letter tokens."""
69
+ return [t for t in tokens if t not in STOPWORDS and len(t) > 1]
70
+
71
+
72
+ def char_ngrams(text: str, n: int = 4) -> set:
73
+ return {text[i : i + n] for i in range(len(text) - n + 1)}
74
+
75
+
76
+ # ---- word-level helpers used by the aligner -----------------------------------------------------------------------
77
+ _ARABIC_LETTER = re.compile(r"[\u0621-\u064A]")
78
+
79
+
80
+ def aligned_words(text: str) -> List[Tuple[str, str]]:
81
+ """``[(original_word, phonetic_skeleton)]``; ayah markers like ``(12)`` and mark-only tokens are dropped."""
82
+ pairs = []
83
+ for word in text.split():
84
+ skeleton = normalize_for_matching(word)
85
+ if skeleton and _ARABIC_LETTER.search(skeleton):
86
+ pairs.append((word, skeleton.replace(" ", "")))
87
+ return pairs
88
+
89
+
90
+ _VOWELS = {"\u064E": "a", "\u064F": "u", "\u0650": "i", "\u064B": "A", "\u064C": "U", "\u064D": "I", "\u0651": "~"}
91
+
92
+
93
+ def vowel_signature(word: str) -> List[Tuple[str, str]]:
94
+ """Per base letter, the set of short vowels / tanween / shadda written on it (sukun and Quranic marks ignored)."""
95
+ word = _prefold(word).replace("ٱ", "ا")
96
+ signature: List[Tuple[str, str]] = []
97
+ for char in word:
98
+ if char in _VOWELS:
99
+ if signature:
100
+ letter, marks = signature[-1]
101
+ signature[-1] = (letter, "".join(sorted(set(marks + _VOWELS[char]))))
102
+ elif "\u0621" <= char <= "\u064A":
103
+ signature.append((char, ""))
104
+ return signature
105
+
106
+
107
+ # ---- phonetic sequence matching --------------------------------------------------------------------------------
108
+ # Letters that are commonly confused in writing or dictation are folded into one class, so an altered or mis-spelled
109
+ # quotation can still be *found*; the verifier then reports the exact differences.
110
+ _PHONETIC_CLASSES = {"ص": "س", "ث": "س", "ذ": "ز", "ظ": "ز", "ض": "ز", "ط": "ت", "ك": "ق", "ح": "ه", "غ": "ع"}
111
+ _PHONETIC_TABLE = str.maketrans(_PHONETIC_CLASSES)
112
+
113
+
114
+ def phonetic_key(skeleton_word: str) -> str:
115
+ """Sound-alike key of a phonetic-skeleton word (confusable consonants folded, doubled letters collapsed)."""
116
+ word = skeleton_word.translate(_PHONETIC_TABLE)
117
+ return "".join(ch for i, ch in enumerate(word) if i == 0 or ch != word[i - 1])
quran.idx.gz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:87eb1db1dffa15d95d757b157a1569cad93e6265c15736c9f4ee62c17b2a2c98
3
+ size 2181000
quran.json ADDED
The diff for this file is too large to render. See raw diff
 
retrieval.py ADDED
@@ -0,0 +1,187 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Source retrieval over the Quran and the six Hadith books: BM25 candidate search with pre-built, lazily loaded indexes.
2
+
3
+ Start-up cost is kept small by shipping the corpora as pre-tokenised, gzip-compressed pickle indexes
4
+ (``index/quran.idx.gz``, ``index/hadith.idx.gz``, produced by ``index_builder.py``):
5
+
6
+ * the small Quran index loads with the retriever;
7
+ * the large Hadith index loads on first use (or via ``warm()``);
8
+ * if an index file is missing it is rebuilt from ``data/*.json`` (slower, a few seconds) so the code never breaks.
9
+
10
+ Retrieval pipeline
11
+ Quran : word-vote F1 (coverage x precision, exact-quote friendly) + BM25 candidates
12
+ Hadith: BM25 recall, then character 4-gram cosine re-rank
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import logging
17
+ import threading
18
+ from collections import OrderedDict
19
+ from pathlib import Path
20
+ from typing import Dict, Iterable, List, Optional, Sequence, Tuple
21
+
22
+ from index_builder import BM25Index, build_hadith_index, build_quran_index, load_index, save_index
23
+ from normalization import char_ngrams, content_words, normalize_for_matching, normalize_lenient, normalize_strict, tokenize
24
+
25
+ logger = logging.getLogger(__name__)
26
+
27
+ BASE_DIR = Path(__file__).resolve().parent
28
+ DATA_DIR = BASE_DIR / "data"
29
+ INDEX_DIR = BASE_DIR / "index"
30
+
31
+
32
+ class CorpusError(RuntimeError):
33
+ """Raised when a corpus or index file is missing or malformed."""
34
+
35
+
36
+ class SourceRetriever:
37
+ """Quran + Hadith retriever. ``quran`` is available immediately; ``hadith`` is loaded lazily."""
38
+
39
+ def __init__(self, index_dir: Path = INDEX_DIR, data_dir: Path = DATA_DIR) -> None:
40
+ self.index_dir, self.data_dir = Path(index_dir), Path(data_dir)
41
+ self._hadith_lock = threading.Lock()
42
+ self._hadith: Optional[dict] = None
43
+ self._norm_cache: "OrderedDict[Tuple[int, str], str]" = OrderedDict()
44
+
45
+ quran = self._load("quran", build_quran_index)
46
+ self.quran: List[dict] = quran["records"]
47
+ self.q_norm_match: List[str] = quran["norm"]
48
+ self.q_word_count: List[int] = quran["word_count"]
49
+ self.q_all_index: Dict[str, List[int]] = quran["all_index"]
50
+ self.quran_by_surah: Dict[int, Dict[int, int]] = quran["by_surah"]
51
+ self.quran_bm25 = BM25Index(quran["postings"], quran["doc_len"])
52
+ self.quran_anchors: Dict[int, List[Tuple[int, int]]] = quran["anchors"]
53
+ logger.info("Quran index ready: %d ayahs", len(self.quran))
54
+
55
+ # ---- loading ----------------------------------------------------------------------------------------------
56
+ def _load(self, name: str, builder) -> dict:
57
+ path = self.index_dir / f"{name}.idx.gz"
58
+ if path.is_file():
59
+ try:
60
+ return load_index(path)
61
+ except Exception:
62
+ logger.warning("Index %s is unreadable; rebuilding from data/", path)
63
+ source = self.data_dir / ("quran.json" if name == "quran" else "hadith.json")
64
+ gz = source.with_name(source.name + ".gz")
65
+ source = source if source.is_file() else gz
66
+ if not source.is_file():
67
+ raise CorpusError(f"Neither {path} nor the source corpus {source} was found")
68
+ index = builder(source)
69
+ try:
70
+ save_index(index, path)
71
+ except OSError:
72
+ logger.info("Could not cache %s (read-only file system); continuing in memory", path)
73
+ return index
74
+
75
+ def warm(self) -> None:
76
+ """Load the Hadith index now (otherwise it loads on the first Hadith query)."""
77
+ _ = self.hadith_data
78
+
79
+ @property
80
+ def hadith_loaded(self) -> bool:
81
+ return self._hadith is not None
82
+
83
+ @property
84
+ def hadith_data(self) -> dict:
85
+ if self._hadith is None:
86
+ with self._hadith_lock:
87
+ if self._hadith is None:
88
+ data = self._load("hadith", build_hadith_index)
89
+ data["bm25"] = BM25Index(data["postings"], data["doc_len"])
90
+ self._hadith = data
91
+ logger.info("Hadith index ready: %d records", len(data["records"]))
92
+ return self._hadith
93
+
94
+ @property
95
+ def hadith(self) -> List[dict]:
96
+ return self.hadith_data["records"]
97
+
98
+ # ---- Quran ------------------------------------------------------------------------------------------------
99
+ @property
100
+ def quran_vocabulary(self) -> frozenset:
101
+ """Phonetic-skeleton words that occur in the Quran (lets the aligner tell spelling variants from real words)."""
102
+ if getattr(self, "_vocab", None) is None:
103
+ self._vocab = frozenset(word for text in self.q_norm_match for word in text.split())
104
+ return self._vocab
105
+
106
+ def search_quran_ayahs(self, query: str, top_k: int = 25, extra_bm25: int = 10) -> List[dict]:
107
+ """Single-ayah candidates: word-vote F1 first, then the best BM25 matches (rare-word hits for partial quotes)."""
108
+ query_words = tokenize(normalize_strict(query))
109
+ if not query_words:
110
+ return []
111
+ votes: Dict[int, int] = {}
112
+ for word in query_words:
113
+ for idx in self.q_all_index.get(word, ()):
114
+ votes[idx] = votes.get(idx, 0) + 1
115
+ scored: List[Tuple[int, float]] = []
116
+ for idx, vote in votes.items():
117
+ coverage = vote / len(query_words)
118
+ precision = vote / self.q_word_count[idx] if self.q_word_count[idx] else 0.0
119
+ f1 = 2 * coverage * precision / (coverage + precision) if coverage + precision > 0 else 0.0
120
+ scored.append((idx, f1))
121
+ scored.sort(key=lambda item: item[1], reverse=True)
122
+ scores = dict(scored)
123
+ ranked = [idx for idx, _ in scored[:top_k]]
124
+ seen = set(ranked)
125
+ bm25_hits = self.quran_bm25.search(content_words(normalize_for_matching(query).split()), extra_bm25)
126
+ ranked += [idx for idx, _ in bm25_hits if idx not in seen]
127
+ results = []
128
+ for idx in ranked:
129
+ candidate = dict(self.quran[idx])
130
+ candidate.update(type="Quran", retrieval_score=scores.get(idx, 0.0))
131
+ results.append(candidate)
132
+ return results
133
+
134
+ def quran_seed_ayahs(self, query_words: Sequence[str], top_k: int = 25) -> List[int]:
135
+ """BM25-ranked ayah indices used to seed the multi-ayah window search."""
136
+ return [idx for idx, _ in self.quran_bm25.search(query_words, top_k)]
137
+
138
+ # ---- Hadith -----------------------------------------------------------------------------------------------
139
+ def hadith_candidates(self, query_words: Sequence[str], top_k: int) -> List[int]:
140
+ return [idx for idx, _ in self.hadith_data["bm25"].search(query_words, top_k)]
141
+
142
+ def hadith_norm(self, idx: int, field: str) -> Optional[str]:
143
+ """Phonetic skeleton of a Hadith field (``matn`` or ``full``), computed on demand and cached."""
144
+ record = self.hadith[idx]
145
+ raw = record.get(field)
146
+ if not raw:
147
+ return None
148
+ key = (idx, field)
149
+ if key in self._norm_cache:
150
+ self._norm_cache.move_to_end(key)
151
+ return self._norm_cache[key]
152
+ value = normalize_for_matching(raw)
153
+ self._norm_cache[key] = value
154
+ if len(self._norm_cache) > 4096:
155
+ self._norm_cache.popitem(last=False)
156
+ return value
157
+
158
+ def search_hadith(self, query: str, top_k: int = 15, pool: int = 60) -> List[dict]:
159
+ """BM25 recall, then character 4-gram cosine re-rank."""
160
+ words = content_words(normalize_for_matching(query).split())
161
+ if not words:
162
+ return []
163
+ query_grams = char_ngrams(normalize_lenient(query))
164
+ results = []
165
+ for idx in self.hadith_candidates(words, pool):
166
+ record = self.hadith[idx]
167
+ text = record["matn"] or record["full"]
168
+ doc_grams = char_ngrams(normalize_lenient(text))
169
+ cosine = (
170
+ len(query_grams & doc_grams) / ((len(query_grams) * len(doc_grams)) ** 0.5)
171
+ if query_grams and doc_grams
172
+ else 0.0
173
+ )
174
+ results.append(
175
+ {
176
+ "type": "Hadith",
177
+ "idx": idx,
178
+ "hadithID": record["hadithID"],
179
+ "book": record["book"],
180
+ "title": record["title"],
181
+ "text": text,
182
+ "has_matn": bool(record["matn"]),
183
+ "retrieval_score": cosine,
184
+ }
185
+ )
186
+ results.sort(key=lambda c: c["retrieval_score"], reverse=True)
187
+ return results[:top_k]
scanner.py ADDED
@@ -0,0 +1,349 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Detection of unannounced and embedded quotations (no quotation marks, no introductory phrase needed).
2
+
3
+ The scanner reads the text as a stream of phonetic-skeleton words and looks for stretches that follow the Quran or a
4
+ Hadith closely:
5
+
6
+ 1. *Seeds* a sliding window of 3 words (plus two gapped variants that survive one changed word) is looked up in the
7
+ pre-built phonetic n-gram index of the Quran.
8
+ 2. *Chaining* seeds of the same ayah on a consistent diagonal are chained into a candidate region; regions of adjacent
9
+ ayahs that touch in the text are merged.
10
+ 3. *Boundaries* the region grows word by word (tolerating one substituted word) while the text keeps following the
11
+ source, and never crosses a sentence boundary on its own; this is the contextual boundary step.
12
+ 4. *Evidence* a region is kept only if enough of its words match and the matched words are rare enough (summed IDF), so
13
+ everyday phrases that merely occur in the Quran are not reported.
14
+ 5. *Hadith* clause-sized windows are sent to BM25, and the best record is aligned word by word with the same rules.
15
+
16
+ Phonetic keys make the search tolerant to sound-alike spelling, but the verifier still reports every real difference.
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import re
21
+ from collections import defaultdict
22
+ from dataclasses import dataclass
23
+ from difflib import SequenceMatcher
24
+ from typing import Dict, List, Optional, Sequence, Tuple
25
+
26
+ from alignment import aligned_words
27
+ from detector import DetectedSpan, RuleDetector, trim_span
28
+ from index_builder import anchor_keys
29
+ from normalization import content_words, normalize_for_matching, phonetic_key
30
+ from retrieval import SourceRetriever
31
+
32
+ _ARABIC = re.compile(r"[\u0621-\u064A]")
33
+ _STRONG_BOUNDARY = re.compile(r"[.؟?!؛;:\n…]")
34
+
35
+
36
+ @dataclass
37
+ class Token:
38
+ skeleton: str
39
+ key: str
40
+ start: int
41
+ end: int
42
+ boundary_after: bool # a sentence-level punctuation mark or line break follows this word
43
+
44
+
45
+ def tokenize_with_offsets(text: str) -> List[Token]:
46
+ matches = list(re.finditer(r"\S+", text))
47
+ tokens: List[Token] = []
48
+ for i, match in enumerate(matches):
49
+ skeleton = normalize_for_matching(match.group()).replace(" ", "")
50
+ if not skeleton or not _ARABIC.search(skeleton):
51
+ if tokens and _STRONG_BOUNDARY.search(match.group()):
52
+ tokens[-1].boundary_after = True
53
+ continue
54
+ gap_end = matches[i + 1].start() if i + 1 < len(matches) else len(text)
55
+ trailing = text[match.end():gap_end] + match.group()[-2:]
56
+ tokens.append(Token(skeleton, phonetic_key(skeleton), match.start(), match.end(),
57
+ bool(_STRONG_BOUNDARY.search(trailing))))
58
+ return tokens
59
+
60
+
61
+ @dataclass
62
+ class _Region:
63
+ start: int # token index in the run (inclusive)
64
+ end: int # exclusive
65
+ label: str
66
+ matched: int
67
+ ratio: float
68
+ idf: float
69
+ surah: Optional[int] = None
70
+ first_ayah: Optional[int] = None
71
+ last_ayah: Optional[int] = None
72
+
73
+
74
+ class CorpusScanner:
75
+ """Finds Quran / Hadith stretches in free text. Thresholds are deliberately conservative."""
76
+
77
+ def __init__(self, retriever: SourceRetriever, min_tokens: int = 4, min_ratio: float = 0.7, min_idf: float = 8.0,
78
+ hadith_min_tokens: int = 6, hadith_min_idf: float = 14.0, scan_hadith: bool = True) -> None:
79
+ self.kb = retriever
80
+ self.min_tokens, self.min_ratio, self.min_idf = min_tokens, min_ratio, min_idf
81
+ self.hadith_min_tokens, self.hadith_min_idf = hadith_min_tokens, hadith_min_idf
82
+ self.scan_hadith = scan_hadith
83
+
84
+ # ---- public -----------------------------------------------------------------------------------------------
85
+ def scan(self, text: str, exclude: Sequence[Tuple[int, int]] = ()) -> List[DetectedSpan]:
86
+ tokens = tokenize_with_offsets(text)
87
+ runs = self._runs(tokens, exclude)
88
+ spans: List[DetectedSpan] = []
89
+ for run in runs:
90
+ for region in self._merge(self._scan_quran(run)):
91
+ spans.append(self._to_span(text, run, region))
92
+ if self.scan_hadith:
93
+ for run in runs:
94
+ taken = [(s.start, s.end) for s in spans]
95
+ for region in self._scan_hadith(run, taken):
96
+ spans.append(self._to_span(text, run, region))
97
+ return sorted(spans, key=lambda s: s.start)
98
+
99
+ # ---- helpers ----------------------------------------------------------------------------------------------
100
+ @staticmethod
101
+ def _runs(tokens: List[Token], exclude: Sequence[Tuple[int, int]]) -> List[List[Token]]:
102
+ runs, current = [], []
103
+ for token in tokens:
104
+ if any(token.start < e and token.end > s for s, e in exclude):
105
+ if current:
106
+ runs.append(current)
107
+ current = []
108
+ else:
109
+ current.append(token)
110
+ if current:
111
+ runs.append(current)
112
+ return runs
113
+
114
+ @staticmethod
115
+ def _to_span(text: str, run: List[Token], region: _Region) -> DetectedSpan:
116
+ start, end = trim_span(text, run[region.start].start, run[region.end - 1].end)
117
+ return DetectedSpan(start, end, region.label, round(region.ratio, 3), "scan", text[start:end])
118
+
119
+ # ---- Quran ------------------------------------------------------------------------------------------------
120
+ def _scan_quran(self, run: List[Token]) -> List[_Region]:
121
+ kb = self.kb
122
+ if len(run) < self.min_tokens:
123
+ return []
124
+ hits: Dict[int, List[Tuple[int, int]]] = defaultdict(list)
125
+ for key_hash, pos in anchor_keys([t.skeleton for t in run]):
126
+ postings = kb.quran_anchors.get(key_hash)
127
+ if not postings or len(postings) > 60: # skip phrases that occur everywhere
128
+ continue
129
+ for ayah, source_pos in postings:
130
+ hits[ayah].append((pos, source_pos))
131
+
132
+ regions: List[_Region] = self._whole_ayahs(run)
133
+ for ayah, found in hits.items():
134
+ found.sort()
135
+ chains: List[dict] = []
136
+ for pos, source_pos in found:
137
+ diagonal = source_pos - pos
138
+ for chain in chains:
139
+ if abs(diagonal - chain["d"]) <= 2 and pos - chain["last"] <= 5:
140
+ chain["hits"].append((pos, source_pos))
141
+ chain["last"], chain["d"] = pos, diagonal
142
+ break
143
+ else:
144
+ chains.append({"d": diagonal, "last": pos, "hits": [(pos, source_pos)]})
145
+ for chain in chains:
146
+ region = self._grow(run, ayah, chain)
147
+ if region is not None:
148
+ regions.append(region)
149
+ return self._non_overlapping(regions)
150
+
151
+ def _whole_ayah_index(self) -> Dict[str, list]:
152
+ """first word -> [(words, ayah index)] for every ayah of at least ``min_tokens`` words (built once)."""
153
+ if getattr(self, "_whole", None) is None:
154
+ index: Dict[str, list] = defaultdict(list)
155
+ for i, norm in enumerate(self.kb.q_norm_match):
156
+ words = tuple(w for w in norm.split() if w)
157
+ if len(words) >= self.min_tokens:
158
+ index[words[0]].append((words, i))
159
+ for entries in index.values():
160
+ entries.sort(key=lambda e: -len(e[0])) # longest first
161
+ self._whole = index
162
+ return self._whole
163
+
164
+ def _whole_ayahs(self, run: List[Token]) -> List[_Region]:
165
+ """A complete ayah typed as it is: found by exact word sequence, whatever the rarity of its words (so a short
166
+ ayah made of common words, like ``قل هو الله احد``, is not missed)."""
167
+ index, skeletons, found, i = self._whole_ayah_index(), [t.skeleton for t in run], [], 0
168
+ while i < len(run):
169
+ for words, ayah in index.get(skeletons[i], ()):
170
+ n = len(words)
171
+ if tuple(skeletons[i:i + n]) == words and not any(t.boundary_after for t in run[i:i + n - 1]):
172
+ record = self.kb.quran[ayah]
173
+ found.append(_Region(i, i + n, "Ayah", n, 1.0, 99.0, record["surah_id"], record["ayah_id"], record["ayah_id"]))
174
+ i += n - 1
175
+ break
176
+ i += 1
177
+ return found
178
+
179
+ def _grow(self, run: List[Token], ayah: int, chain: dict) -> Optional[_Region]:
180
+ kb = self.kb
181
+ source = kb.q_norm_match[ayah].split()
182
+ source_keys = [phonetic_key(w) for w in source]
183
+ diagonal = sorted(source_pos - pos for pos, source_pos in chain["hits"])[len(chain["hits"]) // 2]
184
+ start = min(pos for pos, _ in chain["hits"])
185
+ end = min(len(run), max(pos for pos, _ in chain["hits"]) + 3)
186
+
187
+ def key_at(i: int) -> Optional[str]:
188
+ return source_keys[i + diagonal] if 0 <= i + diagonal < len(source_keys) else None
189
+
190
+ while start > 0 and not run[start - 1].boundary_after: # grow left while the text keeps following the source
191
+ if run[start - 1].key == key_at(start - 1):
192
+ start -= 1
193
+ elif start >= 2 and not run[start - 2].boundary_after and run[start - 2].key == key_at(start - 2):
194
+ start -= 2 # one substituted word
195
+ else:
196
+ break
197
+ while end < len(run) and not run[end - 1].boundary_after:
198
+ if run[end].key == key_at(end):
199
+ end += 1
200
+ elif end + 1 < len(run) and run[end + 1].key == key_at(end + 1):
201
+ end += 2
202
+ else:
203
+ break
204
+
205
+ start = self._soft_left(run, start, diagonal, source_keys)
206
+ end = self._soft_right(run, end, diagonal, source_keys)
207
+ positions = range(start, end)
208
+ matched_idx = [i for i in positions if run[i].key == key_at(i)]
209
+ matched = len(matched_idx)
210
+ n = end - start
211
+ idf = sum(kb.quran_bm25.idf.get(run[i].skeleton, 0.0) for i in matched_idx)
212
+ ratio = matched / n if n else 0.0
213
+ needed_idf = self.min_idf if n >= 5 else self.min_idf + 6.0 # very short stretches must be rare phrases
214
+ if n < self.min_tokens or matched < self.min_tokens or ratio < self.min_ratio or idf < needed_idf:
215
+ return None
216
+ record = kb.quran[ayah]
217
+ return _Region(start, end, "Ayah", matched, ratio, idf, record["surah_id"], record["ayah_id"], record["ayah_id"])
218
+
219
+ @staticmethod
220
+ def _closest(run: List[Token], candidates, target: str):
221
+ """Among candidate token ranges, the one whose joined phonetic key best resembles ``target``; near-ties prefer the
222
+ longer range (words next to a changed word usually belong to the same altered quotation)."""
223
+ scored = [(SequenceMatcher(None, "".join(t.key for t in run[a:b]), target).ratio(), a, b) for a, b in candidates]
224
+ if not scored:
225
+ return None
226
+ top = max(score for score, _, _ in scored)
227
+ if top < 0.6:
228
+ return None
229
+ return max((c for c in scored if c[0] >= top - 0.2), key=lambda c: c[2] - c[1])
230
+
231
+ def _soft_left(self, run: List[Token], start: int, diagonal: int, source_keys: List[str]) -> int:
232
+ """Pull in up to two words before the region that look like the source words missing at its beginning."""
233
+ missing = min(start + diagonal, 2)
234
+ if missing <= 0:
235
+ return start
236
+ target = "".join(source_keys[start + diagonal - missing : start + diagonal])
237
+ candidates = [(start - n, start) for n in range(max(1, missing - 1), missing + 2)
238
+ if start - n >= 0 and not any(run[i].boundary_after for i in range(start - n, start))]
239
+ best = self._closest(run, candidates, target)
240
+ return best[1] if best else start
241
+
242
+ def _soft_right(self, run: List[Token], end: int, diagonal: int, source_keys: List[str]) -> int:
243
+ missing = min(len(source_keys) - (end + diagonal), 2)
244
+ if missing <= 0 or end >= len(run) or run[end - 1].boundary_after:
245
+ return end
246
+ target = "".join(source_keys[end + diagonal : end + diagonal + missing])
247
+ candidates = [(end, end + n) for n in range(max(1, missing - 1), missing + 2)
248
+ if end + n <= len(run) and not any(run[i].boundary_after for i in range(end, end + n - 1))]
249
+ best = self._closest(run, candidates, target)
250
+ return best[2] if best else end
251
+
252
+ @staticmethod
253
+ def _non_overlapping(regions: List[_Region]) -> List[_Region]:
254
+ chosen: List[_Region] = []
255
+ for region in sorted(regions, key=lambda r: (r.matched, r.ratio), reverse=True):
256
+ if all(region.end <= c.start or region.start >= c.end for c in chosen):
257
+ chosen.append(region)
258
+ return sorted(chosen, key=lambda r: r.start)
259
+
260
+ @staticmethod
261
+ def _merge(regions: List[_Region]) -> List[_Region]:
262
+ """Join regions of consecutive ayahs that follow each other in the text (a quotation spanning several ayahs)."""
263
+ merged: List[_Region] = []
264
+ for region in regions:
265
+ last = merged[-1] if merged else None
266
+ if (last and last.surah == region.surah and region.first_ayah == last.last_ayah + 1
267
+ and region.start - last.end <= 1):
268
+ last.end, last.matched = region.end, last.matched + region.matched
269
+ last.ratio = last.matched / (last.end - last.start)
270
+ last.idf += region.idf
271
+ last.last_ayah = region.last_ayah
272
+ else:
273
+ merged.append(region)
274
+ return merged
275
+
276
+ # ---- Hadith -----------------------------------------------------------------------------------------------
277
+ def _scan_hadith(self, run: List[Token], taken: Sequence[Tuple[int, int]]) -> List[_Region]:
278
+ kb = self.kb
279
+ regions: List[_Region] = []
280
+ segment_start = 0
281
+ for i, token in enumerate(run):
282
+ if token.boundary_after or i == len(run) - 1:
283
+ segment = (segment_start, i + 1)
284
+ segment_start = i + 1
285
+ if segment[1] - segment[0] < self.hadith_min_tokens:
286
+ continue
287
+ for a, b in self._windows(*segment):
288
+ region = self._match_hadith(run, a, b)
289
+ if region is not None:
290
+ regions.append(region)
291
+ return self._non_overlapping(regions)
292
+
293
+ @staticmethod
294
+ def _windows(start: int, end: int, size: int = 24, stride: int = 12):
295
+ if end - start <= 40:
296
+ yield start, end
297
+ else:
298
+ for a in range(start, end - 6, stride):
299
+ yield a, min(end, a + size)
300
+
301
+ def _match_hadith(self, run: List[Token], a: int, b: int) -> Optional[_Region]:
302
+ kb = self.kb
303
+ window = run[a:b]
304
+ words = content_words(t.skeleton for t in window)
305
+ if len(words) < 4:
306
+ return None
307
+ window_keys = [t.key for t in window]
308
+ best = None
309
+ for idx in kb.hadith_candidates(words, 3):
310
+ record = kb.hadith[idx]
311
+ source = [p[1] for p in aligned_words(record["matn"] or record["full"])]
312
+ matcher = SequenceMatcher(None, window_keys, [phonetic_key(w) for w in source], autojunk=False)
313
+ blocks = [blk for blk in matcher.get_matching_blocks() if blk.size >= 3]
314
+ matched = sum(blk.size for blk in blocks)
315
+ if matched >= self.hadith_min_tokens and (best is None or matched > best[0]):
316
+ best = (matched, blocks)
317
+ if best is None:
318
+ return None
319
+ matched, blocks = best
320
+ start, end = blocks[0].a, blocks[-1].a + blocks[-1].size
321
+ ratio = matched / (end - start)
322
+ idf = sum(kb.hadith_data["bm25"].idf.get(window[i].skeleton, 0.0) for blk in blocks for i in range(blk.a, blk.a + blk.size))
323
+ if ratio < self.min_ratio or idf < self.hadith_min_idf:
324
+ return None
325
+ return _Region(a + start, a + end, "Hadith", matched, ratio, idf)
326
+
327
+
328
+ class HybridDetector:
329
+ """Rule-based detection first (quotation marks / brackets, typed by the corpora; phrases are only a hint), then the corpus scanner on the rest of the
330
+ text to find unannounced and embedded quotations. Rule spans always win overlaps: an author-delimited quotation is
331
+ verified exactly as written."""
332
+
333
+ def __init__(self, retriever: SourceRetriever, use_scanner: bool = True, rules_use_corpus: bool = True,
334
+ decouple_triggers: bool = True) -> None:
335
+ """``rules_use_corpus`` / ``decouple_triggers``: type delimited quotations from the corpora instead of from
336
+ introductory phrases (set both to False to reproduce the earlier trigger-driven behaviour for ablations)."""
337
+ self.rules = RuleDetector(retriever if rules_use_corpus else None, decouple_triggers=decouple_triggers)
338
+ self.scanner = CorpusScanner(retriever) if use_scanner else None
339
+
340
+ def detect(self, text: str) -> List[DetectedSpan]:
341
+ spans = self.rules.detect(text)
342
+ if self.scanner is not None:
343
+ found = sorted(self.scanner.scan(text, [(s.start, s.end) for s in spans]), key=lambda s: s.end - s.start, reverse=True)
344
+ kept: List[DetectedSpan] = []
345
+ for span in found: # the same words can follow both corpora: keep the longer span, never report one text twice
346
+ if all(span.end <= k.start or span.start >= k.end for k in kept):
347
+ kept.append(span)
348
+ spans += kept
349
+ return sorted(spans, key=lambda s: s.start)
similarity.py ADDED
@@ -0,0 +1,121 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Similarity indicators between a quotation and a candidate source (the signals behind the verification score)."""
2
+ from __future__ import annotations
3
+
4
+ import re
5
+ import unicodedata
6
+ from typing import Dict, List
7
+
8
+ from alignment import best_region, lcs_length
9
+ from normalization import normalize_lenient, normalize_strict, tokenize
10
+
11
+ try: # optional accelerator; identical formula (1 - distance / max_len)
12
+ from rapidfuzz.distance import Levenshtein as _RapidLevenshtein
13
+ except ImportError: # pragma: no cover
14
+ _RapidLevenshtein = None
15
+
16
+
17
+ def _edit_similarity(a: str, b: str, max_len: int = 600) -> float:
18
+ """Normalised Levenshtein similarity in [0, 1] on the first ``max_len`` characters."""
19
+ a, b = a[:max_len], b[:max_len]
20
+ if a == b:
21
+ return 1.0
22
+ if not a or not b:
23
+ return 0.0
24
+ if _RapidLevenshtein is not None:
25
+ return float(_RapidLevenshtein.normalized_similarity(a, b))
26
+ prev = list(range(len(b) + 1))
27
+ for i, ca in enumerate(a):
28
+ curr = [i + 1]
29
+ for j, cb in enumerate(b):
30
+ curr.append(min(curr[j] + 1, prev[j + 1] + 1, prev[j] + (ca != cb)))
31
+ prev = curr
32
+ return 1.0 - prev[-1] / max(len(a), len(b))
33
+
34
+
35
+ def _light_normalize(text: str) -> str:
36
+ """Keeps diacritics (so diacritic changes lower the score) but drops punctuation and tatweel."""
37
+ text = unicodedata.normalize("NFC", text)
38
+ text = re.sub(r"[،؛؟!.,:;'\"()\[\]{}<>«»\-_/\\|@#$%^&*+=~`\u0640]", " ", text)
39
+ return re.sub(r"\s+", " ", text).strip()
40
+
41
+
42
+ def compute_signals(claim: str, candidate: str, content_type: str = "Ayah", local: bool = True) -> Dict[str, float]:
43
+ """Word-, character- and sequence-level similarity indicators between a quotation and a candidate source."""
44
+ is_quran = content_type == "Ayah"
45
+ full_candidate = candidate
46
+ if local: # score the snippet against the region it refers to, not against the whole verse / Hadith
47
+ candidate = best_region(claim, candidate)
48
+ normalize = normalize_strict if is_quran else normalize_lenient
49
+ norm_claim, norm_cand = normalize(claim), normalize(candidate)
50
+
51
+ claim_set, cand_set = set(tokenize(norm_claim)), set(tokenize(norm_cand))
52
+ if claim_set and cand_set:
53
+ shared = claim_set & cand_set
54
+ token_overlap = len(shared) / len(claim_set | cand_set)
55
+ coverage = len(shared) / len(claim_set)
56
+ else:
57
+ token_overlap = coverage = 0.0
58
+
59
+ claim_tokens, cand_tokens = tokenize(norm_claim), tokenize(norm_cand)
60
+ lcs_ratio = lcs_length(claim_tokens, cand_tokens) / len(claim_tokens) if claim_tokens else 0.0
61
+ edit_sim = _edit_similarity(norm_claim, norm_cand)
62
+
63
+ claim_chars, cand_chars = set(norm_claim.replace(" ", "")), set(norm_cand.replace(" ", ""))
64
+ char_overlap = len(claim_chars & cand_chars) / len(claim_chars | cand_chars) if (claim_chars or cand_chars) else 0.0
65
+
66
+ claim_flat, cand_flat = norm_claim.replace(" ", ""), norm_cand.replace(" ", "")
67
+ is_substring = int(bool(claim_flat) and bool(cand_flat) and (claim_flat in cand_flat or cand_flat in claim_flat))
68
+
69
+ diacritic_sim = (
70
+ _edit_similarity(_light_normalize(claim), _light_normalize(candidate), max_len=800) if is_quran else edit_sim
71
+ )
72
+ short = len(claim_tokens) < 4
73
+
74
+ if is_quran:
75
+ w = (
76
+ dict(coverage=0.35, diacritic_sim=0.30, lcs_ratio=0.15, token_overlap=0.10, char_overlap=0.05, edit_sim=0.05)
77
+ if short
78
+ else dict(coverage=0.25, diacritic_sim=0.30, lcs_ratio=0.20, token_overlap=0.10, char_overlap=0.05, edit_sim=0.10)
79
+ )
80
+ composite = (
81
+ w["coverage"] * coverage + w["diacritic_sim"] * diacritic_sim + w["lcs_ratio"] * lcs_ratio
82
+ + w["token_overlap"] * token_overlap + w["char_overlap"] * char_overlap + w["edit_sim"] * edit_sim
83
+ )
84
+ if is_substring and diacritic_sim >= 0.60:
85
+ composite = max(composite, 0.88)
86
+ elif is_substring:
87
+ composite = max(composite, 0.75)
88
+ else:
89
+ w = (
90
+ dict(coverage=0.40, lcs_ratio=0.20, token_overlap=0.20, char_overlap=0.10, edit_sim=0.10)
91
+ if short
92
+ else dict(coverage=0.30, lcs_ratio=0.28, token_overlap=0.18, char_overlap=0.12, edit_sim=0.12)
93
+ )
94
+ composite = (
95
+ w["coverage"] * coverage + w["lcs_ratio"] * lcs_ratio + w["token_overlap"] * token_overlap
96
+ + w["char_overlap"] * char_overlap + w["edit_sim"] * edit_sim
97
+ )
98
+ if is_substring:
99
+ composite = max(composite, 0.82)
100
+
101
+ return {
102
+ "token_overlap": round(token_overlap, 4),
103
+ "coverage": round(coverage, 4),
104
+ "lcs_ratio": round(lcs_ratio, 4),
105
+ "edit_sim": round(edit_sim, 4),
106
+ "diacritic_sim": round(diacritic_sim, 4),
107
+ "char_overlap": round(char_overlap, 4),
108
+ "is_substring": is_substring,
109
+ "source_fraction": round(min(1.0, len(claim_tokens) / max(len(tokenize(normalize(full_candidate))), 1)), 4),
110
+ "composite": round(composite, 4),
111
+ }
112
+
113
+
114
+ def best_match_score(claim: str, candidates: List[dict], content_type: str = "Ayah", local: bool = True):
115
+ """Return ``(score, candidate_with_signals)`` for the best-scoring candidate."""
116
+ best_score, best_candidate = 0.0, None
117
+ for candidate in candidates:
118
+ signals = compute_signals(claim, candidate.get("text", ""), content_type, local)
119
+ if signals["composite"] > best_score:
120
+ best_score, best_candidate = signals["composite"], {**candidate, "signals": signals}
121
+ return best_score, best_candidate
ui.py ADDED
@@ -0,0 +1,534 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Arabic-only presentation layer shared by the Gradio app and the static (in-browser) page.
2
+
3
+ Colour rules: green = verified, red = a real mismatch or a missing reference, amber = needs a human. Verified items
4
+ never use red. Every user-facing string is Arabic.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import html
9
+ from typing import List, Optional
10
+
11
+
12
+ _e = html.escape
13
+
14
+ APP_TITLE = "التحقق من هلوسة القرآن والحديث وتصحيحها"
15
+ APP_TAGLINE = "تحقّق من آيات القرآن والأحاديث النبوية بالدليل"
16
+ COPIED_MESSAGE = "تم نسخ النص بنجاح"
17
+
18
+ GROUP_LABEL = {"verified": "موثّق", "mismatch": "غير مطابق", "review": "تحتاج مراجعة بشرية"}
19
+ TYPE_LABEL = {"Ayah": "آية قرآنية", "Hadith": "حديث نبوي"}
20
+ INDICATORS = [
21
+ ("composite", "الدرجة المركبة"),
22
+ ("coverage", "تغطية الكلمات"),
23
+ ("lcs_ratio", "التسلسل النصي"),
24
+ ("token_overlap", "تداخل الكلمات"),
25
+ ("edit_sim", "تشابه الحروف"),
26
+ ("diacritic_sim", "التشكيل"),
27
+ ]
28
+ METHODS = {
29
+ "substring_match": "الاقتباس واردٌ كاملًا في المصدر",
30
+ "threshold_pass": "تجاوز عتبة التطابق",
31
+ "threshold_fail": "دون عتبة التطابق",
32
+ "borderline_multi_cov": "حالة حدّية: عدة مصادر متقاربة",
33
+ "borderline_default": "حالة حدّية",
34
+ "borderline_low_retrieval": "حالة حدّية: استرجاع ضعيف",
35
+ "no_candidates": "لا توجد مصادر مرشحة",
36
+ "empty_span": "نص فارغ",
37
+ "error": "تعذّرت المعالجة",
38
+ }
39
+
40
+
41
+ def _pct(value: float) -> str:
42
+ return f"{round(value * 100)}%"
43
+
44
+
45
+ def source_label(source: dict) -> str:
46
+ if source["type"] == "Quran":
47
+ start, end = source["ayah_start"], source["ayah_end"]
48
+ verses = f"{start}" if start == end else f"{start}–{end}"
49
+ return f"سورة {source['surah_name']} — الآية {verses}"
50
+ return f"حديث رقم {source['hadithID']} — {source['title']}"
51
+
52
+
53
+ def reason_text(reason: dict) -> str:
54
+ """Arabic explanation of a decision code produced by the pipeline."""
55
+ code = reason["code"]
56
+ src = reason.get("source", "")
57
+ if code == "exact_match":
58
+ return f"تطابق تام مع {src} بعد تجاهل اختلافات الرسم والتشكيل."
59
+ if code == "close_match":
60
+ return f"تطابق شبه كامل مع {src}."
61
+ if code == "altered_passage":
62
+ n = reason.get("n", 0)
63
+ how = "تغيّر ترتيب الكلمات" if reason.get("reordered") else f"{n} موضع مختلف"
64
+ return f"النص يخالف {src} ({how}). التصحيح المقترح هو نص المصدر حرفيًا ولم يُولَّد."
65
+ if code == "weak_match":
66
+ return "وُجد مصدر قريب لكن التطابق غير كافٍ للحكم الآلي."
67
+ if code == "too_short":
68
+ return "الاقتباس قصير جدًا (كلمتان فقط) فلا يمكن تحديد مصدره بيقين؛ يلزم الرجوع إلى مختص."
69
+ if code == "no_source":
70
+ return "لا يوجد في المراجع المضمّنة نصٌّ يشبه هذا الاقتباس، وقد يكون مختلَقًا."
71
+ if code == "insufficient_evidence":
72
+ return "الأدلة غير كافية: لا مصدر واضح، ودرجة اليقين منخفضة."
73
+ if code == "candidate_not_strong":
74
+ return f"وُجد مرشح ({src}، قوة المطابقة {round(reason.get('strength', 0) * 100)}%) لكن الدليل لا يكفي للتصحيح الآلي."
75
+ if code == "hadith_altered":
76
+ return (f"النص قريب جدًا من {src} لكن يختلف عنه في {reason.get('n', 1)} موضع (زيادة أو نقص أو استبدال كلمة)، وقد يغيّر ذلك المعنى؛ "
77
+ "لا يُوثَّق الحديث إلا إذا طابق النص حرفيًا، ولا يُصحَّح آليًا فتلزم مراجعة مختص.")
78
+ if code == "hadith_candidate":
79
+ return f"وُجدت رواية مشابهة ({src}) لكن لا يُصحَّح الحديث آليًا لاختلاف الروايات؛ يلزم الرجوع إلى مختص."
80
+ return "تعذّرت معالجة هذا الاقتباس آليًا."
81
+
82
+
83
+ def note_text(note: dict) -> str:
84
+ n = note.get("n", 0)
85
+ code = note["code"]
86
+ if code == "diacritic_conflict":
87
+ return f"تنبيه: تشكيل {n} كلمة يخالف المصحف (الكلمات صحيحة لكن الحركات مختلفة)."
88
+ if code == "orthographic_variant":
89
+ return f"ملاحظة: {n} اختلاف إملائي (رسم أو مسافات) لا يغيّر الكلمة."
90
+ if note.get("misattributed"):
91
+ if code == "is_ayah":
92
+ return f"تنبيه: النص قرآني ({note['source']}) لكن عبارة التقديم تنسبه إلى الحديث."
93
+ return f"تنبيه: النص حديث ({note['source']}) لكن عبارة التقديم تنسبه إلى القرآن."
94
+ if code == "is_hadith":
95
+ return f"تنبيه: هذا النص وارد في الحديث ({note['source']}) وليس في القرآن؛ ربما نُسب إلى الله تعالى خطأً."
96
+ if code == "is_ayah":
97
+ return f"تنبيه: هذا النص وارد في القرآن ({note['source']}) وليس في الحديث؛ ربما نُسب إلى النبي ﷺ خطأً."
98
+ if code == "hadith_minor_diffs":
99
+ return f"ملاحظة: {n} اختلاف طفيف عن أقرب رواية في المراجع."
100
+ return ""
101
+
102
+
103
+ def _diff_html(comparison: dict) -> str:
104
+ """Red = words only in the quotation; green = the source words that should be there."""
105
+ parts = []
106
+ for op in comparison["word_diff"]:
107
+ if op["op"] == "equal":
108
+ parts.append(f'<span class="w-eq">{_e(op["span"])}</span>')
109
+ continue
110
+ if op["span"]:
111
+ parts.append(f'<span class="w-extra">{_e(op["span"])}</span>')
112
+ if op["source"]:
113
+ parts.append(f'<span class="w-missing">{_e(op["source"])}</span>')
114
+ return " ".join(parts)
115
+
116
+
117
+ def _legend(group: str, comparison: Optional[dict]) -> str:
118
+ if group == "verified" or not comparison or all(op["op"] == "equal" for op in comparison["word_diff"]):
119
+ return ""
120
+ return '<p class="legend"><span class="w-extra">كلمات في الاقتباس تخالف المصدر</span> <span class="w-missing">الصواب من المصدر</span></p>'
121
+
122
+
123
+ def _indicator_table(signals: Optional[dict]) -> str:
124
+ if not signals:
125
+ return ""
126
+ rows = []
127
+ for key, label in INDICATORS:
128
+ value = float(signals.get(key, 0.0))
129
+ rows.append(
130
+ f'<div class="ind"><div class="ind-name">{label}</div>'
131
+ f'<div class="bar"><span style="width:{_pct(value)}"></span></div><div class="ind-val">{_pct(value)}</div></div>'
132
+ )
133
+ rows.append(f'<div class="ind"><div class="ind-name">وروده كاملًا في المصدر</div><div class="ind-val wide">{"نعم" if signals.get("is_substring") else "لا"}</div></div>')
134
+ return '<div class="indicators">' + "".join(rows) + "</div>"
135
+
136
+
137
+ def _match_percent(span: dict) -> int:
138
+ evidence = span["evidence"]
139
+ if evidence and evidence.get("comparison"):
140
+ return round(evidence["comparison"]["word_similarity"] * 100)
141
+ if evidence and evidence.get("signals"):
142
+ return round(evidence["signals"]["composite"] * 100)
143
+ return 0
144
+
145
+
146
+ def _evidence_panel(span: dict) -> str:
147
+ evidence, verification = span["evidence"], span["verification"]
148
+ rows = [
149
+ f'<div class="ev-row"><b>الاقتباس المكتشف</b><p class="quote">{_e(span["text"])}</p></div>',
150
+ f'<div class="ev-row"><b>النوع</b><p>{TYPE_LABEL[span["type"]]}</p></div>',
151
+ ]
152
+ if evidence:
153
+ comparison = evidence["comparison"]
154
+ rows += [
155
+ f'<div class="ev-row"><b>المصدر المرشح</b><p>{_e(source_label(evidence["source"]))}</p></div>',
156
+ f'<div class="ev-row"><b>نص المصدر الأصلي</b><p class="quote">{_e(comparison["source_excerpt"])}</p></div>',
157
+ f'<div class="ev-row"><b>المقارنة كلمةً بكلمة</b><p class="diff">{_diff_html(comparison)}</p>{_legend(span["group"], comparison)}</div>',
158
+ f'<div class="ev-row"><b>مؤشرات التحقق</b>{_indicator_table(evidence.get("signals"))}</div>',
159
+ ]
160
+ if comparison["diacritic_notes"]:
161
+ items = "، ".join(f"{_e(n['word'])} ← {_e(n['source_word'])}" for n in comparison["diacritic_notes"])
162
+ rows.append(f'<div class="ev-row"><b>اختلافات التشكيل</b><p>{items}</p></div>')
163
+ else:
164
+ rows.append('<div class="ev-row"><b>المصدر المرشح</b><p>لم يُسترجع أي مصدر.</p></div>')
165
+ rows += [
166
+ f'<div class="ev-row"><b>درجة اليقين في الحكم</b><p>{_pct(verification["confidence"])} · {METHODS.get(verification["method"], "")}</p></div>',
167
+ f'<div class="ev-row"><b>القرار</b><p>{_e(span["status_ar"])}</p></div>',
168
+ f'<div class="ev-row"><b>سبب القرار</b><p>{_e(reason_text(span["reason"]))}</p></div>',
169
+ ]
170
+ return "".join(rows)
171
+
172
+
173
+ def _card(span: dict) -> str:
174
+ group, evidence = span["group"], span["evidence"]
175
+ label = GROUP_LABEL[group]
176
+ chip = f'<span class="conf">نسبة المطابقة <b>{_match_percent(span)}%</b></span>'
177
+ scan_badge = ('<span class="badge scan" title="اكتُشف بمطابقة النص مع المراجع دون علامات تنصيص">اقتباس غير معلَن</span>'
178
+ if span["detection"]["backend"] == "scan" else "")
179
+
180
+ fields = [f'<div class="field"><label>الاقتباس المكتشف</label><p class="quote">{_e(span["text"])}</p></div>']
181
+ if evidence:
182
+ comparison = evidence["comparison"]
183
+ fields.append(f'<div class="field"><label>المصدر المرشح</label><p>{_e(source_label(evidence["source"]))}</p></div>')
184
+ if group != "verified": # a verified quotation equals its source: no need to print it twice
185
+ fields.append(
186
+ f'<div class="field"><label>المقارنة بالمصدر</label><p class="diff">{_diff_html(comparison)}</p>{_legend(group, comparison)}</div>'
187
+ )
188
+ else:
189
+ fields.append('<div class="field"><label>المصدر المرشح</label><p>لا يوجد</p></div>')
190
+ fields.append(f'<p class="reason">{_e(reason_text(span["reason"]))}</p>')
191
+
192
+ notes = "".join(f'<p class="note">{_e(note_text(n))}</p>' for n in span["notes"] if note_text(n))
193
+ action = ""
194
+ correction, suggestion = span["correction"], span["suggestion"]
195
+ if correction:
196
+ action = (
197
+ '<div class="action ok"><b>التصحيح المقترح من المصدر</b>'
198
+ f'<p class="quote">{_e(correction["display_text"])}</p>'
199
+ f'<p class="legend">{_e(source_label(correction["source"]))}</p></div>'
200
+ )
201
+ elif group == "review":
202
+ closest = ""
203
+ if suggestion:
204
+ closest = (f'<p class="legend">أقرب مصدر وُجد للمراجِع، وليس تصحيحًا آليًا: {_e(source_label(suggestion["source"]))}</p>'
205
+ f'<p class="quote small">{_e(suggestion["display_text"][:420])}</p>')
206
+ action = ('<div class="action warn"><b>الأدلة غير كافية للتصحيح الآلي.</b> يُوصى بالمراجعة البشرية.' + closest + "</div>")
207
+ elif span["status"] == "UNSUPPORTED":
208
+ action = '<div class="action bad"><b>لا يوجد مصدر مطابق في المراجع المتاحة.</b> لم يُقترح أي نص بديل.</div>'
209
+
210
+ return f"""
211
+ <div class="qcard {group}">
212
+ <div class="qhead">
213
+ <span class="idx">{span["id"]}</span>
214
+ <span class="badge type">{TYPE_LABEL[span["type"]]}</span>{scan_badge}
215
+ <span class="badge st {group}">{label}</span>
216
+ {chip}
217
+ </div>
218
+ {"".join(fields)}
219
+ {notes}
220
+ {action}
221
+ <details class="evidence"><summary>عرض الدليل</summary><div class="ev-body">{_evidence_panel(span)}</div></details>
222
+ </div>"""
223
+
224
+
225
+ def _summary(summary: dict) -> str:
226
+ tiles = [
227
+ (summary["n_spans"], "إجمالي الاقتباسات", ""),
228
+ (summary["n_ayah"], "آيات قرآنية", ""),
229
+ (summary["n_hadith"], "أحاديث نبوية", ""),
230
+ (summary["VERIFIED"], "موثّق", "verified"),
231
+ (summary["CORRECTED"] + summary["UNSUPPORTED"], "غير مطابق", "mismatch"),
232
+ (summary["HUMAN_REVIEW"], "تحتاج مراجعة", "review"),
233
+ ]
234
+ return '<div class="summary">' + "".join(
235
+ f'<div class="tile {cls}{" zero" if value == 0 and cls else ""}"><b>{value}</b><span>{label}</span></div>'
236
+ for value, label, cls in tiles
237
+ ) + "</div>"
238
+
239
+
240
+ def _highlighted_text(result: dict, title: str) -> str:
241
+ text, pieces, cursor = result["input_text"], [], 0
242
+ for span in result["spans"]:
243
+ pieces.append(_e(text[cursor:span["start"]]))
244
+ pieces.append(f'<mark class="{span["group"]}">{_e(text[span["start"]:span["end"]])}</mark>')
245
+ cursor = span["end"]
246
+ pieces.append(_e(text[cursor:]))
247
+ legend = ('<div class="hl-legend"><mark class="verified">موثّق</mark><mark class="mismatch">غير مطابق</mark>'
248
+ '<mark class="review">مراجعة بشرية</mark></div>')
249
+ return f'<div class="highlight"><label>{title}</label><p>' + "".join(pieces) + "</p>" + legend + "</div>"
250
+
251
+
252
+ def final_text(result: dict) -> str:
253
+ """Corrected text with an inline Arabic flag after every quotation that still needs attention."""
254
+ text, reports = result["input_text"], result["spans"]
255
+ for report in sorted(reports, key=lambda r: r["start"], reverse=True):
256
+ if report["status"] == "CORRECTED" and report["correction"]:
257
+ text = text[: report["start"]] + report["correction"]["display_text"] + text[report["end"]:]
258
+ elif report["status"] == "UNSUPPORTED":
259
+ text = text[: report["end"]] + " [⚠ لا يوجد مصدر مطابق]" + text[report["end"]:]
260
+ elif report["status"] == "HUMAN_REVIEW":
261
+ text = text[: report["end"]] + " [⚠ يحتاج مراجعة بشرية]" + text[report["end"]:]
262
+ return text
263
+
264
+
265
+ def _final_block(result: dict) -> str:
266
+ summary = result["summary"]
267
+ fixed = summary["CORRECTED"]
268
+ open_items = summary["UNSUPPORTED"] + summary["HUMAN_REVIEW"]
269
+ if fixed == 0 and open_items == 0:
270
+ headline = "كل الاقتباسات المكتشفة مطابقة للمصادر."
271
+ else:
272
+ headline = f"صُحِّح {fixed} اقتباس من نص المصدر، وبقي {open_items} اقتباس يحتاج مراجعة بشرية ومعلَّم بعلامة تنبيه."
273
+ text = final_text(result)
274
+ return (
275
+ '<div class="final"><div class="final-head"><b>النسخة المصحّحة</b>'
276
+ f'<button class="copy" type="button" data-text="{_e(text, quote=True)}" '
277
+ 'onclick="window.icvCopy&&window.icvCopy(this)">نسخ النص</button></div>'
278
+ f'<p class="legend">{headline}</p><p class="final-text">{_e(text)}</p></div>'
279
+ )
280
+
281
+
282
+ def render_results(result: dict, generated_answer: Optional[str] = None) -> str:
283
+ """HTML report for a pipeline result. Pass the model's answer (mode B) to show it first and label the text as generated."""
284
+ header = render_generated_header(generated_answer) if generated_answer is not None else ""
285
+ if not result["spans"]:
286
+ body = (
287
+ '<div class="notice">لم يُعثر على اقتباسات قرآنية أو حديثية في هذا النص. يتعرّف النظام على الاقتباسات بمطابقتها مع المراجع، '
288
+ 'سواء وُضعت بين علامات تنصيص أو أقواس أو وردت داخل الكلام دون أي عبارة تمهيدية.</div>'
289
+ )
290
+ return f'<div class="icv">{header}{body}</div>'
291
+ generated = generated_answer is not None
292
+ title = "إجابة النموذج مع الاقتباسات المكتشفة" if generated else "النص مع الاقتباسات المكتشفة"
293
+ parts = [_summary(result["summary"]), _highlighted_text(result, title), "".join(_card(s) for s in result["spans"])]
294
+ summary = result["summary"]
295
+ if generated or summary["CORRECTED"] or summary["UNSUPPORTED"] or summary["HUMAN_REVIEW"]:
296
+ parts.append(_final_block(result))
297
+ return f'<div class="icv">{header}{"".join(parts)}</div>'
298
+
299
+
300
+ def render_generated_header(answer: str) -> str:
301
+ return f'<div class="generated"><label>إجابة النموذج اللغوي كما وصلت</label><p>{_e(answer)}</p></div>'
302
+
303
+
304
+ def render_message(message: str, kind: str = "info") -> str:
305
+ return f'<div class="icv"><div class="notice {kind}">{_e(message)}</div></div>'
306
+
307
+
308
+ # --------------------------------------------------------------------------------------------------------------
309
+ # Static markup and CSS
310
+ # --------------------------------------------------------------------------------------------------------------
311
+ STAR = (
312
+ '<svg class="mark" viewBox="0 0 64 64" fill="none" stroke="#E2C06E" stroke-width="1.8" aria-hidden="true">'
313
+ '<rect x="14" y="14" width="36" height="36"/><rect x="14" y="14" width="36" height="36" transform="rotate(45 32 32)"/>'
314
+ '<circle cx="32" cy="32" r="7" fill="#E2C06E" stroke="none"/></svg>'
315
+ )
316
+
317
+ HERO = f"""
318
+ <div class="icv"><div class="hero">
319
+ {STAR}
320
+ <h1>{APP_TITLE}</h1>
321
+ <p class="tagline">{APP_TAGLINE}</p>
322
+ <p class="sub">يكتشف النظام الاقتباسات داخل أي نص، ويسترجع نصوصها من المصادر، ويقارنها كلمةً بكلمة، ثم يعرض الدليل.
323
+ وإذا لم تكفِ الأدلة فلن يختلق تصحيحًا، بل يحيل الحالة إلى المراجعة البشرية.</p>
324
+ <div class="flow"><span>كشف</span><i>‹</i><span>استرجاع</span><i>‹</i><span>محاذاة</span><i>‹</i><span>دليل</span><i>‹</i><span>قرار</span></div>
325
+ </div></div>
326
+ """
327
+
328
+ DISCLAIMER = """<div class="icv"><div class="disclaimer">أداة مساعدة للتدقيق النصي وليست فتوى ولا بديلًا عن المراجعة المتخصصة.
329
+ النتائج مبنية على مراجع القرآن الكريم والكتب الستة المضمّنة فقط.</div></div>"""
330
+
331
+ PLACEHOLDER = "الصق هنا النص الذي ولّده نموذج لغوي. يكتشف النظام الآيات والأحاديث الواردة فيه، بعلامات تنصيص أو بدونها…"
332
+ PROMPT_PLACEHOLDER = "اكتب سؤالك، مثل: اشرح لي فضل الصبر في القرآن والسنة مع ذكر الأدلة."
333
+
334
+ _PATTERN = (
335
+ "url(\"data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' width='88' height='88' viewBox='0 0 88 88'%3E"
336
+ "%3Cg fill='none' stroke='%230F4C3A' stroke-opacity='0.07' stroke-width='1'%3E"
337
+ "%3Crect x='22' y='22' width='44' height='44'/%3E%3Crect x='22' y='22' width='44' height='44' transform='rotate(45 44 44)'/%3E"
338
+ "%3C/g%3E%3C/svg%3E\")"
339
+ )
340
+
341
+ CSS = """
342
+ @import url('https://fonts.googleapis.com/css2?family=Cairo:wght@400;600;700&family=Amiri:wght@400;700&display=swap');
343
+ :root { --green:#0F4C3A; --green-2:#17694F; --gold:#B8912F; --gold-2:#E2C06E; --cream:#FBF6EA; --paper:#FFFFFF;
344
+ --ink:#1F2933; --muted:#55626D; --line:#E6DCC3; --ok:#1B7A4B; --ok-bg:#EAF6EF; --ok-line:#BFE3CD;
345
+ --bad:#B3261E; --bad-bg:#FDECEA; --bad-line:#F4B8B3; --warn:#7A5B0C; --warn-bg:#FFF6DA; --warn-line:#EBD28A; }
346
+ .gradio-container, body.icv-page { background:var(--cream) PATTERN !important; font-family:'Cairo','Segoe UI',Tahoma,sans-serif; color:var(--ink); }
347
+ .gradio-container { max-width:1060px !important; --body-text-color:#1F2933; --block-background-fill:#FFFFFF; --block-border-color:#E6DCC3;
348
+ --input-background-fill:#FFFFFF; --block-label-text-color:#55626D; --button-primary-background-fill:#0F4C3A;
349
+ --button-primary-background-fill-hover:#17694F; --button-primary-text-color:#FFFFFF; --button-secondary-background-fill:#FFFFFF;
350
+ --button-secondary-text-color:#0F4C3A; --button-secondary-border-color:#B8912F; --color-accent:#B8912F; }
351
+ .icv { direction:rtl; text-align:right; color:var(--ink); line-height:1.9; }
352
+ .icv .hero { background:linear-gradient(135deg,#0F4C3A,#17694F); border-radius:20px; padding:30px 22px 26px; margin:10px 0 18px; text-align:center;
353
+ box-shadow:0 6px 22px rgba(15,76,58,.18); border-bottom:4px solid var(--gold); }
354
+ .icv .hero .mark { width:54px; height:54px; display:block; margin:0 auto 4px; }
355
+ .icv .hero h1 { font-size:2.3rem; margin:.1rem 0; color:#FFFFFF; font-weight:700; }
356
+ .icv .hero .tagline { color:var(--gold-2); font-size:1.2rem; font-weight:600; margin:.1rem 0 .6rem; }
357
+ .icv .hero .sub { max-width:720px; margin:0 auto; color:#E9F2EE; font-size:1rem; }
358
+ .icv .flow { display:flex; flex-wrap:wrap; justify-content:center; align-items:center; gap:8px; margin-top:16px; }
359
+ .icv .flow span { background:rgba(255,255,255,.12); border:1px solid rgba(226,192,110,.55); color:#FFF; border-radius:999px; padding:3px 16px; font-size:.9rem; }
360
+ .icv .flow i { color:var(--gold-2); font-style:normal; font-size:1.2rem; }
361
+ .icv .disclaimer { font-size:.85rem; color:var(--muted); text-align:center; padding:14px 8px; }
362
+ .icv .en-free { direction:rtl; }
363
+ .input-area textarea { direction:rtl; text-align:right; font-family:'Amiri','Cairo',serif !important; font-size:1.25rem !important; line-height:2.1 !important; background:#FFFFFF !important; color:#1F2933 !important; }
364
+ .llm-row label, .gradio-container label span { font-family:'Cairo',sans-serif; }
365
+ .results { direction:rtl; }
366
+ .icv .summary { display:grid; grid-template-columns:repeat(auto-fit,minmax(130px,1fr)); gap:10px; margin:8px 0 14px; }
367
+ .icv .tile { background:var(--paper); border:1px solid var(--line); border-radius:14px; padding:12px 8px; text-align:center; }
368
+ .icv .tile b { display:block; font-size:1.8rem; color:var(--green); line-height:1.3; } .icv .tile span { font-size:.92rem; color:var(--muted); }
369
+ .icv .tile.verified { background:var(--ok-bg); border-color:var(--ok-line); } .icv .tile.verified b { color:var(--ok); }
370
+ .icv .tile.mismatch { background:var(--bad-bg); border-color:var(--bad-line); } .icv .tile.mismatch b { color:var(--bad); }
371
+ .icv .tile.review { background:var(--warn-bg); border-color:var(--warn-line); } .icv .tile.review b { color:var(--warn); }
372
+ .icv .tile.zero { background:var(--paper); border-color:var(--line); opacity:.6; } .icv .tile.zero b { color:var(--muted); }
373
+ .icv .highlight, .icv .generated, .icv .final { background:var(--paper); border:1px solid var(--line); border-radius:14px; padding:14px 18px; margin-bottom:14px; }
374
+ .icv label { display:block; color:var(--muted); font-size:.82rem; font-weight:600; margin-bottom:4px; }
375
+ .icv .highlight p, .icv .generated p, .icv .final-text { font-family:'Amiri','Cairo',serif; font-size:1.2rem; line-height:2.2; margin:0; white-space:pre-wrap; }
376
+ .icv mark { color:var(--ink); border-radius:6px; padding:1px 5px; }
377
+ .icv mark.verified { background:#D5EEDF; } .icv mark.mismatch { background:#F8CFCB; } .icv mark.review { background:#F7E3A6; }
378
+ .icv .hl-legend { display:flex; flex-wrap:wrap; gap:8px; margin-top:10px; font-size:.78rem; }
379
+ .icv .qcard { background:var(--paper); border:1px solid var(--line); border-inline-start:6px solid var(--line); border-radius:16px; padding:16px 18px; margin:12px 0; }
380
+ .icv .qcard.verified { border-inline-start-color:var(--ok); } .icv .qcard.mismatch { border-inline-start-color:var(--bad); } .icv .qcard.review { border-inline-start-color:var(--gold); }
381
+ .icv .qhead { display:flex; flex-wrap:wrap; gap:8px; align-items:center; margin-bottom:8px; }
382
+ .icv .idx { background:var(--green); color:#FFF; border-radius:50%; width:28px; height:28px; display:inline-flex; align-items:center; justify-content:center; font-weight:700; font-size:.9rem; }
383
+ .icv .badge { padding:2px 14px; border-radius:999px; font-size:.84rem; background:#F3EEDD; border:1px solid var(--line); color:var(--ink); }
384
+ .icv .badge.st.verified { background:var(--ok-bg); border-color:var(--ok-line); color:var(--ok); font-weight:600; }
385
+ .icv .badge.st.mismatch { background:var(--bad-bg); border-color:var(--bad-line); color:var(--bad); font-weight:600; }
386
+ .icv .badge.st.review { background:var(--warn-bg); border-color:var(--warn-line); color:var(--warn); font-weight:600; }
387
+ .icv .badge.scan { background:#EEF3FB; border-color:#C9D8EE; color:#2F4F7F; }
388
+ .icv .conf { margin-inline-start:auto; color:var(--muted); font-size:.88rem; } .icv .conf b { color:var(--green); }
389
+ .icv .field { margin:8px 0; } .icv .field p { margin:0; }
390
+ .icv .quote { font-family:'Amiri','Cairo',serif; font-size:1.25rem; line-height:2.2; color:var(--ink); } .icv .quote.small { font-size:1.05rem; color:#3a4651; }
391
+ .icv .diff { font-family:'Amiri','Cairo',serif; font-size:1.2rem; line-height:2.2; }
392
+ .icv .w-extra { background:var(--bad-bg); color:var(--bad); border-radius:4px; padding:0 4px; text-decoration:line-through; }
393
+ .icv .w-missing { background:var(--ok-bg); color:var(--ok); border-radius:4px; padding:0 4px; font-weight:700; }
394
+ .icv .legend { color:var(--muted); font-size:.82rem; margin:4px 0 0; } .icv .legend span { text-decoration:none; font-size:.8rem; }
395
+ .icv .reason { color:var(--muted); font-size:.92rem; margin:6px 0 0; }
396
+ .icv .note { color:var(--warn); background:var(--warn-bg); border:1px solid var(--warn-line); border-radius:10px; padding:5px 12px; font-size:.88rem; margin:8px 0 0; }
397
+ .icv .action { border-radius:12px; padding:10px 14px; margin-top:10px; }
398
+ .icv .action.ok { background:var(--ok-bg); border:1px solid var(--ok-line); } .icv .action.ok b { color:var(--ok); }
399
+ .icv .action.warn { background:var(--warn-bg); border:1px solid var(--warn-line); } .icv .action.warn b { color:var(--warn); }
400
+ .icv .action.bad { background:var(--bad-bg); border:1px solid var(--bad-line); } .icv .action.bad b { color:var(--bad); }
401
+ .icv .evidence { margin-top:12px; border-top:1px dashed var(--line); padding-top:8px; }
402
+ .icv .evidence summary { cursor:pointer; color:var(--green); font-weight:700; }
403
+ .icv .ev-row { margin:10px 0; } .icv .ev-row b { color:var(--gold); font-size:.88rem; } .icv .ev-row p { margin:2px 0; }
404
+ .icv .indicators { display:grid; gap:6px; margin-top:6px; }
405
+ .icv .ind { display:grid; grid-template-columns:150px 1fr 48px; gap:10px; align-items:center; font-size:.88rem; }
406
+ .icv .bar { background:#EFE8D3; border-radius:999px; height:9px; overflow:hidden; direction:rtl; } .icv .bar span { display:block; height:100%; background:linear-gradient(270deg,var(--green),var(--gold)); }
407
+ .icv .ind-val { text-align:left; direction:ltr; color:var(--ink); } .icv .ind-val.wide { grid-column:2 / span 2; text-align:right; direction:rtl; }
408
+ .icv .final { border-color:var(--gold); background:#FFFDF6; }
409
+ .icv .final-head { display:flex; justify-content:space-between; align-items:center; } .icv .final-head b { color:var(--green); font-size:1.05rem; }
410
+ .icv .copy { background:var(--green); color:#FFF; border:0; border-radius:10px; padding:5px 16px; font-family:inherit; cursor:pointer; }
411
+ .icv .notice { background:var(--paper); border:1px solid var(--line); border-radius:14px; padding:16px; }
412
+ .icv .notice.warn { background:var(--warn-bg); border-color:var(--warn-line); color:var(--warn); } .icv .notice.bad { background:var(--bad-bg); border-color:var(--bad-line); color:var(--bad); }
413
+ @media (max-width:640px){ .icv .hero h1{font-size:1.7rem;} .icv .ind{grid-template-columns:104px 1fr 40px;} }
414
+ """.replace("PATTERN", _PATTERN)
415
+
416
+
417
+ # --------------------------------------------------------------------------------------------------------------
418
+ # Visual layer: soft glassmorphism, dynamic gradients, RTL-safe text, toast and skeleton loaders
419
+ # --------------------------------------------------------------------------------------------------------------
420
+ GLASS_CSS = """
421
+ :root { --glass:rgba(255,255,255,.62); --glass-strong:rgba(255,255,255,.82); --glass-line:rgba(255,255,255,.75);
422
+ --shadow-1:0 1px 2px rgba(15,76,58,.06), 0 8px 24px rgba(15,76,58,.08); --shadow-2:0 2px 4px rgba(15,76,58,.08), 0 18px 44px rgba(15,76,58,.16); }
423
+ .gradio-container, body.icv-page { background:
424
+ radial-gradient(900px 520px at 88% -8%, rgba(226,192,110,.34), transparent 60%),
425
+ radial-gradient(760px 520px at 6% 4%, rgba(23,105,79,.20), transparent 62%),
426
+ linear-gradient(180deg,#FBF6EA 0%,#F3EBD3 100%) !important; background-attachment:fixed !important; }
427
+ /* RTL safety: long words, URLs and mixed Latin/digit runs wrap inside their box instead of spilling out */
428
+ .icv, .icv * { box-sizing:border-box; min-width:0; }
429
+ .icv p, .icv span, .icv b, .icv label, .icv summary, .icv button, .icv .badge, .icv .tile { overflow-wrap:anywhere; }
430
+ .icv .quote, .icv .diff, .icv .final-text, .icv .highlight p, .icv .generated p { unicode-bidi:plaintext; text-align:start; }
431
+ .icv .hero { position:relative; overflow:hidden; border-bottom:0; border:1px solid rgba(226,192,110,.45);
432
+ background:linear-gradient(120deg,#0B3B2D,#17694F 45%,#0F4C3A 70%,#1d7a5c); background-size:240% 240%; animation:icv-flow 16s ease-in-out infinite;
433
+ box-shadow:var(--shadow-2); }
434
+ .icv .hero::before { content:""; position:absolute; inset:-40% -10% auto auto; width:60%; aspect-ratio:1; border-radius:50%;
435
+ background:radial-gradient(circle, rgba(226,192,110,.38), transparent 65%); pointer-events:none; }
436
+ .icv .hero::after { content:""; position:absolute; inset:auto auto 0 0; width:100%; height:4px; background:linear-gradient(90deg,transparent,var(--gold-2),transparent); }
437
+ .icv .hero h1 { font-size:clamp(1.45rem,4.2vw,2.3rem); line-height:1.5; text-wrap:balance; position:relative; }
438
+ .icv .hero .tagline, .icv .hero .sub, .icv .flow { position:relative; }
439
+ .icv .flow span { backdrop-filter:blur(6px); -webkit-backdrop-filter:blur(6px); }
440
+ @keyframes icv-flow { 0%,100%{background-position:0% 50%} 50%{background-position:100% 50%} }
441
+ .icv .tile, .icv .highlight, .icv .generated, .icv .final, .icv .qcard, .icv .notice, .icv .export {
442
+ background:var(--glass); border:1px solid var(--glass-line); box-shadow:var(--shadow-1);
443
+ backdrop-filter:blur(14px) saturate(150%); -webkit-backdrop-filter:blur(14px) saturate(150%); transition:transform .25s ease, box-shadow .25s ease; }
444
+ .icv .qcard { border-inline-start:6px solid var(--line); }
445
+ .icv .qcard:hover, .icv .tile:hover { transform:translateY(-2px); box-shadow:var(--shadow-2); }
446
+ .icv .tile.verified { background:linear-gradient(160deg,rgba(234,246,239,.92),rgba(255,255,255,.6)); }
447
+ .icv .tile.mismatch { background:linear-gradient(160deg,rgba(253,236,234,.92),rgba(255,255,255,.6)); }
448
+ .icv .tile.review { background:linear-gradient(160deg,rgba(255,246,218,.95),rgba(255,255,255,.6)); }
449
+ .icv .final { background:linear-gradient(160deg,rgba(255,253,246,.92),rgba(250,240,208,.55)); border-color:rgba(184,145,47,.55); }
450
+ .icv .final-head { flex-wrap:wrap; gap:10px; }
451
+ .icv .qhead .badge { max-width:100%; white-space:normal; }
452
+ .icv .ind { grid-template-columns:minmax(96px,150px) minmax(0,1fr) 48px; }
453
+ .icv .bar span { background:linear-gradient(270deg,var(--green),var(--gold-2)); transition:width .6s ease; }
454
+ .icv .copy, .icv .dl { background:linear-gradient(135deg,var(--green),var(--green-2)); color:#FFF; border:0; border-radius:12px; padding:7px 18px;
455
+ font-family:inherit; font-weight:600; cursor:pointer; box-shadow:0 4px 12px rgba(15,76,58,.25); transition:transform .15s ease, box-shadow .15s ease, filter .15s ease; }
456
+ .icv .copy:hover, .icv .dl:hover { transform:translateY(-1px); filter:brightness(1.08); box-shadow:0 8px 18px rgba(15,76,58,.3); }
457
+ .icv .copy:active, .icv .dl:active { transform:translateY(0); }
458
+ .icv .copy:focus-visible, .icv .dl:focus-visible { outline:3px solid rgba(184,145,47,.55); outline-offset:2px; }
459
+ .icv .copy.done { background:linear-gradient(135deg,#1B7A4B,#2a9d66); }
460
+ .icv .export { border-radius:16px; padding:12px 18px; margin:12px 0; }
461
+ .icv .export summary { cursor:pointer; color:var(--green); font-weight:700; }
462
+ .icv .dls { display:flex; flex-wrap:wrap; gap:10px; margin-top:10px; }
463
+ /* toast */
464
+ #icv-toast { position:fixed; inset-inline:0; bottom:28px; margin-inline:auto; width:max-content; max-width:calc(100vw - 32px); z-index:99999;
465
+ direction:rtl; text-align:center; font:600 1rem 'Cairo','Segoe UI',Tahoma,sans-serif; color:#FFF; padding:12px 24px; border-radius:999px;
466
+ background:linear-gradient(135deg,rgba(15,76,58,.94),rgba(23,105,79,.94)); border:1px solid rgba(226,192,110,.6);
467
+ box-shadow:0 14px 40px rgba(15,76,58,.35); backdrop-filter:blur(10px); -webkit-backdrop-filter:blur(10px);
468
+ opacity:0; transform:translateY(18px) scale(.97); pointer-events:none; transition:opacity .28s ease, transform .28s cubic-bezier(.2,.9,.3,1.2); }
469
+ #icv-toast.show { opacity:1; transform:none; } #icv-toast.bad { background:linear-gradient(135deg,rgba(179,38,30,.95),rgba(214,69,58,.95)); }
470
+ /* skeleton loaders */
471
+ .skel { position:relative; overflow:hidden; background:rgba(15,76,58,.08); border-radius:10px; }
472
+ .skel::after { content:""; position:absolute; inset:0; transform:translateX(100%); animation:icv-shimmer 1.4s infinite;
473
+ background:linear-gradient(90deg,transparent,rgba(255,255,255,.75),transparent); }
474
+ @keyframes icv-shimmer { 100% { transform:translateX(-100%); } }
475
+ .skel-card { background:var(--glass); border:1px solid var(--glass-line); border-radius:16px; padding:16px 18px; margin:12px 0; box-shadow:var(--shadow-1); }
476
+ .skel-line { height:14px; margin:10px 0; } .skel-line.w60 { width:60%; } .skel-line.w85 { width:85%; } .skel-line.w40 { width:40%; }
477
+ .skel-tiles { display:grid; grid-template-columns:repeat(auto-fit,minmax(120px,1fr)); gap:10px; margin:8px 0 14px; } .skel-tile { height:78px; border-radius:14px; }
478
+ .icv-spinner { width:18px; height:18px; border-radius:50%; border:3px solid rgba(15,76,58,.18); border-top-color:var(--gold); display:inline-block;
479
+ vertical-align:middle; margin-inline-end:10px; animation:icv-spin .8s linear infinite; }
480
+ @keyframes icv-spin { to { transform:rotate(360deg); } }
481
+ @media (prefers-reduced-motion: reduce) { .icv .hero, .skel::after, .icv-spinner { animation:none; } .icv .qcard, .icv .tile, #icv-toast { transition:none; } }
482
+ @media (max-width:640px){ .icv .ind{grid-template-columns:minmax(84px,104px) minmax(0,1fr) 40px;} .icv .final-head{flex-direction:column; align-items:stretch;} .icv .copy{width:100%;} }
483
+ /* layout polish: centred, compact, professional */
484
+ .icv .notice, .icv .disclaimer, .icv .export { text-align:center; }
485
+ .icv .export .legend { max-width:640px; margin:8px auto 0; font-size:.9rem; line-height:1.9; color:var(--muted); }
486
+ .icv .dls { justify-content:center; }
487
+ .icv .dl { display:inline-flex; flex-direction:column; align-items:center; gap:2px; min-width:170px; }
488
+ .icv .dl .t { font-weight:700; } .icv .dl .s { font-size:.78rem; font-weight:500; opacity:.85; }
489
+ .icv .hero .flow { display:flex; justify-content:center; flex-wrap:wrap; gap:8px; }
490
+ """
491
+ CSS += GLASS_CSS
492
+
493
+ SKELETON = (
494
+ '<div class="icv" aria-busy="true" aria-label="جارٍ التحقق">'
495
+ '<div class="skel-tiles">' + '<div class="skel skel-tile"></div>' * 4 + '</div>'
496
+ + ('<div class="skel-card"><div class="skel skel-line w40"></div><div class="skel skel-line w85"></div>'
497
+ '<div class="skel skel-line"></div><div class="skel skel-line w60"></div></div>') * 2
498
+ + '</div>'
499
+ )
500
+
501
+ COPY_JS = r"""
502
+ (function () {
503
+ if (window.icvCopy) return;
504
+ function toast(message, bad) {
505
+ var t = document.getElementById('icv-toast');
506
+ if (!t) { t = document.createElement('div'); t.id = 'icv-toast'; t.setAttribute('role', 'status'); t.setAttribute('aria-live', 'polite'); document.body.appendChild(t); }
507
+ t.textContent = message; t.className = 'show' + (bad ? ' bad' : '');
508
+ clearTimeout(window.__icvToast); window.__icvToast = setTimeout(function () { t.className = bad ? 'bad' : ''; }, 2400);
509
+ }
510
+ function legacyCopy(text) {
511
+ var area = document.createElement('textarea'); area.value = text; area.setAttribute('readonly', '');
512
+ area.style.cssText = 'position:fixed;top:0;left:0;opacity:0;pointer-events:none'; document.body.appendChild(area);
513
+ area.select(); area.setSelectionRange(0, text.length); var ok = false;
514
+ try { ok = document.execCommand('copy'); } catch (e) { ok = false; }
515
+ document.body.removeChild(area); return ok;
516
+ }
517
+ window.icvToast = toast;
518
+ window.icvCopy = function (button) {
519
+ var text = button.getAttribute('data-text') || '';
520
+ function finish(ok) {
521
+ toast(ok ? '__COPIED__' : 'تعذّر النسخ، حدّد النص وانسخه يدويًا', !ok);
522
+ if (ok) { button.classList.add('done'); setTimeout(function () { button.classList.remove('done'); }, 1600); }
523
+ }
524
+ if (navigator.clipboard && window.isSecureContext) {
525
+ navigator.clipboard.writeText(text).then(function () { finish(true); }, function () { finish(legacyCopy(text)); });
526
+ } else { finish(legacyCopy(text)); }
527
+ };
528
+ window.icvDownload = function (button) {
529
+ var blob = new Blob([button.getAttribute('data-text') || ''], { type: 'text/tab-separated-values;charset=utf-8' });
530
+ var link = document.createElement('a'); link.href = URL.createObjectURL(blob); link.download = button.getAttribute('data-name') || 'results.tsv';
531
+ document.body.appendChild(link); link.click(); document.body.removeChild(link); setTimeout(function () { URL.revokeObjectURL(link.href); }, 1000);
532
+ };
533
+ })();
534
+ """.replace("__COPIED__", COPIED_MESSAGE)
verifier.py ADDED
@@ -0,0 +1,571 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Evidence-based verification and source-backed correction of Quran and Hadith quotations.
2
+
3
+ Pipeline (see ``IslamicContentVerifier``):
4
+
5
+ input text -> detection -> BM25 retrieval -> alignment (phonetic skeleton, sliding window, LCS, dynamic gap)
6
+ -> exact match : verified
7
+ -> altered Quran passage : mismatch with the exact source text as the correction
8
+ -> anything uncertain : human review (no correction is produced)
9
+ -> nothing similar found : no matching source
10
+
11
+ Safety rule: a correction is only ever the exact text of a retrieved source. Nothing is generated.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import difflib
16
+ import logging
17
+ import time
18
+ from dataclasses import dataclass, field
19
+ from typing import Dict, List, Optional
20
+
21
+ from alignment import align
22
+ from detector import DetectedSpan
23
+ from scanner import HybridDetector
24
+ from idgham import apply_idgham
25
+ from normalization import content_words, normalize_for_matching
26
+ from retrieval import SourceRetriever
27
+ from similarity import best_match_score, compute_signals
28
+
29
+ logger = logging.getLogger(__name__)
30
+
31
+ MAX_INPUT_CHARS = 20_000
32
+ SHORT_QUOTE_WORDS = 3 # quotations shorter than this (found by the rules) always go to human review
33
+ MIN_EXACT_TOKENS = 3 # shorter quotations are too generic to be verified by exact containment alone
34
+
35
+
36
+ # --------------------------------------------------------------------------------------------------------------
37
+ # Configuration (all tunable numbers live here)
38
+ # --------------------------------------------------------------------------------------------------------------
39
+ @dataclass
40
+ class VerifierConfig:
41
+ """Thresholds calibrated in Subtask 1B of the research notebook."""
42
+
43
+ quran_correct_threshold: float = 0.94
44
+ quran_uncertain_low: float = 0.45
45
+ quran_min_coverage: float = 0.40
46
+ hadith_correct_threshold: float = 0.88
47
+ hadith_uncertain_low: float = 0.30
48
+ hadith_min_coverage: float = 0.70
49
+ quran_top_k: int = 25
50
+ hadith_top_k: int = 15
51
+ hadith_retrieval_guard: float = 0.20
52
+
53
+
54
+ @dataclass
55
+ class CorrectorConfig:
56
+ """Correction is proposed only when match strength >= ``*_strong``; between ``*_low`` and ``*_strong`` the
57
+ case goes to human review. Hadith is never auto-corrected (``hadith_strong`` > 1), by design: the exact
58
+ fragment boundaries of a Hadith quotation cannot be reproduced reliably, and narrations differ legitimately."""
59
+
60
+ max_window: int = 8
61
+ hadith_top_k: int = 40
62
+ quran_strong: float = 0.65
63
+ min_full_ratio: float = 0.40
64
+ quran_low: float = 0.55
65
+ hadith_strong: float = 1.01
66
+ hadith_low: float = 0.45
67
+
68
+
69
+ @dataclass
70
+ class PipelineConfig:
71
+ verifier: VerifierConfig = field(default_factory=VerifierConfig)
72
+ corrector: CorrectorConfig = field(default_factory=CorrectorConfig)
73
+ verified_min_conf: float = 0.75 # "Correct" verdicts weaker than this -> human review
74
+ unsupported_min_conf: float = 0.70 # "Incorrect + no source" needs this confidence to abstain confidently
75
+ unsupported_strength: float = 0.35 # ...or the best source match is this weak (nothing similar exists)
76
+
77
+
78
+ # --------------------------------------------------------------------------------------------------------------
79
+ # Verification (Subtask 1B)
80
+ # --------------------------------------------------------------------------------------------------------------
81
+ @dataclass
82
+ class Verification:
83
+ verdict: str # 'Correct' | 'Incorrect'
84
+ confidence: float
85
+ best_score: float
86
+ method: str
87
+ source: Optional[dict] = None # best matching source record (with 'signals')
88
+ n_candidates: int = 0
89
+ retrieval_top: float = 0.0
90
+
91
+
92
+ class Verifier:
93
+ """Scores a quotation against retrieved candidates and returns a verdict with its supporting source."""
94
+
95
+ def __init__(self, retriever: SourceRetriever, config: Optional[VerifierConfig] = None) -> None:
96
+ self.kb = retriever
97
+ self.cfg = config or VerifierConfig()
98
+
99
+ def verify(self, span_text: str, content_type: str) -> Verification:
100
+ if not span_text or not span_text.strip():
101
+ return self._result("Incorrect", 0.95, 0.0, None, 0, "empty_span")
102
+ if content_type == "Ayah":
103
+ return self._verify_quran(span_text)
104
+ if content_type == "Hadith":
105
+ return self._verify_hadith(span_text)
106
+ return self._result("Incorrect", 0.5, 0.0, None, 0, "unknown_type")
107
+
108
+ def _verify_quran(self, span: str) -> Verification:
109
+ cfg = self.cfg
110
+ candidates = self.kb.search_quran_ayahs(span, top_k=cfg.quran_top_k)
111
+ if not candidates:
112
+ return self._result("Incorrect", 0.8, 0.0, None, 0, "no_candidates")
113
+ score, best = best_match_score(span, candidates, "Ayah")
114
+ signals = best.get("signals", {}) if best else {}
115
+ coverage, is_substring = signals.get("coverage", 0.0), signals.get("is_substring", 0)
116
+ n = len(candidates)
117
+
118
+ if is_substring and coverage >= cfg.quran_min_coverage:
119
+ return self._result("Correct", min(0.98, 0.85 + score * 0.15), score, best, n, "substring_match")
120
+ if score >= cfg.quran_correct_threshold and coverage >= cfg.quran_min_coverage:
121
+ return self._result("Correct", min(0.95, 0.70 + score * 0.25), score, best, n, "threshold_pass")
122
+ if score <= cfg.quran_uncertain_low:
123
+ return self._result("Incorrect", min(0.95, 0.70 + (1 - score) * 0.25), score, best, n, "threshold_fail")
124
+
125
+ strong = sum(
126
+ 1 for cand in candidates[:10]
127
+ if (s := compute_signals(span, cand.get("text", ""), "Ayah"))["coverage"] >= 0.80 and s["lcs_ratio"] >= 0.75
128
+ )
129
+ if strong >= 2:
130
+ return self._result("Correct", 0.60 + min(0.20, strong * 0.05), score, best, n, "borderline_multi_cov")
131
+ return self._result("Incorrect", 0.58, score, best, n, "borderline_default")
132
+
133
+ def _verify_hadith(self, span: str) -> Verification:
134
+ cfg = self.cfg
135
+ candidates = self.kb.search_hadith(span, top_k=cfg.hadith_top_k)
136
+ if not candidates:
137
+ return self._result("Incorrect", 0.75, 0.0, None, 0, "no_candidates")
138
+ top_retrieval = candidates[0].get("retrieval_score", 0.0)
139
+ score, best = best_match_score(span, candidates, "Hadith")
140
+ signals = best.get("signals", {}) if best else {}
141
+ coverage, is_substring = signals.get("coverage", 0.0), signals.get("is_substring", 0)
142
+ n = len(candidates)
143
+
144
+ if is_substring and coverage >= cfg.hadith_min_coverage and top_retrieval >= cfg.hadith_retrieval_guard:
145
+ return self._result("Correct", min(0.97, 0.80 + score * 0.17), score, best, n, "substring_match", top_retrieval)
146
+ if score >= cfg.hadith_correct_threshold and coverage >= cfg.hadith_min_coverage:
147
+ return self._result("Correct", min(0.92, 0.65 + score * 0.27), score, best, n, "threshold_pass", top_retrieval)
148
+ if score <= cfg.hadith_uncertain_low:
149
+ return self._result("Incorrect", min(0.90, 0.65 + (1 - score) * 0.25), score, best, n, "threshold_fail", top_retrieval)
150
+
151
+ moderate = sum(
152
+ 1 for cand in candidates[:8]
153
+ if (s := compute_signals(span, cand.get("text", ""), "Hadith"))["coverage"] >= 0.65 and s["lcs_ratio"] >= 0.55
154
+ )
155
+ if moderate >= 2 and top_retrieval >= 0.30:
156
+ return self._result("Correct", 0.58 + min(0.22, moderate * 0.06), score, best, n, "borderline_multi_cov", top_retrieval)
157
+ if top_retrieval < 0.25 or score < 0.45:
158
+ return self._result("Incorrect", 0.60, score, best, n, "borderline_low_retrieval", top_retrieval)
159
+ return self._result("Incorrect", 0.55, score, best, n, "borderline_default", top_retrieval)
160
+
161
+ @staticmethod
162
+ def _result(verdict, confidence, score, best, n_candidates, method, top_retrieval=0.0) -> Verification:
163
+ return Verification(verdict, round(confidence, 4), round(score, 4), method, best, n_candidates, round(top_retrieval, 4))
164
+
165
+
166
+ # --------------------------------------------------------------------------------------------------------------
167
+ # Correction (Subtask 1C): locate the true ayah window / Hadith record and return its exact text
168
+ # --------------------------------------------------------------------------------------------------------------
169
+ @dataclass
170
+ class CorrectionMatch:
171
+ kind: str # 'Ayah' | 'Hadith'
172
+ strength: float # coverage (Quran) / symmetric containment (Hadith)
173
+ full_ratio: float
174
+ text: str # proposed correction in the official 1C format (idgham + '(n)' ayah markers)
175
+ source: dict # reference metadata
176
+ display: str = "" # clean human-readable version
177
+
178
+
179
+ class Corrector:
180
+ def __init__(self, retriever: SourceRetriever, config: Optional[CorrectorConfig] = None) -> None:
181
+ self.kb = retriever
182
+ self.cfg = config or CorrectorConfig()
183
+
184
+ def match(self, span_text: str, content_type: str) -> Optional[CorrectionMatch]:
185
+ return self.match_quran(span_text) if content_type == "Ayah" else self.match_hadith(span_text)
186
+
187
+ def match_quran(self, query_text: str) -> Optional[CorrectionMatch]:
188
+ """Best window of 1-8 consecutive ayahs (dynamic sliding window over BM25 seeds, exact-result pruning)."""
189
+ kb = self.kb
190
+ query_norm = normalize_for_matching(query_text)
191
+ query_words = content_words(query_norm.split())
192
+ if not query_words:
193
+ return None
194
+ query_len = len(query_norm)
195
+ memo: Dict[tuple, tuple] = {}
196
+ best = None # (key, coverage, ratio, surah, start, end)
197
+ for seed in kb.quran_seed_ayahs(query_words, top_k=25):
198
+ surah, ayah = kb.quran[seed]["surah_id"], kb.quran[seed]["ayah_id"]
199
+ ayahs = kb.quran_by_surah[surah]
200
+ min_ayah, max_ayah = min(ayahs), max(ayahs)
201
+ for offset in range(3):
202
+ start = ayah - offset
203
+ if start < min_ayah:
204
+ continue
205
+ window_len = -1
206
+ for length in range(1, self.cfg.max_window + 1):
207
+ end = start + length - 1
208
+ if end > max_ayah:
209
+ break
210
+ window_len += len(kb.q_norm_match[ayahs[end]]) + 1
211
+ len_diff = abs(window_len - query_len)
212
+ upper_bound = min(1.0, window_len / max(query_len, 1))
213
+ if best is not None: # exact-result pruning
214
+ best_cov, best_neg = best[0][0], best[0][1]
215
+ if upper_bound < best_cov or (upper_bound == best_cov and -len_diff < best_neg):
216
+ continue
217
+ key_pos = (surah, start, end)
218
+ if key_pos in memo:
219
+ continue
220
+ window = " ".join(kb.q_norm_match[ayahs[a]] for a in range(start, end + 1))
221
+ matcher = difflib.SequenceMatcher(None, query_norm, window, autojunk=False)
222
+ matched = sum(b.size for b in matcher.get_matching_blocks() if b.size >= 4)
223
+ coverage = matched / max(query_len, 1)
224
+ key = (coverage, -len_diff, matcher.ratio())
225
+ memo[key_pos] = key
226
+ if best is None or key > best[0]:
227
+ best = (key, coverage, key[2], surah, start, end)
228
+ if best is None:
229
+ return None
230
+ _, coverage, ratio, surah, start, end = best
231
+ display = " ".join(kb.quran[kb.quran_by_surah[surah][a]]["text"] for a in range(start, end + 1))
232
+ return CorrectionMatch(
233
+ "Ayah", coverage, ratio, self._ayah_text(surah, start, end),
234
+ {"type": "Quran", "surah_id": surah, "surah_name": kb.quran[kb.quran_by_surah[surah][start]]["surah_name"],
235
+ "ayah_start": start, "ayah_end": end},
236
+ display,
237
+ )
238
+
239
+ def _ayah_text(self, surah: int, start: int, end: int) -> str:
240
+ kb, multi = self.kb, end > start
241
+ parts = []
242
+ for a in range(start, end + 1):
243
+ text = kb.quran[kb.quran_by_surah[surah][a]]["text"]
244
+ parts.append(f"{text} ({a})" if multi else text)
245
+ return apply_idgham(" ".join(parts)).replace("\u0640", "")
246
+
247
+ def match_hadith(self, query_text: str) -> Optional[CorrectionMatch]:
248
+ kb = self.kb
249
+ query_norm = normalize_for_matching(query_text)
250
+ query_words = content_words(query_norm.split())
251
+ if not query_words:
252
+ return None
253
+ query_len = len(query_norm)
254
+ best = None # (key, idx, field, ratio)
255
+ for idx in kb.hadith_candidates(query_words, self.cfg.hadith_top_k):
256
+ for field_name in ("matn", "full"):
257
+ text = kb.hadith_norm(idx, field_name)
258
+ if not text:
259
+ continue
260
+ upper_bound = min(1.0, query_len / max(len(text), 1))
261
+ if best is not None and upper_bound < best[0][0]:
262
+ continue
263
+ matcher = difflib.SequenceMatcher(None, query_norm, text, autojunk=False)
264
+ matched = sum(b.size for b in matcher.get_matching_blocks() if b.size >= 4)
265
+ coverage, candidate_cov = matched / max(query_len, 1), matched / max(len(text), 1)
266
+ key = (min(coverage, candidate_cov), matcher.ratio())
267
+ if best is None or key > best[0]:
268
+ best = (key, idx, field_name, key[1])
269
+ if best is None:
270
+ return None
271
+ key, idx, field_name, ratio = best
272
+ record = kb.hadith[idx]
273
+ text = record[field_name].strip()
274
+ return CorrectionMatch(
275
+ "Hadith", key[0], ratio, text,
276
+ {"type": "Hadith", "hadithID": record["hadithID"], "book": record["book"], "title": record["title"],
277
+ "field": "matn" if field_name == "matn" else "full_text"},
278
+ text,
279
+ )
280
+
281
+
282
+ # --------------------------------------------------------------------------------------------------------------
283
+ # End-to-end pipeline
284
+ # --------------------------------------------------------------------------------------------------------------
285
+ STATUS_INFO = {
286
+ "VERIFIED": {"ar": "موثّق", "group": "verified"},
287
+ "CORRECTED": {"ar": "غير مطابق — يوجد تصحيح من المصدر", "group": "mismatch"},
288
+ "UNSUPPORTED": {"ar": "غير مطابق — لا يوجد مصدر مطابق", "group": "mismatch"},
289
+ "HUMAN_REVIEW": {"ar": "يحتاج مراجعة بشرية", "group": "review"},
290
+ }
291
+
292
+
293
+ class IslamicContentVerifier:
294
+ """Detect quotations, verify them against the corpora and decide: verified, corrected, unsupported or review."""
295
+
296
+ def __init__(self, retriever: Optional[SourceRetriever] = None, config: Optional[PipelineConfig] = None,
297
+ use_scanner: bool = True, decouple_triggers: bool = True) -> None:
298
+ self.cfg = config or PipelineConfig()
299
+ self.retriever = retriever or SourceRetriever()
300
+ self.verifier = Verifier(self.retriever, self.cfg.verifier)
301
+ self.corrector = Corrector(self.retriever, self.cfg.corrector)
302
+ self.detector = HybridDetector(self.retriever, use_scanner, rules_use_corpus=decouple_triggers,
303
+ decouple_triggers=decouple_triggers)
304
+ self.detector_name = type(self.detector).__name__
305
+
306
+ # ---- public API -----------------------------------------------------------------------------------------
307
+ def detect(self, text: str) -> List[DetectedSpan]:
308
+ return self.detector.detect(self._validate(text)) if text.strip() else []
309
+
310
+ def needs_hadith(self, text: str) -> bool:
311
+ """True if the text contains a Hadith quotation (so the large Hadith index must be loaded)."""
312
+ return any(span.label == "Hadith" for span in self.detect(text))
313
+
314
+ def analyze(self, text: str) -> dict:
315
+ """Run the full pipeline on a generated text."""
316
+ text = self._validate(text)
317
+ started = time.time()
318
+ spans = self.detector.detect(text) if text.strip() else []
319
+ detect_seconds = time.time() - started
320
+ result = self._analyze_spans(text, spans)
321
+ result["timings"] = {"detect_s": round(detect_seconds, 3), "total_s": round(time.time() - started, 3)}
322
+ return result
323
+
324
+ def analyze_spans(self, text: str, spans: List[dict]) -> dict:
325
+ """Skip detection and use given spans ``[{label, start, end}]`` (evaluation / oracle mode)."""
326
+ given = [DetectedSpan(s["start"], s["end"], s["label"], None, "given", text[s["start"]:s["end"]]) for s in spans]
327
+ return self._analyze_spans(self._validate(text), given)
328
+
329
+ def analyze_detected(self, text: str, spans: List[DetectedSpan]) -> dict:
330
+ """Verify spans produced elsewhere (e.g. detector + hosted model merged by ``camelbert_adapter.merge_spans``)."""
331
+ return self._analyze_spans(self._validate(text), spans)
332
+
333
+ # ---- internals ------------------------------------------------------------------------------------------
334
+ @staticmethod
335
+ def _validate(text: str) -> str:
336
+ if not isinstance(text, str):
337
+ raise TypeError("Input text must be a string")
338
+ if len(text) > MAX_INPUT_CHARS:
339
+ raise ValueError(f"Input is too long ({len(text)} characters); the limit is {MAX_INPUT_CHARS}")
340
+ return text
341
+
342
+ def _analyze_spans(self, text: str, spans: List[DetectedSpan]) -> dict:
343
+ reports = [self._process_span(i + 1, span) for i, span in enumerate(sorted(spans, key=lambda s: s.start))]
344
+ counts = {status: 0 for status in STATUS_INFO}
345
+ for report in reports:
346
+ counts[report["status"]] += 1
347
+ return {
348
+ "input_text": text,
349
+ "detector": self.detector_name,
350
+ "spans": reports,
351
+ "corrected_text": self._apply_corrections(text, reports),
352
+ "summary": {
353
+ "n_spans": len(reports),
354
+ "n_ayah": sum(r["type"] == "Ayah" for r in reports),
355
+ "n_hadith": sum(r["type"] == "Hadith" for r in reports),
356
+ **counts,
357
+ "needs_human_review": counts["HUMAN_REVIEW"] > 0,
358
+ },
359
+ }
360
+
361
+ def _process_span(self, index: int, span: DetectedSpan) -> dict:
362
+ try:
363
+ return self._decide(index, span)
364
+ except Exception: # a single failing quotation must not break the whole report
365
+ logger.exception("Failed to process span %d", index)
366
+ report = self._empty_report(index, span)
367
+ self._finalize(report, "HUMAN_REVIEW", {"code": "internal_error"})
368
+ return report
369
+
370
+ @staticmethod
371
+ def _empty_report(index: int, span: DetectedSpan) -> dict:
372
+ return {
373
+ "id": index, "type": span.label, "start": span.start, "end": span.end, "text": span.text,
374
+ "detection": {"backend": span.source, "confidence": None if span.confidence is None else round(span.confidence, 4)},
375
+ "verification": {"verdict": "Incorrect", "confidence": 0.0, "score": 0.0, "method": "error", "n_candidates": 0},
376
+ "evidence": None, "correction": None, "suggestion": None, "notes": [],
377
+ }
378
+
379
+ @staticmethod
380
+ def _finalize(report: dict, status: str, reason: dict) -> None:
381
+ info = STATUS_INFO[status]
382
+ report.update(status=status, status_ar=info["ar"], group=info["group"], reason=reason)
383
+
384
+ @staticmethod
385
+ def source_label(source: dict) -> str:
386
+ """Plain reference used in reasons, e.g. ``سورة البقرة 153`` or ``حديث رقم 5``."""
387
+ if source["type"] == "Quran":
388
+ start, end = source["ayah_start"], source["ayah_end"]
389
+ return f"سورة {source['surah_name']} {start}" + (f"–{end}" if end != start else "")
390
+ return f"حديث رقم {source['hadithID']}"
391
+
392
+ def _decide(self, index: int, span: DetectedSpan) -> dict:
393
+ report = self._empty_report(index, span)
394
+ verification = self.verifier.verify(span.text, span.label)
395
+ report["verification"] = {
396
+ "verdict": verification.verdict, "confidence": verification.confidence, "score": verification.best_score,
397
+ "method": verification.method, "n_candidates": verification.n_candidates,
398
+ }
399
+ whole = self._whole_ayah(span.text) if span.label == "Ayah" else None
400
+ if whole is not None:
401
+ self._verified_whole_ayah(report, span, whole) # a complete ayah, however short or common its words
402
+ elif span.label == "Ayah":
403
+ self._decide_quran(report, span, verification)
404
+ else:
405
+ self._decide_hadith(report, span, verification)
406
+ if whole is None and span.source == "rules" and len(normalize_for_matching(span.text).split()) < SHORT_QUOTE_WORDS:
407
+ # two words cannot identify a source: never claim "verified" or "wrong", and never correct
408
+ self._finalize(report, "HUMAN_REVIEW", {"code": "too_short"})
409
+ report["correction"] = None
410
+ report["suggestion"] = None
411
+ elif report["status"] in ("UNSUPPORTED", "HUMAN_REVIEW"):
412
+ self._cross_check(report, span)
413
+ if report["status"] == "VERIFIED" and span.hint and span.hint != span.label:
414
+ report["notes"].append({"code": "is_ayah" if span.label == "Ayah" else "is_hadith", "source": self.source_label(report["evidence"]["source"]), "misattributed": True})
415
+ return report
416
+
417
+ # ---- helpers for the decision ---------------------------------------------------------------------------
418
+ def _whole_ayah(self, text: str) -> Optional[int]:
419
+ """Index of the ayah whose complete text equals ``text`` (two words or more), else None."""
420
+ if getattr(self, "_whole_map", None) is None:
421
+ mapping: Dict[str, int] = {}
422
+ for i, norm in enumerate(self.retriever.q_norm_match):
423
+ if len(norm.split()) >= 2:
424
+ mapping.setdefault(norm, i)
425
+ self._whole_map = mapping
426
+ norm = normalize_for_matching(text)
427
+ return self._whole_map.get(norm) if len(norm.split()) >= 2 else None
428
+
429
+ def _verified_whole_ayah(self, report: dict, span: DetectedSpan, idx: int) -> None:
430
+ record = self.retriever.quran[idx]
431
+ source_ref = {"type": "Quran", "surah_id": record["surah_id"], "surah_name": record["surah_name"],
432
+ "ayah_start": record["ayah_id"], "ayah_end": record["ayah_id"]}
433
+ alignment = align(span.text, record["text"], self.retriever.quran_vocabulary)
434
+ report["evidence"] = {"source": source_ref, "signals": compute_signals(span.text, record["text"], "Ayah"),
435
+ "comparison": alignment}
436
+ report["verification"].update(verdict="Correct", confidence=0.98, score=1.0, method="whole_ayah")
437
+ self._finalize(report, "VERIFIED", {"code": "exact_match", "source": self.source_label(source_ref)})
438
+ if alignment["diacritic_notes"]:
439
+ report["notes"].append({"code": "diacritic_conflict", "n": len(alignment["diacritic_notes"])})
440
+
441
+ def _cross_check(self, report: dict, span: DetectedSpan) -> None:
442
+ """A quotation that fails against its claimed corpus but is found verbatim in the other one is probably
443
+ mis-attributed (a Hadith presented as an ayah, or the reverse): say so."""
444
+ if len(normalize_for_matching(span.text).split()) < 3:
445
+ return
446
+ try:
447
+ if span.label == "Ayah":
448
+ other = self.verifier._verify_hadith(span.text)
449
+ if other.verdict == "Correct" and other.method == "substring_match" and other.source:
450
+ report["notes"].append({"code": "is_hadith", "source": f"حديث رقم {other.source['hadithID']}"})
451
+ else:
452
+ other = self.verifier._verify_quran(span.text)
453
+ if other.verdict == "Correct" and other.method == "substring_match" and other.source:
454
+ c = other.source
455
+ report["notes"].append({"code": "is_ayah", "source": f"سورة {c['surah_name']} {c['ayah_id']}"})
456
+ except Exception: # a diagnostic note must never break the decision
457
+ logger.exception("cross-check failed")
458
+
459
+ @staticmethod
460
+ def _proposal(span: DetectedSpan, match: Optional[CorrectionMatch]) -> Optional[dict]:
461
+ if match is None:
462
+ return None
463
+ return {"text": match.text, "display_text": match.display, "source": match.source,
464
+ "match_strength": round(match.strength, 4), "full_ratio": round(match.full_ratio, 4)}
465
+
466
+ def _decide_quran(self, report: dict, span: DetectedSpan, verification: Verification) -> None:
467
+ cfg, corr_cfg = self.cfg, self.cfg.corrector
468
+ match = self.corrector.match_quran(span.text)
469
+ if match is not None:
470
+ source_text, source_ref = match.display, match.source
471
+ elif verification.source:
472
+ candidate = verification.source
473
+ source_text = candidate["text"]
474
+ source_ref = {"type": "Quran", "surah_id": candidate["surah_id"], "surah_name": candidate["surah_name"],
475
+ "ayah_start": candidate["ayah_id"], "ayah_end": candidate["ayah_id"]}
476
+ else:
477
+ source_text, source_ref = "", None
478
+ alignment = align(span.text, source_text, self.retriever.quran_vocabulary) if source_text else None
479
+ if source_ref:
480
+ report["evidence"] = {"source": source_ref, "signals": compute_signals(span.text, source_text, "Ayah"),
481
+ "comparison": alignment}
482
+ proposal = self._proposal(span, match)
483
+ n_tokens = len(content_words(normalize_for_matching(span.text).split())) if span.text else 0
484
+ has_tokens = alignment is not None and alignment["exact"] and len(alignment["word_diff"]) >= 1
485
+
486
+ if has_tokens and sum(len(op["span"].split()) for op in alignment["word_diff"]) >= MIN_EXACT_TOKENS:
487
+ self._finalize(report, "VERIFIED", {"code": "exact_match", "source": self.source_label(source_ref)})
488
+ if alignment["diacritic_notes"]:
489
+ report["notes"].append({"code": "diacritic_conflict", "n": len(alignment["diacritic_notes"])})
490
+ if alignment["orthographic_variants"]:
491
+ report["notes"].append({"code": "orthographic_variant", "n": alignment["orthographic_variants"]})
492
+ return
493
+
494
+ local = bool(alignment and alignment.get("near") and alignment.get("word_similarity", 0) >= 0.75 and alignment.get("source_excerpt"))
495
+ strong = match is not None and match.strength >= corr_cfg.quran_strong and (match.full_ratio >= corr_cfg.min_full_ratio or local)
496
+ if strong:
497
+ self._finalize(report, "CORRECTED", {"code": "altered_passage", "source": self.source_label(source_ref),
498
+ "n": alignment["mismatches"] if alignment else 0,
499
+ "reordered": bool(alignment and alignment["reordered"])})
500
+ # replace only the passage the author quoted, not the whole ayah that contains it
501
+ excerpt = alignment["source_excerpt"] if alignment and alignment["source_excerpt"] else proposal["display_text"]
502
+ report["correction"] = {**proposal, "display_text": excerpt, "applied": True}
503
+ return
504
+
505
+ if verification.verdict == "Correct":
506
+ self._finalize(report, "HUMAN_REVIEW", {"code": "weak_match"})
507
+ report["suggestion"] = proposal
508
+ elif match is None or match.strength < corr_cfg.quran_low:
509
+ confident = verification.confidence >= cfg.unsupported_min_conf and not verification.method.startswith("borderline")
510
+ if match is None or match.strength < cfg.unsupported_strength or confident:
511
+ self._finalize(report, "UNSUPPORTED", {"code": "no_source"})
512
+ else:
513
+ self._finalize(report, "HUMAN_REVIEW", {"code": "insufficient_evidence"})
514
+ report["suggestion"] = proposal
515
+ else:
516
+ self._finalize(report, "HUMAN_REVIEW", {"code": "candidate_not_strong", "source": self.source_label(source_ref),
517
+ "strength": round(match.strength, 2)})
518
+ report["suggestion"] = proposal
519
+
520
+ def _decide_hadith(self, report: dict, span: DetectedSpan, verification: Verification) -> None:
521
+ cfg, corr_cfg = self.cfg, self.cfg.corrector
522
+ best = verification.source
523
+ alignment = None
524
+ if best:
525
+ alignment = align(span.text, best["text"])
526
+ report["evidence"] = {
527
+ "source": {"type": "Hadith", "hadithID": best["hadithID"], "book": best["book"], "title": best["title"]},
528
+ "signals": best.get("signals"), "comparison": alignment,
529
+ }
530
+ exact_tokens = alignment is not None and alignment["exact"] and sum(len(op["span"].split()) for op in alignment["word_diff"]) >= 4
531
+
532
+ if verification.verdict == "Correct" or exact_tokens:
533
+ altered = (not exact_tokens and alignment is not None and not alignment["exact"] and alignment["mismatches"] >= 1)
534
+ if altered: # any added, dropped or changed word can alter the meaning (خيرًا / شرًّا, or a missing "لا"): only an exact match is verified
535
+ match = self.corrector.match_hadith(span.text)
536
+ self._finalize(report, "HUMAN_REVIEW", {"code": "hadith_altered", "n": alignment["mismatches"],
537
+ "source": f"حديث رقم {best['hadithID']}"})
538
+ report["suggestion"] = self._proposal(span, match)
539
+ report["verification"]["verdict"] = "Incorrect"
540
+ elif exact_tokens or (not verification.method.startswith("borderline") and verification.confidence >= cfg.verified_min_conf):
541
+ self._finalize(report, "VERIFIED", {"code": "exact_match" if exact_tokens else "close_match",
542
+ "source": f"حديث رقم {best['hadithID']}"})
543
+ if alignment and not alignment["exact"]:
544
+ report["notes"].append({"code": "hadith_minor_diffs", "n": alignment["mismatches"]})
545
+ if alignment and alignment["diacritic_notes"]:
546
+ report["notes"].append({"code": "diacritic_conflict", "n": len(alignment["diacritic_notes"])})
547
+ else:
548
+ self._finalize(report, "HUMAN_REVIEW", {"code": "weak_match"})
549
+ return
550
+
551
+ match = self.corrector.match_hadith(span.text)
552
+ proposal = self._proposal(span, match)
553
+ if match is None or match.strength < corr_cfg.hadith_low:
554
+ confident = verification.confidence >= cfg.unsupported_min_conf and not verification.method.startswith("borderline")
555
+ if match is None or match.strength < cfg.unsupported_strength or confident:
556
+ self._finalize(report, "UNSUPPORTED", {"code": "no_source"})
557
+ else:
558
+ self._finalize(report, "HUMAN_REVIEW", {"code": "insufficient_evidence"})
559
+ report["suggestion"] = proposal
560
+ else: # a similar narration exists, but Hadith is never corrected automatically
561
+ self._finalize(report, "HUMAN_REVIEW", {"code": "hadith_candidate", "source": self.source_label(match.source),
562
+ "strength": round(match.strength, 2)})
563
+ report["suggestion"] = proposal
564
+
565
+ @staticmethod
566
+ def _apply_corrections(text: str, reports: List[dict]) -> str:
567
+ out = text
568
+ for report in sorted(reports, key=lambda r: r["start"], reverse=True):
569
+ if report["status"] == "CORRECTED" and report["correction"] and report["correction"].get("applied"):
570
+ out = out[: report["start"]] + report["correction"]["display_text"] + out[report["end"]:]
571
+ return out