File size: 53,664 Bytes
df9c352 e47737d | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578 579 580 581 582 583 584 585 586 587 588 589 590 591 592 593 594 595 596 597 598 599 600 601 602 603 604 605 606 607 608 609 610 611 612 613 614 615 616 617 618 619 620 621 622 623 624 625 626 627 628 629 630 631 632 633 634 635 636 637 638 639 640 641 642 643 644 645 646 647 648 649 650 651 652 653 654 655 656 657 658 659 660 661 662 663 664 665 666 667 668 669 670 671 672 673 674 675 676 677 678 679 680 681 682 683 684 685 686 687 688 689 690 691 692 693 694 695 696 697 698 699 700 701 702 703 704 705 706 707 708 709 710 711 712 713 714 715 716 717 718 719 720 721 722 723 724 725 726 727 728 729 730 731 732 733 734 735 736 737 738 739 740 741 742 743 744 745 746 747 748 749 750 751 752 753 754 755 756 757 758 759 760 761 762 763 764 765 766 767 768 769 770 771 772 773 774 775 776 777 778 779 780 781 782 783 784 785 786 787 788 789 790 791 792 793 794 795 796 797 798 799 800 801 802 803 804 805 806 807 808 809 810 811 812 813 814 815 816 817 818 819 820 821 822 823 824 825 826 827 828 829 830 831 832 833 834 835 836 837 838 839 840 841 842 843 844 845 846 847 848 849 850 851 852 853 854 855 856 857 858 859 860 861 862 863 864 865 866 867 868 869 870 871 872 873 874 875 876 877 878 879 880 881 882 883 884 885 886 887 888 889 890 891 892 893 894 895 896 897 898 899 900 901 902 903 904 905 906 907 908 909 910 911 912 913 914 915 916 917 918 919 920 921 922 923 924 925 926 927 928 929 930 931 932 933 934 935 936 937 938 939 940 941 942 943 944 945 946 947 948 949 950 951 952 953 954 955 956 957 958 959 960 961 962 963 964 965 966 967 968 969 970 971 972 973 974 975 976 977 978 979 980 981 982 983 984 985 986 987 988 989 990 991 992 993 994 995 996 997 998 999 1000 1001 1002 1003 1004 1005 1006 1007 1008 1009 1010 1011 1012 1013 1014 1015 1016 1017 1018 1019 1020 1021 1022 1023 1024 1025 1026 1027 1028 1029 1030 1031 1032 1033 1034 1035 1036 1037 | import re
import math
import json
import string
import tempfile
import os
from collections import Counter
from html.parser import HTMLParser
from typing import Dict, List, Tuple, Any, Optional
import xml.etree.ElementTree as ET
import gradio as gr
# =====================================================================
# CUSTOM CORE PARSERS & HELPERS (NO EXTERNAL NLP LIBRARIES)
# =====================================================================
class SimpleHTMLInspector(HTMLParser):
"""
A lightweight, pure-Python HTML inspector utilizing standard HTMLParser.
Fulfills educational HTML analysis without external BeautifulSoup or library dependencies.
"""
def __init__(self):
super().__init__()
self.tags: List[str] = []
self.content_map: List[Tuple[str, str]] = []
self.tag_stack: List[str] = []
def handle_starttag(self, tag: str, attrs: List[Tuple[str, Optional[str]]]):
self.tags.append(tag)
self.tag_stack.append(tag)
def handle_data(self, data: str):
cleaned = data.strip()
if cleaned:
current_tag = self.tag_stack[-1] if self.tag_stack else "text"
# Ignore style/script data for clean text preview
if current_tag not in ["script", "style"]:
self.content_map.append((current_tag, cleaned))
def handle_endtag(self, tag: str):
if self.tag_stack and self.tag_stack[-1] == tag:
self.tag_stack.pop()
def parse_csv_custom(text: str, delimiter: str = ",") -> Tuple[List[str], List[List[str]], int, int]:
"""
A custom state-machine CSV parser built from scratch.
Complies with rules: No import csv, manages quote boundaries, escaped quotes, and newlines correctly.
"""
lines = text.splitlines()
if not lines:
return [], [], 0, 0
parsed_rows: List[List[str]] = []
for line in lines:
if not line.strip():
continue
row_fields = []
current_field = []
in_quotes = False
i = 0
n = len(line)
while i < n:
char = line[i]
if in_quotes:
if char == '"':
# Lookahead for escaped quote
if i + 1 < n and line[i+1] == '"':
current_field.append('"')
i += 2
continue
else:
in_quotes = False
else:
current_field.append(char)
else:
if char == '"':
in_quotes = True
elif char == delimiter:
row_fields.append("".join(current_field))
current_field = []
else:
current_field.append(char)
i += 1
row_fields.append("".join(current_field))
parsed_rows.append(row_fields)
if not parsed_rows:
return [], [], 0, 0
headers = parsed_rows[0]
data_rows = parsed_rows[1:] if len(parsed_rows) > 1 else []
col_count = len(headers)
row_count = len(parsed_rows)
return headers, data_rows, row_count, col_count
def detect_dominant_delimiter(text: str) -> str:
"""
Analyzes lines to guess if the delimiter is comma, semicolon, or tab.
Looks for the highest count with consistency across initial lines.
"""
candidates = [",", ";", "\t"]
sample_lines = [l for l in text.splitlines()[:5] if l.strip()]
if not sample_lines:
return ","
best_delim = ","
best_score = -1
for delim in candidates:
counts = [line.count(delim) for line in sample_lines]
avg_count = sum(counts) / len(counts)
# Minimize standard deviation for structural consistency
variance = sum((c - avg_count)**2 for c in counts) / len(counts)
if avg_count > 0.5:
score = avg_count / (1.0 + variance) # high frequency, low variation
if score > best_score:
best_score = score
best_delim = delim
return best_delim
# =====================================================================
# TEXT ANALYZER CORE LOGIC
# =====================================================================
class TextAnalyzer:
"""
The monolithic analysis processor wrapping all character,
unicode, word, sentence, regex, and structured analyses.
"""
def __init__(self, text: str):
self.text = text
self.bytes_data = text.encode("utf-8")
# Simple structural tokenizer
self.lines = text.splitlines()
# Word extraction without NLP: lowercase for counts, standard regex boundary matches
self.words_raw = re.findall(r'\b[a-zA-Z0-9_\'-]+\b', text)
self.words = [w.strip() for w in self.words_raw if w.strip()]
# Sentence segmentation heuristic (split on . ! ? followed by space/line bounds)
self.sentences = [s.strip() for s in re.split(r'[.!?]+(?=\s|$)', text) if s.strip()]
def get_overview(self) -> Dict[str, Any]:
"""Module 1: General document measurements."""
char_count = len(self.text)
word_count = len(self.words)
sentence_count = max(1 if word_count > 0 and not self.sentences else 0, len(self.sentences))
line_count = len(self.lines)
# Paragraphs: split by multiple empty lines
paragraphs = [p for p in re.split(r'\n\s*\n', self.text) if p.strip()]
paragraph_count = len(paragraphs)
byte_count = len(self.bytes_data)
avg_word_length = sum(len(w) for w in self.words) / word_count if word_count > 0 else 0.0
avg_sentence_length = word_count / sentence_count if sentence_count > 0 else 0.0
return {
"char_count": char_count,
"word_count": word_count,
"sentence_count": sentence_count,
"line_count": line_count,
"paragraph_count": paragraph_count,
"byte_count": byte_count,
"avg_word_length": round(avg_word_length, 2),
"avg_sentence_length": round(avg_sentence_length, 2),
}
def generate_encoding_table(self, max_chars: int = 150) -> List[List[str]]:
"""
Module 2: Technical character encodings mappings.
Limits rows to avoid crashing browser windows with massive datasets.
"""
rows = []
for char in self.text[:max_chars]:
dec = ord(char)
hex_val = f"0x{dec:X}"
bin_val = f"{dec:08b}"
# ASCII validation
ascii_repr = char if dec < 128 else "N/A"
if dec < 32 or dec == 127:
ascii_repr = "Control Char" if dec != 10 and dec != 9 else ("[LF]" if dec == 10 else "[TAB]")
unicode_point = f"U+{dec:04X}"
# UTF-8 details
utf8_bytes = char.encode("utf-8")
utf8_len = len(utf8_bytes)
utf8_bytes_str = " ".join(f"{b:02X}" for b in utf8_bytes)
rows.append([
char if dec >= 32 else " ",
ascii_repr,
unicode_point,
str(dec),
hex_val,
bin_val,
f"{utf8_len} byte(s)",
utf8_bytes_str
])
return rows
def generate_unicode_explorer(self, max_chars: int = 100) -> str:
"""
Module 3: Educational breakdown highlighting the prefix bit structure of UTF-8.
Provides highly descriptive, interactive visual representation of multibyte UTF-8.
"""
html = ['<div class="space-y-4">']
limit_text = self.text[:max_chars]
for idx, char in enumerate(limit_text):
dec_val = ord(char)
cp = f"U+{dec_val:04X}"
utf8_b = char.encode("utf-8")
html.append('<div class="p-3 rounded-lg border border-slate-200 dark:border-slate-700 bg-white dark:bg-slate-800 flex items-center justify-between shadow-sm hover:shadow-md transition">')
# Character Display Panel
char_display = char if dec_val >= 32 else f"<span class='text-xs text-amber-500 font-mono'>CTRL({dec_val})</span>"
html.append(f'<div class="flex items-center space-x-4">')
html.append(f' <div class="w-12 h-12 bg-teal-50 dark:bg-teal-950 text-teal-700 dark:text-teal-300 rounded-lg flex items-center justify-center font-bold text-2xl border border-teal-200">{char_display}</div>')
html.append(f' <div>')
html.append(f' <div class="text-sm font-bold text-slate-800 dark:text-slate-100">Character #{idx + 1}</div>')
html.append(f' <div class="text-xs font-mono text-slate-500">{cp}</div>')
html.append(f' </div>')
html.append(f'</div>')
# Binary byte representation detail
html.append('<div class="text-right font-mono text-xs">')
html.append('<div class="text-slate-400 mb-1">UTF-8 Bit Structure:</div>')
for b in utf8_b:
bin_str = f"{b:08b}"
# Format markers with color spans according to UTF-8 rule
if len(utf8_b) == 1:
# Single Byte (ASCII): 0xxxxxxx
formatted_bin = f"<span class='text-green-600 dark:text-green-400 font-bold'>0</span>{bin_str[1:]}"
lbl = "ASCII Pattern"
elif len(utf8_b) == 2:
if b == utf8_b[0]:
formatted_bin = f"<span class='text-blue-600 dark:text-blue-400 font-bold'>110</span>{bin_str[3:]}"
lbl = "2-Byte Header"
else:
formatted_bin = f"<span class='text-purple-600 dark:text-purple-400 font-bold'>10</span>{bin_str[2:]}"
lbl = "Data Byte"
elif len(utf8_b) == 3:
if b == utf8_b[0]:
formatted_bin = f"<span class='text-orange-600 dark:text-orange-400 font-bold'>1110</span>{bin_str[4:]}"
lbl = "3-Byte Header"
else:
formatted_bin = f"<span class='text-purple-600 dark:text-purple-400 font-bold'>10</span>{bin_str[2:]}"
lbl = "Data Byte"
else: # 4-byte sequences
if b == utf8_b[0]:
formatted_bin = f"<span class='text-red-600 dark:text-red-400 font-bold'>11110</span>{bin_str[5:]}"
lbl = "4-Byte Header (Emoji/Rare)"
else:
formatted_bin = f"<span class='text-purple-600 dark:text-purple-400 font-bold'>10</span>{bin_str[2:]}"
lbl = "Data Byte"
html.append(f'<div class="flex items-center justify-end space-x-2">')
html.append(f' <span class="text-[10px] text-slate-400 italic">({lbl})</span>')
html.append(f' <span class="bg-slate-100 dark:bg-slate-900 px-1.5 py-0.5 rounded tracking-widest">{formatted_bin}</span>')
html.append(f'</div>')
html.append('</div>') # end byte display
html.append('</div>') # end card
html.append('</div>')
if len(self.text) > max_chars:
html.append(f'<div class="p-3 text-center text-xs text-amber-600 bg-amber-50 rounded border border-amber-200 mt-2">Displaying first {max_chars} characters. Your input has {len(self.text)} total chars.</div>')
return "\n".join(html)
def get_char_frequencies(self) -> List[Tuple[str, int, float]]:
"""Module 4: Sort and measure character densities."""
total = len(self.text)
if total == 0:
return []
cnt = Counter(self.text)
sorted_chars = cnt.most_common()
return [(c, count, round((count / total) * 100, 2)) for c, count in sorted_chars]
def get_word_frequencies(self) -> List[Tuple[str, int]]:
"""Module 5: Tokenize, clean and compute high vocabulary counts."""
if not self.words:
return []
normalized_words = [w.lower() for w in self.words]
cnt = Counter(normalized_words)
return cnt.most_common()
def get_string_statistics(self) -> Dict[str, Any]:
"""Module 6: Text measurements and distributions."""
unique_words = len(set(w.lower() for w in self.words))
unique_chars = len(set(self.text))
whitespace_count = sum(1 for c in self.text if c.isspace())
tab_count = self.text.count('\t')
newline_count = self.text.count('\n')
digit_count = sum(1 for c in self.text if c.isdigit())
uppercase_count = sum(1 for c in self.text if c.isupper())
lowercase_count = sum(1 for c in self.text if c.islower())
# Punctuation set (Standard English keys + Unicode counterparts)
punct_set = set(string.punctuation) | {'โ', 'โ', 'โ', 'โ', 'โ', 'โ', 'โฆ', 'ยฟ', 'ยก'}
punctuation_count = sum(1 for c in self.text if c in punct_set)
longest_word = max(self.words, key=len) if self.words else ""
shortest_word = min(self.words, key=len) if self.words else ""
return {
"longest_word": longest_word,
"shortest_word": shortest_word,
"unique_words": unique_words,
"unique_characters": unique_chars,
"whitespace_count": whitespace_count,
"tab_count": tab_count,
"newline_count": newline_count,
"digit_count": digit_count,
"uppercase_count": uppercase_count,
"lowercase_count": lowercase_count,
"punctuation_count": punctuation_count
}
def run_regex_explorer(self) -> Dict[str, List[str]]:
"""
Module 7: Matches semantic groups matching pre-defined structural regex rules.
"""
patterns = {
"Emails": r'[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}',
"URLs": r'https?://(?:www\.)?[-a-zA-Z0-9@:%._\+~#=]{1,256}\.[a-zA-Z0-9()]{1,6}\b(?:[-a-zA-Z0-9()@:%_\+.~#?&//=]*)',
"Phone numbers": r'\+?[0-9]{1,4}?[-.\s]?(?:\([0-9]{1,3}\)|[0-9]{1,3})[-.\s]?[0-9]{1,4}[-.\s]?[0-9]{1,4}[-.\s]?[0-9]{1,9}',
"Dates": r'\d{4}[-/.]\d{1,2}[-/.]\d{1,2}|\d{1,2}[-/.]\d{1,2}[-/.]\d{2,4}',
"Times": r'\b(?:[01]?\d|2[0-3]):[0-5]\d(?::[0-5]\d)?\s?(?:AM|PM|am|pm)?\b',
"Numbers": r'\b\d+(?:\.\d+)?\b',
"Currency values": r'(?:\$|โฌ|ยฃ|ยฅ|โน)\s?\d+(?:,\d{3})*(?:\.\d+)?\b|\b\d+(?:\.\d+)?\s?(?:USD|EUR|GBP|JPY|INR)\b',
"Hashtags": r'#[a-zA-Z0-9_]+',
"Mentions": r'@[a-zA-Z0-9_]+',
"IP addresses": r'\b(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)\.){3}(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)\b'
}
matches = {}
for title, pattern in patterns.items():
found = re.findall(pattern, self.text)
# Remove empty strings or simple fragments
matches[title] = [f.strip() for f in found if f.strip()]
return matches
def detect_format(self) -> Tuple[str, str]:
"""
Module 8: Text Format Detection.
Evaluates criteria and supplies natural explaining logic for findings.
"""
txt = self.text.strip()
if not txt:
return "Empty Input", "No text provided to detect structural formats."
# JSON detection
try:
json.loads(txt)
return "JSON (Valid)", "Parsed cleanly into key-value trees or arrays using standard python serialization (json.loads)."
except json.JSONDecodeError:
if (txt.startswith('{') and txt.endswith('}')) or (txt.startswith('[') and txt.endswith(']')):
return "JSON (Malformed)", "Begins/ends with brackets ({}, []), but contains syntactic errors like misplaced commas or string literals."
# XML detection
try:
ET.fromstring(txt)
return "XML (Valid)", "Parsed successfully using standard ElementTree XML architecture."
except ET.ParseError:
if re.search(r'<\?xml', txt, re.I) or (txt.startswith('<') and txt.endswith('>')):
return "XML (Malformed)", "Starts/ends with tags or contains XML headers, but fails parser schemas."
# HTML detection
html_sig_tags = r'<!DOCTYPE html|<html|<body|<div\s+|<span\s+|<p\s+|<a\s+href'
tags_found = re.findall(r'<[a-zA-Z1-6]+(?:\s+[^>]*)*>', txt)
if re.search(html_sig_tags, txt, re.I) or len(tags_found) > 4:
return "HTML", f"Contains signature tags (matched {len(tags_found)} raw HTML elements) or document structures like '<!DOCTYPE>'."
# Markdown detection
md_points = 0
reasons = []
if re.search(r'^(?:#|##|###|####|#####|######)\s+.+', txt, re.M):
md_points += 2
reasons.append("Contains structural section indicators (# Header)")
if re.search(r'\[.+?\]\(https?://.+?\)', txt):
md_points += 2
reasons.append("Identified Markdown link structures: [text](URL)")
if re.search(r'^[*-]\s+\w+', txt, re.M):
md_points += 1
reasons.append("Identified bullet lists (- or *)")
if re.search(r'^```\w*\n', txt, re.M):
md_points += 3
reasons.append("Identified structural code fencing tags (```)")
if md_points >= 3:
return "Markdown", f"Identified markdown layout markers: {', '.join(reasons)}."
# CSV/TSV detection
delim = detect_dominant_delimiter(self.text)
lines_with_delim = [l for l in self.text.splitlines() if delim in l]
if len(lines_with_delim) >= 2:
# Check consistency of delimiter count
counts = [l.count(delim) for l in lines_with_delim[:5]]
avg_count = sum(counts) / len(counts)
variance = sum((c - avg_count)**2 for c in counts) / len(counts)
if avg_count > 0 and variance < 1.5:
delim_name = "Comma" if delim == "," else ("Semicolon" if delim == ";" else "Tab")
return f"CSV / Delimited Text", f"Structured grid properties detected using consistently spaced delimiter: '{delim_name}' (average spacing density: {avg_count:.1f} per line)."
return "Plain Text (General)", "Default categorization. Contains no specialized programmatic structure, markup schemas, or clear delimiters."
# =====================================================================
# UI GENERATION HELPER FUNCTIONS
# =====================================================================
def analyze_all_inputs(text: str) -> List[Any]:
"""
Main callback updating all visual components in Gradio blocks concurrently.
"""
if not text or not text.strip():
# Fallback values
empty_overview_html = "<div class='text-center text-slate-500 py-6'>Please supply valid text or load a sample.</div>"
empty_tbl = []
return [
empty_overview_html, empty_tbl, empty_overview_html, empty_tbl, empty_tbl,
empty_overview_html, empty_overview_html, empty_overview_html, empty_overview_html, ""
]
analyzer = TextAnalyzer(text)
# Overview
ov = analyzer.get_overview()
ov_html = f"""
<div class="grid grid-cols-2 md:grid-cols-4 gap-4 mb-4">
<div class="bg-indigo-50 dark:bg-indigo-950 p-4 rounded-xl border border-indigo-100 text-center">
<span class="block text-indigo-500 text-xs font-semibold uppercase tracking-wider mb-1">Characters</span>
<span class="text-3xl font-extrabold text-indigo-900 dark:text-indigo-100">{ov['char_count']:,}</span>
</div>
<div class="bg-blue-50 dark:bg-blue-950 p-4 rounded-xl border border-blue-100 text-center">
<span class="block text-blue-500 text-xs font-semibold uppercase tracking-wider mb-1">Words</span>
<span class="text-3xl font-extrabold text-blue-900 dark:text-blue-100">{ov['word_count']:,}</span>
</div>
<div class="bg-emerald-50 dark:bg-emerald-950 p-4 rounded-xl border border-emerald-100 text-center">
<span class="block text-emerald-500 text-xs font-semibold uppercase tracking-wider mb-1">Sentences</span>
<span class="text-3xl font-extrabold text-emerald-900 dark:text-emerald-100">{ov['sentence_count']:,}</span>
</div>
<div class="bg-purple-50 dark:bg-purple-950 p-4 rounded-xl border border-purple-100 text-center">
<span class="block text-purple-500 text-xs font-semibold uppercase tracking-wider mb-1">Bytes (UTF-8)</span>
<span class="text-3xl font-extrabold text-purple-900 dark:text-purple-100">{ov['byte_count']:,}</span>
</div>
</div>
<div class="grid grid-cols-2 md:grid-cols-4 gap-4">
<div class="bg-slate-50 dark:bg-slate-900 p-4 rounded-xl border border-slate-100 text-center">
<span class="block text-slate-500 text-xs font-semibold uppercase tracking-wider mb-1">Lines</span>
<span class="text-xl font-bold text-slate-800 dark:text-slate-200">{ov['line_count']:,}</span>
</div>
<div class="bg-slate-50 dark:bg-slate-900 p-4 rounded-xl border border-slate-100 text-center">
<span class="block text-slate-500 text-xs font-semibold uppercase tracking-wider mb-1">Paragraphs</span>
<span class="text-xl font-bold text-slate-800 dark:text-slate-200">{ov['paragraph_count']:,}</span>
</div>
<div class="bg-slate-50 dark:bg-slate-900 p-4 rounded-xl border border-slate-100 text-center">
<span class="block text-slate-500 text-xs font-semibold uppercase tracking-wider mb-1">Avg Word Length</span>
<span class="text-xl font-bold text-slate-800 dark:text-slate-200">{ov['avg_word_length']} <small class="text-xs text-slate-400">chars</small></span>
</div>
<div class="bg-slate-50 dark:bg-slate-900 p-4 rounded-xl border border-slate-100 text-center">
<span class="block text-slate-500 text-xs font-semibold uppercase tracking-wider mb-1">Avg Sentence Length</span>
<span class="text-xl font-bold text-slate-800 dark:text-slate-200">{ov['avg_sentence_length']} <small class="text-xs text-slate-400">words</small></span>
</div>
</div>
"""
# Visual Charts using Tailwind (Avoid high load graphics libraries)
chars_stat = analyzer.get_string_statistics()
total_measurable = max(1, chars_stat["uppercase_count"] + chars_stat["lowercase_count"] + chars_stat["digit_count"] + chars_stat["punctuation_count"] + chars_stat["whitespace_count"])
def pct(val):
return (val / total_measurable) * 100
overview_chart_html = f"""
<div class="mt-6 p-4 rounded-xl border border-slate-200 dark:border-slate-700 bg-white dark:bg-slate-800">
<h4 class="text-sm font-bold text-slate-800 dark:text-slate-200 mb-3">Character Class Distribution Ratio:</h4>
<div class="w-full flex h-6 rounded-lg overflow-hidden border border-slate-300 dark:border-slate-600 mb-3">
<div style="width: {pct(chars_stat['lowercase_count'])}%" class="bg-emerald-500 hover:opacity-90" title="Lowercase ({chars_stat['lowercase_count']})"></div>
<div style="width: {pct(chars_stat['uppercase_count'])}%" class="bg-blue-500 hover:opacity-90" title="Uppercase ({chars_stat['uppercase_count']})"></div>
<div style="width: {pct(chars_stat['digit_count'])}%" class="bg-amber-500 hover:opacity-90" title="Digits ({chars_stat['digit_count']})"></div>
<div style="width: {pct(chars_stat['punctuation_count'])}%" class="bg-purple-500 hover:opacity-90" title="Punctuation ({chars_stat['punctuation_count']})"></div>
<div style="width: {pct(chars_stat['whitespace_count'])}%" class="bg-slate-400 hover:opacity-90" title="Whitespace ({chars_stat['whitespace_count']})"></div>
</div>
<div class="grid grid-cols-2 sm:grid-cols-5 gap-2 text-xs">
<div class="flex items-center space-x-1.5"><span class="w-3 h-3 bg-emerald-500 rounded-sm"></span> <span>Lowercase ({chars_stat['lowercase_count']})</span></div>
<div class="flex items-center space-x-1.5"><span class="w-3 h-3 bg-blue-500 rounded-sm"></span> <span>Uppercase ({chars_stat['uppercase_count']})</span></div>
<div class="flex items-center space-x-1.5"><span class="w-3 h-3 bg-amber-500 rounded-sm"></span> <span>Digits ({chars_stat['digit_count']})</span></div>
<div class="flex items-center space-x-1.5"><span class="w-3 h-3 bg-purple-500 rounded-sm"></span> <span>Punctuation ({chars_stat['punctuation_count']})</span></div>
<div class="flex items-center space-x-1.5"><span class="w-3 h-3 bg-slate-400 rounded-sm"></span> <span>Whitespace ({chars_stat['whitespace_count']})</span></div>
</div>
</div>
"""
# Encodings Table
encodings_data = analyzer.generate_encoding_table()
# Unicode Inspector
unicode_html = analyzer.generate_unicode_explorer()
# Frequencies
char_freqs = [[repr(c)[1:-1] if c != '\n' else '[LF]', count, f"{pct:.1f}%"] for c, count, pct in analyzer.get_char_frequencies()]
word_freqs = [[word, count] for word, count in analyzer.get_word_frequencies()[:100]]
# Detailed statistics mappings
stats_data = [
["Property Metric", "Value", "Educational Context"],
["Longest word length", f"{len(chars_stat['longest_word'])} chars ('{chars_stat['longest_word']}')" if chars_stat['longest_word'] else "N/A", "Useful for finding outlier anomalies or unspaced strings."],
["Shortest word length", f"{len(chars_stat['shortest_word'])} chars ('{chars_stat['shortest_word']}')" if chars_stat['shortest_word'] else "N/A", "Averages lower for basic grammar prepositions."],
["Unique vocabulary (Words)", str(chars_stat['unique_words']), "Vocabulary density indicator before lemmatization / stemming."],
["Unique character keys", str(chars_stat['unique_characters']), "Alphabet size of current document schema."],
["Whitespace instances", str(chars_stat['whitespace_count']), "Sum of space, newline, tab occurrences."],
["Tab character counts", str(chars_stat['tab_count']), "Important identifier for tab-separated grids or indent properties."],
["Line feeds (Newlines)", str(chars_stat['newline_count']), "Tells us structural grouping spacing parameters."],
["Number instances", str(chars_stat['digit_count']), "Shows the count of raw numerals."],
["Uppercase keys", str(chars_stat['uppercase_count']), "Casing properties indicating structural sentences or nouns."],
["Lowercase keys", str(chars_stat['lowercase_count']), "Standard core text body."],
["Punctuation marks", str(chars_stat['punctuation_count']), "Sentence partitions and code punctuation keys."]
]
# Regex Matches HTML Builder
regex_matches = analyzer.run_regex_explorer()
regex_html = ['<div class="space-y-4">']
for category, list_matches in regex_matches.items():
count = len(list_matches)
color = "emerald" if count > 0 else "slate"
badge_class = f"bg-{color}-100 text-{color}-800 dark:bg-{color}-950 dark:text-{color}-300"
regex_html.append(f'<div class="p-4 rounded-xl border border-slate-200 dark:border-slate-700 bg-white dark:bg-slate-800 shadow-sm">')
regex_html.append(f' <div class="flex items-center justify-between border-b border-slate-100 dark:border-slate-700 pb-2 mb-2">')
regex_html.append(f' <h4 class="text-sm font-bold text-slate-800 dark:text-slate-100">{category}</h4>')
regex_html.append(f' <span class="px-2 py-0.5 text-xs font-semibold rounded-full {badge_class}">{count} matches</span>')
regex_html.append(f' </div>')
if count > 0:
# Highlight items in tags
items_html = " ".join(f'<span class="inline-block px-2.5 py-1 bg-slate-100 dark:bg-slate-900 text-slate-800 dark:text-slate-200 rounded font-mono text-xs m-1 border border-slate-200 dark:border-slate-800">{m}</span>' for m in set(list_matches[:30]))
if len(list_matches) > 30:
items_html += f' <span class="text-xs text-slate-400 italic">...and {len(list_matches)-30} more</span>'
regex_html.append(f' <div class="flex flex-wrap">{items_html}</div>')
else:
regex_html.append(f' <p class="text-xs text-slate-400 italic">No matches detected in this pattern configuration.</p>')
regex_html.append('</div>')
regex_html.append('</div>')
regex_html_str = "\n".join(regex_html)
# Text Formats Detection
format_type, format_desc = analyzer.detect_format()
format_html_str = f"""
<div class="p-6 rounded-xl border border-indigo-100 dark:border-indigo-900/50 bg-indigo-50/50 dark:bg-indigo-950/30">
<span class="text-[10px] uppercase font-extrabold tracking-widest text-indigo-500 block mb-1">Identified Schema Format</span>
<h3 class="text-2xl font-black text-indigo-900 dark:text-indigo-100 mb-2">{format_type}</h3>
<p class="text-sm text-indigo-700 dark:text-indigo-300 font-medium">{format_desc}</p>
</div>
"""
# Sub-Inspectors (CSV, HTML)
csv_html_str = "<p class='text-xs text-slate-400 italic'>CSV structures not detected. This explorer will display formatting tables if text looks tabular.</p>"
if "CSV" in format_type or "," in text or ";" in text or "\t" in text:
delim = detect_dominant_delimiter(text)
try:
headers, data_rows, r_cnt, c_cnt = parse_csv_custom(text, delim)
if r_cnt > 0:
# Limit row views
preview_rows = data_rows[:10]
rows_html = "".join(f"<tr class='border-b border-slate-100 dark:border-slate-800 text-slate-600 dark:text-slate-300'>" + "".join(f"<td class='px-3 py-2 text-xs'>{col}</td>" for col in row) + "</tr>" for row in preview_rows)
headers_html = "".join(f"<th class='px-3 py-2 bg-slate-100 dark:bg-slate-900 text-left text-xs font-bold text-slate-700 dark:text-slate-300'>{h}</th>" for h in headers)
csv_html_str = f"""
<div class="mt-4 border border-slate-200 dark:border-slate-700 rounded-lg overflow-hidden bg-white dark:bg-slate-800">
<div class="p-3 bg-slate-50 dark:bg-slate-900 border-b border-slate-200 dark:border-slate-700 flex justify-between items-center text-xs">
<span class="font-bold text-slate-700 dark:text-slate-300">Parser Output: {r_cnt} Rows ร {c_cnt} Columns</span>
<span class="px-2 py-0.5 bg-slate-200 dark:bg-slate-800 rounded font-mono text-slate-600 dark:text-slate-400">Delimiter: '{delim}'</span>
</div>
<div class="overflow-x-auto">
<table class="w-full text-left border-collapse">
<thead><tr>{headers_html}</tr></thead>
<tbody>{rows_html}</tbody>
</table>
</div>
{f'<div class="p-2 text-center text-[11px] text-slate-400 bg-slate-50 border-t border-slate-200">Showing top 10 preview rows</div>' if len(data_rows) > 10 else ''}
</div>
"""
except Exception as e:
csv_html_str = f"<p class='text-xs text-red-500 italic'>Failed standard CSV matrix processing: {str(e)}</p>"
# HTML parsing extraction using SimpleHTMLInspector
html_details_str = "<p class='text-xs text-slate-400 italic'>HTML markup pattern matches not found.</p>"
if "HTML" in format_type or "<" in text:
try:
parser = SimpleHTMLInspector()
parser.feed(text)
tag_counts = Counter(parser.tags)
tags_stat_html = " ".join(f"<span class='inline-flex items-center px-2 py-0.5 rounded text-xs font-semibold bg-blue-100 text-blue-800 dark:bg-blue-950 dark:text-blue-300 m-1'><{tag}> ({cnt})</span>" for tag, cnt in tag_counts.items())
# Content mapping lists
content_preview_html = "".join(f"<div class='p-2 bg-slate-50 dark:bg-slate-900 border-b border-slate-100 dark:border-slate-800 flex justify-between items-start text-xs'><span class='font-mono text-teal-600 font-bold'><{tag}></span><span class='text-slate-600 dark:text-slate-300 text-right w-2/3'>{data}</span></div>" for tag, data in parser.content_map[:15])
html_details_str = f"""
<div class="mt-4 space-y-4">
<div class="p-3 border border-slate-200 dark:border-slate-700 bg-white dark:bg-slate-800 rounded-lg">
<h5 class="text-xs font-bold text-slate-700 dark:text-slate-300 mb-2">Tag Counts (Structure Streams)</h5>
<div class="flex flex-wrap">{tags_stat_html if tags_stat_html else '<span class="text-xs text-slate-400 italic">No tag keys found.</span>'}</div>
</div>
<div class="border border-slate-200 dark:border-slate-700 bg-white dark:bg-slate-800 rounded-lg overflow-hidden">
<div class="p-2.5 bg-slate-50 dark:bg-slate-900 border-b border-slate-200 text-xs font-bold text-slate-700 dark:text-slate-300">
Tag Text Extraction Preview
</div>
<div>
{content_preview_html if content_preview_html else '<div class="p-3 text-xs italic text-slate-400">No clean text stream mapped inside tags.</div>'}
</div>
{f'<div class="p-2 text-center text-[11px] text-slate-400 bg-slate-50 border-t border-slate-200">Displaying first 15 mapped nodes</div>' if len(parser.content_map) > 15 else ''}
</div>
</div>
"""
except Exception as e:
html_details_str = f"<p class='text-xs text-red-500 italic'>HTML parser trace error: {str(e)}</p>"
return [
ov_html,
encodings_data,
unicode_html,
char_freqs,
word_freqs,
stats_data,
regex_html_str,
format_html_str,
csv_html_str,
html_details_str,
overview_chart_html
]
def handle_file_upload(file_obj) -> str:
"""
Reads the file path securely and returns decoded text.
Handles decoding failures gracefully by falling back to errors or latin-1.
"""
if file_obj is None:
return ""
try:
# File object passed is a tempfile wrapper path in Gradio
with open(file_obj.name, "rb") as f:
bytes_content = f.read()
try:
return bytes_content.decode("utf-8")
except UnicodeDecodeError:
# Fallback to Latin-1 encoding to preserve characters cleanly
return bytes_content.decode("latin-1")
except Exception as e:
return f"File load error: {str(e)}"
def generate_report(text: str) -> str:
"""
Fulfills Module 11: Export downloadable analysis reports.
Formats analysis sections into a beautifully structured Markdown report file.
"""
if not text or not text.strip():
# Fallback empty path
temp_file = tempfile.NamedTemporaryFile(delete=False, suffix=".md")
temp_file.write("# No analysis data available\nPlease submit text first.".encode('utf-8'))
temp_file.close()
return temp_file.name
analyzer = TextAnalyzer(text)
ov = analyzer.get_overview()
chars_stat = analyzer.get_string_statistics()
fmt, desc = analyzer.detect_format()
regex_matches = analyzer.run_regex_explorer()
char_freqs = analyzer.get_char_frequencies()[:20]
word_freqs = analyzer.get_word_frequencies()[:20]
report = []
report.append("# Text Explorer Analysis Report")
report.append(f"Generated on: 2026-07-05 (System Analysis)\n")
report.append("---")
report.append("## 1. Document Overview Statistics")
report.append(f"- **Characters (Count)**: {ov['char_count']}")
report.append(f"- **Words (Count)**: {ov['word_count']}")
report.append(f"- **Sentences**: {ov['sentence_count']}")
report.append(f"- **Line count**: {ov['line_count']}")
report.append(f"- **Paragraphs**: {ov['paragraph_count']}")
report.append(f"- **Bytes (Size)**: {ov['byte_count']} bytes")
report.append(f"- **Average Word Length**: {ov['avg_word_length']} characters")
report.append(f"- **Average Sentence Length**: {ov['avg_sentence_length']} words\n")
report.append("## 2. Text Format Classification")
report.append(f"- **Detected Structure Format**: {fmt}")
report.append(f"- **Deduction Reason**: {desc}\n")
report.append("## 3. String & Lexical Statistics")
report.append(f"- **Longest word**: '{chars_stat['longest_word']}'")
report.append(f"- **Shortest word**: '{chars_stat['shortest_word']}'")
report.append(f"- **Unique words**: {chars_stat['unique_words']}")
report.append(f"- **Unique characters**: {chars_stat['unique_characters']}")
report.append(f"- **Whitespaces total**: {chars_stat['whitespace_count']}")
report.append(f"- **Tabs**: {chars_stat['tab_count']}")
report.append(f"- **Newlines**: {chars_stat['newline_count']}")
report.append(f"- **Numerics (Digits)**: {chars_stat['digit_count']}")
report.append(f"- **Uppercase characters**: {chars_stat['uppercase_count']}")
report.append(f"- **Lowercase characters**: {chars_stat['lowercase_count']}")
report.append(f"- **Punctuation keys**: {chars_stat['punctuation_count']}\n")
report.append("## 4. Top 20 Character Frequencies")
report.append("| Character | Count | Percentage |")
report.append("| :--- | :--- | :--- |")
for char, cnt, pct in char_freqs:
c_repr = repr(char)[1:-1] if char != '\n' else '[LF]'
report.append(f"| `{c_repr}` | {cnt} | {pct:.1f}% |")
report.append("\n")
report.append("## 5. Top 20 Vocabulary Word Frequencies")
report.append("| Word | Count |")
report.append("| :--- | :--- |")
for word, cnt in word_freqs:
report.append(f"| {word} | {cnt} |")
report.append("\n")
report.append("## 6. Regex Inspector Discoveries")
for key, items in regex_matches.items():
if items:
report.append(f"- **{key}** ({len(items)} found): {', '.join(set(items[:15]))}")
else:
report.append(f"- **{key}**: 0 instances detected.")
report.append("\n---\n*Report generated by Text Explorer Space (Pure Python Educational Analyzer).*")
# Write out to temporary local file path
temp_dir = tempfile.gettempdir()
file_path = os.path.join(temp_dir, "text_explorer_report.md")
with open(file_path, "w", encoding="utf-8") as f:
f.write("\n".join(report))
return file_path
# =====================================================================
# CUSTOM CSS STYLE DEFINITIONS FOR THE GRADIO THEME
# =====================================================================
CSS = """
body {
background-color: #f8fafc;
}
.gradio-container {
font-family: ui-sans-serif, system-ui, -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, "Helvetica Neue", Arial, sans-serif;
}
.header-hero {
background: linear-gradient(135deg, #4f46e5 0%, #2563eb 100%);
color: white !important;
border-radius: 1rem;
padding: 2.5rem 2rem;
margin-bottom: 2rem;
text-align: center;
box-shadow: 0 4px 6px -1px rgb(0 0 0 / 0.1), 0 2px 4px -2px rgb(0 0 0 / 0.1);
}
.header-hero h1 {
color: white !important;
font-weight: 900 !important;
letter-spacing: -0.025em;
font-size: 2.25rem !important;
margin-bottom: 0.5rem;
}
.header-hero p {
color: #e0e7ff !important;
font-size: 1rem;
}
.educational-card {
background-color: #f0fdfa;
border: 1px solid #ccfbf1;
border-radius: 0.75rem;
padding: 1rem;
margin-bottom: 1rem;
}
.educational-card h4 {
color: #0f766e !important;
margin-top: 0 !important;
}
.educational-card p, .educational-card li {
color: #115e59 !important;
font-size: 0.875rem !important;
}
"""
# =====================================================================
# SAMPLE TEXTSETS FOR CONVENIENT USER EXPERIENCES
# =====================================================================
SAMPLES = {
"Simple English": "Hello World! Feel free to paste any text here to test the NLP Text Explorer system. It works immediately without external libraries.",
"Multilingual & Emoji": "Multilingual test: Bonjour, ใใใซใกใฏ, ์๋
ํ์ธ์! Here is an emoji breakdown to inspect UTF-8 bytes: ๐ ๐ ๐. What is their representation in binary memory space?",
"HTML document sample": '<!DOCTYPE html>\n<html>\n<head>\n <title>Sample Sandbox</title>\n</head>\n<body>\n <div class="content">\n <h1>Welcome to Text Explorer!</h1>\n <p>This is standard markup designed to verify our pure HTML-Parser extraction utilities.</p>\n <a href="https://example.com">Visit our project</a>\n </div>\n</body>\n</html>',
"Tabular CSV format": "username,email,role,joined_date\nsmith_john,john.smith@gmail.com,Administrator,2024-03-24\elizabeth_k,k.elizabeth@yahoo.com,Contributor,2025-05-15\ntech_support,support@domain.org,User,2026-01-10",
"Markdown document": (
"# Markdown Document Guide\n\n"
"Welcome to this structural text parsing review. Here are some key attributes:\n\n"
"- Support lists\n- Supports link items: [Hugging Face Space](https://huggingface.co/spaces)\n\n"
"## Sub-headers\n"
"```python\n"
"def test():\n"
" return 'Hello, Markdown analyzer!'\n"
"```"
)
}
# =====================================================================
# GRADIO BLOCKS APPLICATION VIEW
# =====================================================================
with gr.Blocks(css=CSS, title="Text Explorer - Educational Pre-NLP Inspector") as demo:
# Hero Title Banner
gr.HTML("""
<div class="header-hero">
<h1>๐ Text Explorer</h1>
<p>Analyze and demystify raw character representation, encodings, byte streams, and structural text layouts before high-level NLP tokenization.</p>
</div>
""")
with gr.Row():
# Input column
with gr.Column(scale=4):
gr.Markdown("### ๐ฅ Document Source Input")
input_text = gr.Textbox(
label="Source Text Area",
placeholder="Paste your paragraphs, CSV structure, HTML content, or multilingual strings here...",
lines=10,
max_lines=25,
elem_id="input-textbox"
)
with gr.Row():
file_upload = gr.File(
label="Or Upload Text File (.txt, .csv, .json, .html, .md)",
file_types=[".txt", ".csv", ".json", ".html", ".md"],
type="filepath",
)
gr.Markdown("๐ก **Educational Presets**")
with gr.Row():
# Loop samples to create buttons
for name, content in SAMPLES.items():
gr.Button(name, size="sm").click(
fn=lambda val=content: val,
outputs=input_text
)
# Output / Analysis display tab matrix
with gr.Column(scale=6):
gr.Markdown("### โก Real-Time Analytical Breakdown")
with gr.Tabs():
# Overview
with gr.TabItem("๐ Overview"):
with gr.Accordion("๐ Educational Spotlight: The Lexical Units", open=True):
gr.HTML("""
<div class="educational-card">
<h4>How Computers Partition Text</h4>
<p>Before an NLP model can comprehend context, the document must be mathematically measured:</p>
<ul style="list-style-type: disc; margin-left: 1.25rem; margin-top: 0.5rem;">
<li><b>Bytes</b>: The actual digital footprint in physical memory (UTF-8 allocates 1 to 4 bytes per character).</li>
<li><b>Words & Sentences</b>: Fundamental logic nodes. Without standard tools like NLTK/spaCy, computers extract tokens using punctuation borders and boundary regex.</li>
</ul>
</div>
""")
overview_html = gr.HTML("<div class='text-center text-slate-400 py-6'>Analyze text to generate document insights.</div>")
overview_chart = gr.HTML()
# Character Encodings
with gr.TabItem("๐ข Encodings Table"):
with gr.Accordion("๐ซ Educational Spotlight: Encodings Simplified", open=False):
gr.HTML("""
<div class="educational-card">
<h4>Encoding Architecture Explained</h4>
<p>Every character is assigned a unique tracking coordinate:</p>
<ul style="list-style-type: disc; margin-left: 1.25rem; margin-top: 0.5rem;">
<li><b>ASCII</b>: Traditional 7-bit system spanning decimal positions 0 through 127. Standard English keys only.</li>
<li><b>Unicode</b>: A universal catalog assigning identical codepoint indices (e.g., U+1F60A) for all global languages and emojis.</li>
<li><b>UTF-8</b>: An optimized encoding translating those codepoints into binary arrays, using smaller slices (1 byte) for English and wider slices for eastern languages or emojis.</li>
</ul>
</div>
""")
encoding_table = gr.DataFrame(
headers=["Char", "ASCII Representation", "Unicode Point", "Decimal", "Hexadecimal", "Binary Value", "UTF-8 Length", "UTF-8 Bytes"],
datatype=["str", "str", "str", "str", "str", "str", "str", "str"],
wrap=True
)
# Unicode Explorer
with gr.TabItem("๐ Unicode Explorer"):
with gr.Accordion("๐ Educational Spotlight: Bit Patterns of UTF-8", open=False):
gr.HTML("""
<div class="educational-card">
<h4>Deciphering the UTF-8 Multi-byte Rules</h4>
<p>UTF-8 dynamically sizes bytes based on prefix triggers. Inspect the binary tags below:</p>
<ul style="list-style-type: disc; margin-left: 1.25rem; margin-top: 0.5rem;">
<li><b>0xxxxxxx</b> (1 Byte): Matches normal English ASCII indices seamlessly.</li>
<li><b>110xxxxx 10xxxxxx</b> (2 Bytes): Directs unicode sequences (e.g. Greek / Cyrillic).</li>
<li><b>1110xxxx 10xxxxxx 10xxxxxx</b> (3 Bytes): Covers general Asian characters (CJK blocks).</li>
<li><b>11110xxx 10xxxxxx 10xxxxxx 10xxxxxx</b> (4 Bytes): Represents deep emoticons and rare historical glyphs.</li>
</ul>
</div>
""")
unicode_explorer_html = gr.HTML("<div class='text-center text-slate-400 py-6'>Displaying individual unicode bit structures...</div>")
# Frequencies
with gr.TabItem("๐ Frequencies"):
with gr.Row():
with gr.Column():
gr.Markdown("#### Character Frequency Analysis")
char_freq_table = gr.DataFrame(
headers=["Character Key", "Occurrence", "Ratio"],
datatype=["str", "number", "str"]
)
with gr.Column():
gr.Markdown("#### Word Frequency Analysis (Vocabulary)")
word_freq_table = gr.DataFrame(
headers=["Lowercased Token", "Occurrence"],
datatype=["str", "number"]
)
# Advanced Statistics
with gr.TabItem("๐งฎ Detailed Statistics"):
stats_table = gr.DataFrame(
headers=["Property Metric", "Value", "Educational Context"],
datatype=["str", "str", "str"],
wrap=True
)
# Regex Explorer
with gr.TabItem("๐ Regex Explorer"):
with gr.Accordion("๐ง Educational Spotlight: Structured Pattern Matching", open=False):
gr.HTML("""
<div class="educational-card">
<h4>Regex: The Structural Extraction Language</h4>
<p>Regular expressions look for matching text layouts. They are extremely effective before models compile lexical vectors:</p>
<ul style="list-style-type: disc; margin-left: 1.25rem; margin-top: 0.5rem;">
<li><b>Emails</b>: Matches user accounts bounded by domains (<code>username@host.ext</code>).</li>
<li><b>IP Addresses</b>: Recognizes IPv4 address streams (digits separated by decimal boundaries).</li>
</ul>
</div>
""")
regex_explorer_html = gr.HTML("<div class='text-center text-slate-400 py-6'>Matches category analysis lists...</div>")
# Format Detectors
with gr.TabItem("๐งฑ Layout Formats"):
with gr.Accordion("๐ฆ Educational Spotlight: Layout Analysis Systems", open=False):
gr.HTML("""
<div class="educational-card">
<h4>Markup and Matrix Structure Detection Rules</h4>
<p>Computers parse document systems using distinctive delimiters and structural tags:</p>
<ul style="list-style-type: disc; margin-left: 1.25rem; margin-top: 0.5rem;">
<li><b>HTML/XML</b>: Uses bracket boundaries (<code><tag></code>) to represent nested elements.</li>
<li><b>CSV Grid Structure</b>: Uses consistent grid markers (such as commas) containing identical segments per line row.</li>
</ul>
</div>
""")
format_overview_html = gr.HTML()
with gr.Accordion("CSV Preview Engine (Non-Library Machine)", open=True):
csv_preview_html = gr.HTML("<p class='text-xs text-slate-400 italic'>No CSV parsed grid display loaded yet.</p>")
with gr.Accordion("HTML Tree Parser Detail", open=True):
html_preview_html = gr.HTML("<p class='text-xs text-slate-400 italic'>No HTML tag analysis details loaded yet.</p>")
# Download Export
with gr.TabItem("๐พ Export Report"):
gr.Markdown("### Create Inspection Summary File")
gr.Markdown("Download a complete, beautifully structured Markdown text report mapping the calculations of the explorer for external study.")
report_btn = gr.Button("๐จ Compile Markdown Analysis Report", variant="primary")
report_file = gr.File(label="Downloadable Analysis Report", type="filepath")
# =====================================================================
# CONTROLLER WIREFRAMES & ACTIONS
# =====================================================================
# Upload handler updating input text
file_upload.change(
fn=handle_file_upload,
inputs=file_upload,
outputs=input_text
)
# Core reactive interface processing all panels together
ui_outputs = [
overview_html,
encoding_table,
unicode_explorer_html,
char_freq_table,
word_freq_table,
stats_table,
regex_explorer_html,
format_overview_html,
csv_preview_html,
html_preview_html,
overview_chart
]
# Input changes trigger analytical updates instantly
input_text.change(
fn=analyze_all_inputs,
inputs=input_text,
outputs=ui_outputs
)
# Export report trigger
report_btn.click(
fn=generate_report,
inputs=input_text,
outputs=report_file
)
# Run the Gradio Space application
if __name__ == "__main__":
demo.launch() |