import re import os import nltk import random from flask_cors import CORS import numpy as np from datetime import datetime, timedelta from collections import defaultdict from flask import Flask, request, jsonify from transformers import pipeline from huggingface_hub import login from sentence_transformers import SentenceTransformer from keybert import KeyBERT from sklearn.cluster import KMeans from nltk.corpus import stopwords CORNELIA = r""" /$$$$$$ /$$$$$$ /$$$$$$$ /$$ /$$ /$$$$$$$$ /$$ /$$$$$$ /$$$$$$ /$$__ $$ /$$__ $$| $$__ $$| $$$ | $$| $$_____/| $$ |_ $$_/ /$$__ $$ | $$ \__/| $$ \ $$| $$ \ $$| $$$$| $$| $$ | $$ | $$ | $$ \ $$ | $$ | $$ | $$| $$$$$$$/| $$ $$ $$| $$$$$ | $$ | $$ | $$$$$$$$ | $$ | $$ | $$| $$__ $$| $$ $$$$| $$__/ | $$ | $$ | $$__ $$ | $$ $$| $$ | $$| $$ \ $$| $$\ $$$| $$ | $$ | $$ | $$ | $$ | $$$$$$/| $$$$$$/| $$ | $$| $$ \ $$| $$$$$$$$| $$$$$$$$ /$$$$$$| $$ | $$ \______/ \______/ |__/ |__/|__/ \__/|________/|________/|______/|__/ |__/ *#================#@@#***%%##%*+====+=====%++========++=======+*&&@@@*%%#@@@@#&%+==========&%*%+===+=====+==*@@@@@@@@@#==========*@&&%==+========% @======**%%%%*====@%%#&@@@&&@#*++*%%##%%+%#%*#%%%###%%#%#%#%*@&&@@@%*#&@@#%#%@*##%####%#*+@##*++****%=+%#%=#@@@@@@@@@*==+%%%**=+@@@@%===++++***% %=====*%%%%%+==%#&&@@@@@&&@@*+*****%#%%+#%%=%%%%#*#%##%##%%###%=*#+**#%%++======###%####%%%**#%%*+*++=*==**=#@@@@@@@@@%==+%%%%==&@@@@@+=+##%+++* *====+%%%%%+=*%&&@@@@@@&@@%++%*%%%%*%**#*++%%*%*%#%%&%##%#####++==============***#%%%%##%%*%%*%#@&#%*+===++==%&@@@@@@@@#==*%%%+=+@@@@@@+==+*%#%% %====*%***%=*#@@@@@@@@#@#%**%%%%%*%*%##%=+%*%%*%#%%&%%#%%%%#%+==========++++===**##%%%##%%%*****&&@@@@%*+==+++=%#@@@@@@@%==*%%%==+@@@@@@+==*+++# %===+%%%%%%=%@@@@@@@@&@@##*#%%**%%%#&%%=++*%+*%%*%&*%#%%#%#%====+++++++++++*++==++#%%%%###%*++***%#&&&&@@@%*%%#*==@@@@@@@*=+**%%==%@@@@@&==**+** %+==*%%%%%%+%@@@@@@@&@&&*##%%%%%%%++*%*+*%*+*#+*%#%%%#%===++*************++==+*#%%%*%#%*==+#%%%%%@&&@@@@@@&#%=&@@@@@@@==%%%%%==&@@@@%====**% %%+%%%%%%%*======*#@#@#@*###%%##%**#*++***+*@=+%%%*##*==++++**************++===%#%****%%+++*#%#%%#@&&&&@@@@@##@@@@@@@@%=+%%%%*=%@@@&=======# %**%%%%+===========%&@#@*##%%&%%#%*#%+++***+%@==*%**#%===+++++++++*******++++++===+#%++++*%*****#&&%**%@@@@@@@#&@@@@@@@@@==%%%%%==&@@&====+==+ %=#*%*==============&@#@*%%#&%%+%*=+*+***#@==+=+%##+======++++++++++++++++++++++*+#*++*+%*+*%#**#&&%%%*++&&&%&@@@@@@@@@%=%%%%%*=%@@@=====+=% %=&++=====+===++====%@#@@#%#%*%@%*+=+++*+%&&+=+==##%*@&&&*++=======+++++========***%###*+*%**++%#%*%#@&%%%#**%@@@@@@@@@@@#=*%%%%%=*@@@&======% %+*====*%%%*==++====%@#@#%+@#*+*++=++%#&&%=+=+++=======*%%***+##=+++++=*#@+#*%*+=++*%&*+=++=+%%%%%*%#@%*#%**&@@@@@@@@@@@==#####=+@@@@#=====# %+====*%%##+==++===+#@&@@%%#*%@%+%%*++++%#@&%**=+%#@@%%%%+*+****===%+**%+===%*===+*@++**===+**%%#%%%*%#&%%#*+%@@@@@@@@@@*=#####==@@@@@+====+ %====*###%%==+++===#&&&@*%#%@&%*%%%*++%##&%+*#@%#==+*==#===@+=+++++++++++++*=&&@@*====#==#@@*==+%%#%##%**%&@%#***#@@@@@@@@*=#####+=&@@@@&===== ======%**%===+====*@@@**%%#&&%%%%%%**###&@%#%%@**=========+%=#=##=++++++++++*#=====%#%*@=====*&@&+++**%###%*%#&%#%###@@@@@@@*=#####*=&@@@@@*==== ====@=***===+====+&&%**%*#@@#*%##%*%###@@%%%*@=#&*=+*=+**=========++++===+++=============+**#*=*@@&*****%###%*%&@#####@@@@@@%=#####*=&@@@@@&==== ====*======++===*&@*%%%+@&@#%%#%**%%#&@@*#%*@@*+%#==*=====++++++++++++=*=+++++++++++++++++=%#%=&@@@@@#****%##%*%&@#%#@@##@@@@@%=#####*=&@@@@@@===+ +=#+======++===*&@**#%*#&@&%*#%%%#%@@@&%#%%@@#@#=*#==**+++++++++****++=%=+++***++++++++++=+&**%#@@@@@@@@%***%%#*%&#@@@%@@@@@%=#####+=@@@@@@@*==+ *@======++====**@+%%#*%&@@&%%%%%%#@@##%#&&@&@&&@&*+===++=+++++*****+++===+++*********+++==*+%#@&&@@@&&@@%**+*#%%&&@@@&&@@@@@%=#####=*@@@@@@@*==+ @============%%*%*##%*@@@@#%%%#%&@&&&##@@&@#@@@%#&=========+*****+++=@=+++********++===+%%@&@@&@#@@&@&@@##*%%%#&&@@@@@@@*=#####=*@@@@@@@+==+ @==========+%%*&%###%%&@@#%##@@#@@&@#&@#&&@@&&%+=====+=++****+++=&=+++++****++====#&@@&@#@#@@&&@@&@&@@#%%%#&@@@#&@@@@&=+####%=%@@@@@@@===+ ==+=======*##+*#*###%%&@@##@@#@#@@#@##@@###%%%#&%+=*++++++++===+++++++++=====#@@#@@#@@#@#@@@@&&@#&&&@@#%#@@@@@@@@%=%%*+===+&@@@@@#==== ==========%##=&%%%##%%&@@@@&&&@@&@@%###@##%@@@@@@@@@@&*==*+++==============++====*%&@#@@@&@@#@#@@@@@&@@@&&@@&@@&%@@@@@@*=%+========&@@@*=*== ========*%###=#*%#%%#@@@&@&@@@&&@&&*%##&&&@@@@#####&@@&%%===++*=*#%%*%%%##%*+===+++@&&@@@#&@@&@@&@@@@@@#@@@#@&&&@&&%&@@@@@@+=*==========*@&=+%=+ ========*%###=&%*%#*++==+==+#@@@&@@@%%&@@@#&@@@@@&@&%&=+#====+===============@*=%#@@&&&@@@@#@@@@@@&@@#@@&@@@&@@&@@@@@@#====++====++==+*=##=* +======+=+###=&&@*===========*&@@&&@@@###&@@@@@@&&@@#@*=**@====*+====++====&@=*=&@#&@@@@@@@@@&@@@@@@#@#&&&@#@@&@@@@@@@*====+==+**%*====*#%+% ========+*==*##*==============*&@@#@@@@@@@&@##@@&&@#&@*=%+=@@=====++=====&@%=+==@@&@@#**++=*%@@@@@@#@@&@@&@@&@@@#*@@@@@@+======+%%#%%===+##%*% ======+@@#==+%&@*=====*+*+=======*@#%@@@&@@@#&@@@@&&&@&=+*===&@*======+#@*===+=%@#=+%=======%&@@@&&@@#&@@&@@#&@@@##@@@@@&======%*##%%====%#%+% ======@@@@&&@@%=====*%%%%%%==+===*+*++@@@@@@@@&%&%%##&&&&*=======%@@@@@#*+======%&*====%==+%*=+#&&&@@&&@@@###@@#@@@&%@@@@@@======#++*%*+===*#**% ======@@@@@@@======*#%###%%======*%#%+=++#@@@@*#@@@#%*=====+======================**==#&=%==&@@@@&&@&#@@@@@&&@@@%@@@@@@=======#%+======*#+%% =====*@@@@@@======#*####%%%=====+%%**%#%*=%&@@&%#@@#&%%&&%=**====*========#=*==+====%==+&&=====*#@#%%@@@@@@@@@@@%@@@@@@====+====@======%*+%% *====&@@@@@=======%*##%++%======@&%%**%%++#&@#%&&@@#%=*%*====++=====*+==+=+==+==%#&*+*%*==%@@@@@@@@@@@@@@@@@@@#@@&&@@@@@&====++===@=====+#+**% %====@@@@@========*===*##==+===#@@@@@%%*#**%&@&%*==*%%*===++%+==%====*=========*%=++==*&@@@@@@@@@@@@@@@@@@@@#@@#@@@@@@%====+====%=====*=+*=& %+==+@@@@=========@*##+=======&@&&@@@@@@%%#&@&@#&&%=*%%%*===***+=======%========+=&%==+&@@@@@@@@@@@@@@@@@@@@@@#@@@@@@@*===++===@======++*=## %%*=*@@@*========@===========#&&@@&&@@@@@@##@#&@#%=*%%%%*===+**=======+=========*+**#&@@@@@@@@@@@@@@@@@@@@@@@&&&%@@@@@@@+===++===&======*%+### %%*=%@@#==+=====&=====+====+%&@@&&@@#&@@@@@@&&@#@#@&@%=+%%%%*====*++======*=========*%=%@@@@@@@@@@@@@@@@@@@@@@@@@#&&@@@@@@@&========@======**+=+%# %%*=%@@+=***===#+=========*@@@@@@&&@@@@@@@@#@#@@#@*=*%%%*====++====+==*=========*%==&@@@@@@@@@@@@@@@@@@@@@@@&&&@@@@@@@@%=*+=====*=====*+====+# %%*=%@%=+%%%*=+@=+=======%@@@%&@@@@&&@@#@@@@@@&@@@&@#@*+=%%%*=====#===+==%====+=====***=#@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@*=**====@=====#=======# %%*=%@==%%%%==@=+%%%%%==%@@@@@%&@@@@@#@@#@@@@@&@@#@@&@#*=%%%*======+================%**=*@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@%==%%*==@=====+========% %===%&=======&+========*@@@@@@&%%%%%%##@@%%%%%%@##%%@##+===========+===================+=&@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@+=======*=============== @===&@======+@========*@@@@@@@@&&&&&&&&@@&&&&&&@&&&&@*=================================#@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@*======#===============& /$$$$$$ /$$$$$$ /$$$$$$$ /$$ /$$ /$$$$$$$$ /$$ /$$$$$$ /$$$$$$ /$$__ $$ /$$__ $$| $$__ $$| $$$ | $$| $$_____/| $$ |_ $$_/ /$$__ $$ | $$ \__/| $$ \ $$| $$ \ $$| $$$$| $$| $$ | $$ | $$ | $$ \ $$ | $$ | $$ | $$| $$$$$$$/| $$ $$ $$| $$$$$ | $$ | $$ | $$$$$$$$ | $$ | $$ | $$| $$__ $$| $$ $$$$| $$__/ | $$ | $$ | $$__ $$ | $$ $$| $$ | $$| $$ \ $$| $$\ $$$| $$ | $$ | $$ | $$ | $$ | $$$$$$/| $$$$$$/| $$ | $$| $$ \ $$| $$$$$$$$| $$$$$$$$ /$$$$$$| $$ | $$ \______/ \______/ |__/ |__/|__/ \__/|________/|________/|______/|__/ |__/ """ print(CORNELIA) MIN_CLUSTER_SIZE = 3 # ── Use env var for token — never hardcode ─────────────────────────────────── _hf_token = os.environ.get("HF_TOKEN") if _hf_token: login(_hf_token) # ───────────────────────────────────────────── # CONSTANTS # ───────────────────────────────────────────── CONTRACTIONS = { r"don't": "do not", r"doesn't": "does not", r"didn't": "did not", r"can't": "cannot", r"couldn't": "could not", r"won't": "will not", r"wouldn't": "would not", r"isn't": "is not", r"aren't": "are not", r"wasn't": "was not", r"weren't": "were not", r"haven't": "have not", r"hasn't": "has not", r"hadn't": "had not", r"it's": "it is", r"that's": "that is", r"there's": "there is", r"i'm": "i am", r"i've": "i have", r"i'd": "i would", r"i'll": "i will", r"you're": "you are", r"they're": "they are", r"we're": "we are", r"he's": "he is", r"she's": "she is", r"let's": "let us", } CONTRACTION_PATTERNS = [ (re.compile(pattern, re.IGNORECASE), replacement) for pattern, replacement in CONTRACTIONS.items() ] nltk.download("stopwords", quiet=True) KEYWORD_BLOCKLIST = set(stopwords.words("english")) | { "wi", "fi", "app", "really", "like", "love", "work", "works", "great", "good", "nice", "use", "needs", "need", "feels", "feel", "makes", "make", "just", "also", "even", "much", "many", "very", } AGE_BUCKETS = [ ("18–24", 18, 24), ("25–34", 25, 34), ("35–44", 35, 44), ("45–54", 45, 54), ("55–64", 55, 64), ("65–74", 65, 74), ("75–84", 75, 84), ("85–94", 85, 94), ("95–98", 95, 98), ("99+", 99, 999), ] # ───────────────────────────────────────────── # CPU-FRIENDLY BATCH SIZE # Smaller batches reduce memory pressure on CPU # and prevent the tokenizer from stalling. # ───────────────────────────────────────────── SENTIMENT_BATCH_SIZE = 8 # Truncate at 128 tokens — comments are short, # 512 is wasteful and slows every inference call. MAX_TOKEN_LENGTH = 128 # ───────────────────────────────────────────── # HELPERS # ───────────────────────────────────────────── def expand_contractions(text: str) -> str: for pattern, replacement in CONTRACTION_PATTERNS: text = pattern.sub(replacement, text) return text def clean_keywords(kw_list: list[str]) -> list[str]: return [ kw for kw in kw_list if len(kw) >= 4 and kw.lower() not in KEYWORD_BLOCKLIST ] def run_sentiment_batch(texts: list[str]) -> list[str]: """ Run sentiment classifier over a list of texts and return a list of lowercase label strings in the same order. Single batched call — never call this inside a loop. """ if not texts: return [] results = sentiment_classifier( texts, truncation=True, max_length=MAX_TOKEN_LENGTH, batch_size=SENTIMENT_BATCH_SIZE, ) return [r["label"].lower() for r in results] def labels_to_breakdown(labels: list[str]) -> dict: """Convert a list of sentiment label strings to a percentage breakdown dict.""" counts = {"positive": 0, "neutral": 0, "negative": 0} for label in labels: if label in counts: counts[label] += 1 total = len(labels) or 1 return { "positive": round(counts["positive"] / total * 100, 1), "neutral": round(counts["neutral"] / total * 100, 1), "negative": round(counts["negative"] / total * 100, 1), } def get_sentiment_breakdown(texts: list[str]) -> dict: """Convenience wrapper — batch sentiment → percentage breakdown.""" return labels_to_breakdown(run_sentiment_batch(texts)) def cluster_and_label(texts: list[str], embeddings: np.ndarray, reduced: np.ndarray): """ Shared clustering + keyword-extraction logic. Returns (labels_array, cluster_keywords_dict). """ try: import hdbscan clusterer = hdbscan.HDBSCAN( min_cluster_size=2, min_samples=1, metric="euclidean", cluster_selection_method="leaf" ) labels = clusterer.fit_predict(reduced) except ImportError: from sklearn.preprocessing import normalize labels = KMeans(n_clusters=7, random_state=42, n_init=10).fit_predict( normalize(embeddings) ) cluster_keywords = {} for label in sorted(set(labels)): if label == -1: continue indices = [j for j, l in enumerate(labels) if l == label] cluster_docs = [texts[j] for j in indices] expanded = " ".join(expand_contractions(doc) for doc in cluster_docs) keywords = kw_model.extract_keywords( expanded, keyphrase_ngram_range=(1, 1), stop_words="english", top_n=10, use_mmr=True, diversity=0.6, ) cluster_keywords[label] = clean_keywords([kw for kw, _ in keywords])[:6] return labels, cluster_keywords def reduce_embeddings(embeddings: np.ndarray, n_components: int = 5): """ UMAP reduction with PCA fallback. On CPU, PCA is often faster and accurate enough for clustering. UMAP is used only when available and the dataset is large enough to benefit from non-linear reduction. """ try: import umap # Only bother with UMAP on larger datasets; PCA is fine for small ones if embeddings.shape[0] < 50: raise ImportError("Skipping UMAP for small dataset — using PCA") reducer = umap.UMAP( n_components=n_components, random_state=42, min_dist=0.0, metric="cosine", ) return reducer.fit_transform(embeddings) except (ImportError, Exception): from sklearn.decomposition import PCA n = min(n_components, embeddings.shape[0] - 1) return PCA(n_components=n, random_state=42).fit_transform(embeddings) # ───────────────────────────────────────────── # MODELS — loaded once at startup # ───────────────────────────────────────────── sentiment_classifier = pipeline( "sentiment-analysis", model="cardiffnlp/twitter-roberta-base-sentiment-latest", truncation=True, max_length=MAX_TOKEN_LENGTH, ) emotion_classifier = pipeline( "text-classification", model="SamLowe/roberta-base-go_emotions", top_k=None, truncation=True, max_length=MAX_TOKEN_LENGTH, ) embedder = SentenceTransformer("all-MiniLM-L6-v2") kw_model = KeyBERT(model=embedder) # ───────────────────────────────────────────── # HELPER FUNCS # ───────────────────────────────────────────── def find_optimal_k(embeddings, k_min=3, k_max=5): inertias = [] k_range = range(k_min, min(k_max + 1, len(embeddings))) for k in k_range: km = KMeans(n_clusters=k, random_state=42, n_init=10) km.fit(embeddings) inertias.append(km.inertia_) drops = [inertias[i] - inertias[i + 1] for i in range(len(inertias) - 1)] return list(k_range)[np.argmax(drops) + 1] def merge_small_clusters(cluster_ids, k, embeddings, kmeans: KMeans): counts = defaultdict(int) for c in cluster_ids: counts[c] += 1 small_clusters = {c for c, count in counts.items() if count < MIN_CLUSTER_SIZE} if not small_clusters: return cluster_ids large_cluster_ids = [c for c in range(k) if c not in small_clusters] large_centroids = kmeans.cluster_centers_[large_cluster_ids] new_ids = [] for i, c in enumerate(cluster_ids): if c in small_clusters: dists = np.linalg.norm(large_centroids - embeddings[i], axis=1) new_ids.append(large_cluster_ids[np.argmin(dists)]) else: new_ids.append(c) return new_ids # ───────────────────────────────────────────── # APP # ───────────────────────────────────────────── app = Flask(__name__) CORS(app) @app.route('/') def index(): return """
This is the inference server powering CORNELIA's comment analysis pipeline. It exposes a set of NLP endpoints consumed by the Flutter app to generate real-time sentiment, emotion, topic, and demographic insights from user comments.
A RoBERTa model fine-tuned on ~124M tweets for 3-class sentiment classification: positive, neutral, and negative. Used to produce sentiment distributions, sentiment over time, country-level gender breakdowns, and negative outlier scoring.
A RoBERTa model fine-tuned on Google's GoEmotions dataset, capable of classifying text into 28 fine-grained emotion categories. CORNELIA uses the top 5 non-neutral emotions weighted by score to produce the emotion distribution chart.
A lightweight sentence transformer that maps text to dense 384-dimensional vectors. Used as the backbone for topic clustering (HDBSCAN/KMeans), intertopic distance mapping (UMAP/PCA), keyword co-occurrence graph construction, emerging trend detection over time, and age-group aspect sentiment correlation.
Built on top of all-MiniLM-L6-v2, KeyBERT extracts the most semantically representative keywords from each topic cluster using MMR (Maximal Marginal Relevance) to balance relevance and diversity. Keywords label clusters, drive the network graph, and surface emerging issues.
✅ All models loaded and ready.