| import re |
| import os |
| import nltk |
| import random |
| from flask_cors import CORS |
| import numpy as np |
| from datetime import datetime, timedelta |
| from collections import defaultdict |
| from flask import Flask, request, jsonify |
| from transformers import pipeline |
| from huggingface_hub import login |
| from sentence_transformers import SentenceTransformer |
| from keybert import KeyBERT |
| from sklearn.cluster import KMeans |
| from nltk.corpus import stopwords |
|
|
|
|
|
|
| CORNELIA = r""" |
| /$$$$$$ /$$$$$$ /$$$$$$$ /$$ /$$ /$$$$$$$$ /$$ /$$$$$$ /$$$$$$ |
| /$$__ $$ /$$__ $$| $$__ $$| $$$ | $$| $$_____/| $$ |_ $$_/ /$$__ $$ |
| | $$ \__/| $$ \ $$| $$ \ $$| $$$$| $$| $$ | $$ | $$ | $$ \ $$ |
| | $$ | $$ | $$| $$$$$$$/| $$ $$ $$| $$$$$ | $$ | $$ | $$$$$$$$ |
| | $$ | $$ | $$| $$__ $$| $$ $$$$| $$__/ | $$ | $$ | $$__ $$ |
| | $$ $$| $$ | $$| $$ \ $$| $$\ $$$| $$ | $$ | $$ | $$ | $$ |
| | $$$$$$/| $$$$$$/| $$ | $$| $$ \ $$| $$$$$$$$| $$$$$$$$ /$$$$$$| $$ | $$ |
| \______/ \______/ |__/ |__/|__/ \__/|________/|________/|______/|__/ |__/ |
| |
| |
| *#================#@@#***%%##%*+====+=====%++========++=======+*&&@@@*%%#@@@@#&%+==========&%*%+===+=====+==*@@@@@@@@@#==========*@&&%==+========% |
| @======**%%%%*====@%%#&@@@&&@#*++*%%##%%+%#%*#%%%###%&#%#%#%#%*@&&@@@%*#&@@#%#%@*##%####%#*+@##*++****%=+%#%=#@@@@@@@@@*==+%%%**=+@@@@%===++++***% |
| %=====*%%%%%+==%#&&@@@@@&&@@*+*****%#%%+#%%=%%%%#*#%##%##%%###%=*#+**#%%++======###%####%%%**#&#%%*+*++=*==**=#@@@@@@@@@%==+%%%%==&@@@@@+=+##%+++* |
| *====+%%%%%+=*%&&@@@@@@&@@%++%*%%%%*%**#*++%%*%*%#%%&%##%#####++==============***#%%%%##%%*%%*%#@&&##%*+===++==%&@@@@@@@@#==*%%%+=+@@@@@@+==+*%#%% |
| %====*%***%=*#@@@@@@@@#@#%**%%%%%*%*%##%=+%*%%*%#%%&%%#%%%%#%+==========++++===**##%%%##%%%*****&&@@@@&#%*+==+++=%#@@@@@@@%==*%%%==+@@@@@@+==*+++# |
| %===+%%%%%%=%@@@@@@@@&@@##*#%%**%%%#&%%=++*%+*%%*%&*%#%%#%#%====+++++++++++*++==++#%%%%###%*++***%#&&&&@@@&#%*%%#*==@@@@@@@*=+**%%==%@@@@@&==**+** |
| %+==*%%%%%%+%@@@@@@@&@&&&#*##%%%%%&#%%++*%*+*%*+*#+*%#%%%#%===++*************++==+*#%%%*%#%*==+#%%%%%@&&@@@@@@&&##%=&@@@@@@@==%%%%%==&@@@@%====**% |
| %%+%%%%%%%*======*#@#@#@&#*###%%#&##%**#*++***+*@=+%%%*##*==++++**************++===%#%****%%+++*#&#%#%%#@&&&&@@@@@##@@@@@@@@%=+%%%%*=%@@@&=======# |
| %**%%%%+===========%&@#@&#*##%%&%%#%*#%+++***+%@==*%**#%===+++++++++*******++++++===+#%++++*%*****#&&#&%**%@@@@@@@#&@@@@@@@@@==%%%%%==&@@&====+==+ |
| %=#*%*==============&@#@&#*%%#&%%+%&#*=+*+***#@==+=+%##+======++++++++++++++++++++++*+#*++*+%*+*%#**#&&%%%*++&#&&&%&@@@@@@@@@%=%%%%%*=%@@@=====+=% |
| %=&++=====+===++====%@#@@#%#&#%*%@%*+=+++*+%&&+=+==##%*@&&&*++=======+++++========***%###*+*%**++%#%*%#@&%%%#**%@&#@@@@@@@@@@#=*%%%%%=*@@@&======% |
| %+*====*%%%*==++====%@#@&##&#%+@#*+*++=++%#&&%=+=+++=======*%%***+##=+++++=*#@+#*%*+=++*%&*+=++=+%%%%%*%#@&#%*#%**&@@@@@@@@@@@==#####=+@@@@#=====# |
| %+====*%%##+==++===+#@&@@%%#*%@%+%%*++++%#@&%**=+%#@@&#%%%%+*+****===%+**%+===%*===+&#*@++**===+**%%#%%%*%#&&#%%#*+%@@@@@@@@@@*=#####==@@@@@+====+ |
| %====*###%%==+++===#&&&@*%#%@&%*%%%*++%##&&#%+*#@%#==+*==#===@+=+++++++++++++*=&&@@*====#==#@@*==+%%#%##%**%&@&#%#***#@@@@@@@@*=#####+=&@@@@&===== |
| ======%**%===+====*@@@**%%#&&%%%%%%**###&@%#%%@**=========+%=#=##=++++++++++*#=====%#%*@=====*&@&+++**%###%*%#&&#%#%###@@@@@@@*=#####*=&@@@@@*==== |
| ====@=***===+====+&&%**%*#@@#*%##%*%###@@%%%*@=#&*=+*=+**=========++++===+++=============+**#*=*@@&*****%###%*%&@####&##@@@@@@%=#####*=&@@@@@&==== |
| ====*======++===*&@*%%%+@&@#%%#%**%%#&@@*#%*@@*+%#==*=====++++++++++++=*=+++++++++++++++++=%#%=&@@@@@#****%##%*%&@#%#@@##@@@@@%=#####*=&@@@@@@===+ |
| +=#+======++===*&@**#%*#&@&%*#%%%#%@@@&%#%%@@#@#=*#==**+++++++++****++=%=+++***++++++++++=+&**%#@@@@@@@@%***%%#*%&&##@@@%@@@@@%=#####+=@@@@@@@*==+ |
| *@======++====**@+%%#*%&@@&%%%%%%#@@##%#&&@&@&&@&*+===++=+++++*****+++===+++*********+++==*+%#@&&@&#@@&&@@%**+*#%%&&@@@&&@@@@@%=#####=*@@@@@@@*==+ |
| @============%%*%*##%*@@@@#%%%#%&@&&&&###@@&@#@@@%#&=========+*****+++=@=+++********++===+%%@&@@&@#&#@@&@&@@##*%%%#&&@@&#@@@@@*=#####=*@@@@@@@+==+ |
| @==========+%%*&%###%%&@@&##%##@@#@@&#&&#@#&@#&&@@&&%+=====+=++****+++=&=+++++****++====#&&#@@&@#@#@@&&@@&@&@@#%%%#&@@@#&@@@@&=+####%=%@@@@@@@===+ |
| ==+=======*##+*#*###%%&@@&###@@#@#@@&##&#@##@@###&#%%%#&&#%+=*++++++++===+++++++++=====#@@#@@#@@#@#@@@@&&@#&&&@@#%#@@@&#@@@@@%=%%*+===+&@@@@@#==== |
| ==========%##=&%%%##%%&@@@&#@&&&@@&@@%###@##%@@@@@@@@@@&&#*==*+++==============++====*%&@#@@@&@@#@#@@@@@&@@&#@&&@@&@@&%@@@@@@*=%+========&@@@*=*== |
| ========*%###=&##*%#%%#@@@&@&@@@&&@&&*%##&&&@@@@#####&@@&%%===++*=*#%%*%%%##%*+===+++@&&@@@#&@@&@@&@@@@@@#@@@#@&&&@&&%&@@@@@@+=*==========*@&=+%=+ |
| ========*%###=&#&%*%#*++==+==+#@@@&@@@%%&@@@&##&@@@@@&@&%&=+#====+===============@*=%#@@&&&@@@&#@#@@@@@@&@@#@@&@@@&@@&@@@@@@#====++====++==+*=##=* |
| +======+=+###=&&@&#*===========*&@@&&@@@###&@@@@@@&&@@#&#@*=**@====*+====++====&@=*=&@#&@@@@@@@@@&@@@@@@#@#&&&@#@@&#&@@@@@@@*====+==+**%*====*#%+% |
| =======&#=+*==*##*==============*&@@#@@@@@@@&@&###@@&&@#&@*=%+=@@=====++=====&@%=+==@@&@@#**++=*%@@@@@@#@@&@@&@@&@@@#*@@@@@@+======+%%#%%===+##%*% |
| ======+@@#==+%&@*=====*+*+=======*@#%@@@&@@@#&@@@@&&#&&#&@&=+*===&@*======+#@*===+=%@#=+%=======%&@@@&&@@#&@@&@@#&@@@##@@@@@&======%*##%%====%#%+% |
| ======@@@@&&@@%=====*%%%%%%==+===*+*++@@@@@@@@&%&%%##&&#&&&*=======%@@@@@#*+======%&*====%==+%*=+#&&&@@&&@@@###@@#@@@&%@@@@@@======#++*%*+===*#**% |
| ======@@@@@@@======*#%###%%======*%#%+=++#@@@@&#&#*#@@@#%*=====+======================**==#&=%==&@@@@&&@&&##@@@@@&&@@@%@@@@@@=======#%+======*#+%% |
| =====*@@@@@@======#*####%%%=====+%%**%#%*=%&@@&%#@@#&%%&&%=**====*========#=*==+====%==+&&=====*&##@#%%&#@@@@@@@@&#@@@%@@@@@@====+====@======%*+%% |
| *====&@@@@@=======%*##%++%======@&&#%%**%%++#&@#&#%&&@@#%=*%*====++=====*+==+=+==+==%#&*+*%*==%@@@@@@@@@@@@@@@@@@@#@@&&@@@@@&====++===@=====+#+**% |
| %====@@@@@========*===*##==+===#@@@@@&#%%*#**%&@&&#&#%*==*%%*===++%+==%====*=========*%=++==*&@@@@@@@@@@@@@@@@@@@@#@@#@@@@@@%====+====%=====*=+*=& |
| %+==+@@@@=========@*##+=======&@&&@@@@@@&#%%#&#&@&@#&&%=*%%%*===***+=======%========+=&%==+&@@@@@@@@@@@@@@@@@@@@@@#@&#@@@@@@*===++===@======++*=## |
| %%*=*@@@*========@===========#&&@@&&@@@@@@&###&#@#&@#%=*%%%%*===+**=======+=========*+**#&@@@@@@@@@@@@@@@@@@@@@@@&&&%@@@@@@@+===++===&======*%+### |
| %%*=%@@#==+=====&=====+====+%&@@&&@@#&@@@@@@&&@#@#@&@%=+%%%%*====*++======*=========*%=%@@@@@@@@@@@@@@@@@@@@@@@@@#&&@@@@@@@&========@======**+=+%# |
| %%*=%@@+=***===#+=========*@&#@@@&#@@&&@@@@@@@@#@#@@#@*=*%%%*====++====+==*=========*%==&@@@@@@@@@@@@@@@@@@@@@@@&&&@@@@@@@@%=*+=====*=====*+====+# |
| %%*=%@%=+%%%*=+@=+=======%@@@%&@@@@&&@@#@@@@@@&@@@&@#@*+=%%%*=====#===+==%====+=====***=#@@@@@@@@@@@@@@@@@@@@@@&#@@@@@@@@@@*=**====@=====#=======# |
| %%*=%@==%%%%==@=+%%%%%==%@@@@@%&@@@@@#@@#@@@@@&@@#@@&@#*=%%%*======+================%**=*@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@%==%%*==@=====+========% |
| %===%&=======&+========*@@@@@@&%%%%%%##@@%%%%%%@##%%@##+===========+===================+=&@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@+=======*=============== |
| @===&@======+@========*@@@@@@@@&&&&&&&&@@&&&&&&@&&&&@&#*=================================#@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@@*======#===============& |
| |
| /$$$$$$ /$$$$$$ /$$$$$$$ /$$ /$$ /$$$$$$$$ /$$ /$$$$$$ /$$$$$$ |
| /$$__ $$ /$$__ $$| $$__ $$| $$$ | $$| $$_____/| $$ |_ $$_/ /$$__ $$ |
| | $$ \__/| $$ \ $$| $$ \ $$| $$$$| $$| $$ | $$ | $$ | $$ \ $$ |
| | $$ | $$ | $$| $$$$$$$/| $$ $$ $$| $$$$$ | $$ | $$ | $$$$$$$$ |
| | $$ | $$ | $$| $$__ $$| $$ $$$$| $$__/ | $$ | $$ | $$__ $$ |
| | $$ $$| $$ | $$| $$ \ $$| $$\ $$$| $$ | $$ | $$ | $$ | $$ |
| | $$$$$$/| $$$$$$/| $$ | $$| $$ \ $$| $$$$$$$$| $$$$$$$$ /$$$$$$| $$ | $$ |
| \______/ \______/ |__/ |__/|__/ \__/|________/|________/|______/|__/ |__/ |
| |
| """ |
|
|
| print(CORNELIA) |
|
|
| MIN_CLUSTER_SIZE = 3 |
|
|
| |
| _hf_token = os.environ.get("HF_TOKEN") |
| if _hf_token: |
| login(_hf_token) |
|
|
| |
| |
| |
| CONTRACTIONS = { |
| r"don't": "do not", |
| r"doesn't": "does not", |
| r"didn't": "did not", |
| r"can't": "cannot", |
| r"couldn't": "could not", |
| r"won't": "will not", |
| r"wouldn't": "would not", |
| r"isn't": "is not", |
| r"aren't": "are not", |
| r"wasn't": "was not", |
| r"weren't": "were not", |
| r"haven't": "have not", |
| r"hasn't": "has not", |
| r"hadn't": "had not", |
| r"it's": "it is", |
| r"that's": "that is", |
| r"there's": "there is", |
| r"i'm": "i am", |
| r"i've": "i have", |
| r"i'd": "i would", |
| r"i'll": "i will", |
| r"you're": "you are", |
| r"they're": "they are", |
| r"we're": "we are", |
| r"he's": "he is", |
| r"she's": "she is", |
| r"let's": "let us", |
| } |
|
|
| CONTRACTION_PATTERNS = [ |
| (re.compile(pattern, re.IGNORECASE), replacement) |
| for pattern, replacement in CONTRACTIONS.items() |
| ] |
|
|
| nltk.download("stopwords", quiet=True) |
|
|
| KEYWORD_BLOCKLIST = set(stopwords.words("english")) | { |
| "wi", "fi", "app", "really", "like", "love", |
| "work", "works", "great", "good", "nice", "use", |
| "needs", "need", "feels", "feel", "makes", "make", |
| "just", "also", "even", "much", "many", "very", |
| } |
|
|
| AGE_BUCKETS = [ |
| ("18β24", 18, 24), |
| ("25β34", 25, 34), |
| ("35β44", 35, 44), |
| ("45β54", 45, 54), |
| ("55β64", 55, 64), |
| ("65β74", 65, 74), |
| ("75β84", 75, 84), |
| ("85β94", 85, 94), |
| ("95β98", 95, 98), |
| ("99+", 99, 999), |
| ] |
|
|
| |
| |
| |
| |
| |
| SENTIMENT_BATCH_SIZE = 8 |
|
|
| |
| |
| MAX_TOKEN_LENGTH = 128 |
|
|
| |
| |
| |
| def expand_contractions(text: str) -> str: |
| for pattern, replacement in CONTRACTION_PATTERNS: |
| text = pattern.sub(replacement, text) |
| return text |
|
|
|
|
| def clean_keywords(kw_list: list[str]) -> list[str]: |
| return [ |
| kw for kw in kw_list |
| if len(kw) >= 4 and kw.lower() not in KEYWORD_BLOCKLIST |
| ] |
|
|
|
|
| def run_sentiment_batch(texts: list[str]) -> list[str]: |
| """ |
| Run sentiment classifier over a list of texts and return |
| a list of lowercase label strings in the same order. |
| Single batched call β never call this inside a loop. |
| """ |
| if not texts: |
| return [] |
| results = sentiment_classifier( |
| texts, |
| truncation=True, |
| max_length=MAX_TOKEN_LENGTH, |
| batch_size=SENTIMENT_BATCH_SIZE, |
| ) |
| return [r["label"].lower() for r in results] |
|
|
|
|
| def labels_to_breakdown(labels: list[str]) -> dict: |
| """Convert a list of sentiment label strings to a percentage breakdown dict.""" |
| counts = {"positive": 0, "neutral": 0, "negative": 0} |
| for label in labels: |
| if label in counts: |
| counts[label] += 1 |
| total = len(labels) or 1 |
| return { |
| "positive": round(counts["positive"] / total * 100, 1), |
| "neutral": round(counts["neutral"] / total * 100, 1), |
| "negative": round(counts["negative"] / total * 100, 1), |
| } |
|
|
|
|
| def get_sentiment_breakdown(texts: list[str]) -> dict: |
| """Convenience wrapper β batch sentiment β percentage breakdown.""" |
| return labels_to_breakdown(run_sentiment_batch(texts)) |
|
|
|
|
| def cluster_and_label(texts: list[str], embeddings: np.ndarray, reduced: np.ndarray): |
| """ |
| Shared clustering + keyword-extraction logic. |
| Returns (labels_array, cluster_keywords_dict). |
| """ |
| try: |
| import hdbscan |
| clusterer = hdbscan.HDBSCAN( |
| min_cluster_size=2, |
| min_samples=1, |
| metric="euclidean", |
| cluster_selection_method="leaf" |
| ) |
| labels = clusterer.fit_predict(reduced) |
| except ImportError: |
| from sklearn.preprocessing import normalize |
| labels = KMeans(n_clusters=7, random_state=42, n_init=10).fit_predict( |
| normalize(embeddings) |
| ) |
|
|
| cluster_keywords = {} |
| for label in sorted(set(labels)): |
| if label == -1: |
| continue |
| indices = [j for j, l in enumerate(labels) if l == label] |
| cluster_docs = [texts[j] for j in indices] |
| expanded = " ".join(expand_contractions(doc) for doc in cluster_docs) |
| keywords = kw_model.extract_keywords( |
| expanded, |
| keyphrase_ngram_range=(1, 1), |
| stop_words="english", |
| top_n=10, |
| use_mmr=True, |
| diversity=0.6, |
| ) |
| cluster_keywords[label] = clean_keywords([kw for kw, _ in keywords])[:6] |
|
|
| return labels, cluster_keywords |
|
|
|
|
| def reduce_embeddings(embeddings: np.ndarray, n_components: int = 5): |
| """ |
| UMAP reduction with PCA fallback. |
| On CPU, PCA is often faster and accurate enough for clustering. |
| UMAP is used only when available and the dataset is large enough |
| to benefit from non-linear reduction. |
| """ |
| try: |
| import umap |
| |
| if embeddings.shape[0] < 50: |
| raise ImportError("Skipping UMAP for small dataset β using PCA") |
| reducer = umap.UMAP( |
| n_components=n_components, |
| random_state=42, |
| min_dist=0.0, |
| metric="cosine", |
| ) |
| return reducer.fit_transform(embeddings) |
| except (ImportError, Exception): |
| from sklearn.decomposition import PCA |
| n = min(n_components, embeddings.shape[0] - 1) |
| return PCA(n_components=n, random_state=42).fit_transform(embeddings) |
|
|
|
|
| |
| |
| |
| sentiment_classifier = pipeline( |
| "sentiment-analysis", |
| model="cardiffnlp/twitter-roberta-base-sentiment-latest", |
| truncation=True, |
| max_length=MAX_TOKEN_LENGTH, |
| ) |
|
|
| emotion_classifier = pipeline( |
| "text-classification", |
| model="SamLowe/roberta-base-go_emotions", |
| top_k=None, |
| truncation=True, |
| max_length=MAX_TOKEN_LENGTH, |
| ) |
|
|
| embedder = SentenceTransformer("all-MiniLM-L6-v2") |
| kw_model = KeyBERT(model=embedder) |
|
|
| |
| |
| |
| def find_optimal_k(embeddings, k_min=3, k_max=5): |
| inertias = [] |
| k_range = range(k_min, min(k_max + 1, len(embeddings))) |
| for k in k_range: |
| km = KMeans(n_clusters=k, random_state=42, n_init=10) |
| km.fit(embeddings) |
| inertias.append(km.inertia_) |
| drops = [inertias[i] - inertias[i + 1] for i in range(len(inertias) - 1)] |
| return list(k_range)[np.argmax(drops) + 1] |
|
|
|
|
| def merge_small_clusters(cluster_ids, k, embeddings, kmeans: KMeans): |
| counts = defaultdict(int) |
| for c in cluster_ids: |
| counts[c] += 1 |
|
|
| small_clusters = {c for c, count in counts.items() if count < MIN_CLUSTER_SIZE} |
| if not small_clusters: |
| return cluster_ids |
|
|
| large_cluster_ids = [c for c in range(k) if c not in small_clusters] |
| large_centroids = kmeans.cluster_centers_[large_cluster_ids] |
|
|
| new_ids = [] |
| for i, c in enumerate(cluster_ids): |
| if c in small_clusters: |
| dists = np.linalg.norm(large_centroids - embeddings[i], axis=1) |
| new_ids.append(large_cluster_ids[np.argmin(dists)]) |
| else: |
| new_ids.append(c) |
|
|
| return new_ids |
|
|
|
|
| |
| |
| |
| app = Flask(__name__) |
| CORS(app) |
|
|
| @app.route('/') |
| def index(): |
| return """ |
| <html> |
| <head> |
| <title>CORNELIA Backend</title> |
| <style> |
| body { font-family: monospace; background: #0d0d0d; color: #e0e0e0; max-width: 860px; margin: 60px auto; padding: 0 20px; } |
| h1 { color: #a78bfa; font-size: 2em; } |
| h2 { color: #818cf8; margin-top: 40px; } |
| .model { background: #1a1a2e; border-left: 4px solid #a78bfa; padding: 16px 20px; margin: 16px 0; border-radius: 4px; } |
| .model h3 { margin: 0 0 8px 0; color: #c4b5fd; } |
| .model p { margin: 0; color: #9ca3af; line-height: 1.6; } |
| .endpoint { color: #34d399; font-size: 0.85em; margin-top: 6px; } |
| .tag { display: inline-block; background: #312e81; color: #a5b4fc; padding: 2px 8px; border-radius: 12px; font-size: 0.75em; margin-right: 4px; } |
| hr { border: none; border-top: 1px solid #1f2937; margin: 40px 0; } |
| </style> |
| </head> |
| <body> |
| <h1>β CORNELIA β AI Analytics Backend</h1> |
| <p>This is the inference server powering CORNELIA's comment analysis pipeline. It exposes a set of NLP endpoints consumed by the Flutter app to generate real-time sentiment, emotion, topic, and demographic insights from user comments.</p> |
| |
| <hr> |
| <h2>Models</h2> |
| |
| <div class="model"> |
| <h3>cardiffnlp/twitter-roberta-base-sentiment-latest</h3> |
| <span class="tag">Sentiment</span> |
| <p>A RoBERTa model fine-tuned on ~124M tweets for 3-class sentiment classification: positive, neutral, and negative. Used to produce sentiment distributions, sentiment over time, country-level gender breakdowns, and negative outlier scoring.</p> |
| <div class="endpoint">POST /roberta-base-sentiment Β· POST /roberta-base-sentiment-SOT Β· POST /roberta-base-sentimentCO Β· POST /roberta-base-sentiment-SCORES</div> |
| </div> |
| |
| <div class="model"> |
| <h3>SamLowe/roberta-base-go_emotions</h3> |
| <span class="tag">Emotion</span> |
| <p>A RoBERTa model fine-tuned on Google's GoEmotions dataset, capable of classifying text into 28 fine-grained emotion categories. CORNELIA uses the top 5 non-neutral emotions weighted by score to produce the emotion distribution chart.</p> |
| <div class="endpoint">POST /roberta-base-go</div> |
| </div> |
| |
| <div class="model"> |
| <h3>all-MiniLM-L6-v2</h3> |
| <span class="tag">Embeddings</span><span class="tag">Clustering</span><span class="tag">Keywords</span> |
| <p>A lightweight sentence transformer that maps text to dense 384-dimensional vectors. Used as the backbone for topic clustering (HDBSCAN/KMeans), intertopic distance mapping (UMAP/PCA), keyword co-occurrence graph construction, emerging trend detection over time, and age-group aspect sentiment correlation.</p> |
| <div class="endpoint">POST /all-MiniLM-L6-v2 Β· POST /all-MiniLM-L6-v2-ETO Β· POST /all-MiniLM-L6-v2-G Β· POST /all-MiniLM-L6-v2-COR</div> |
| </div> |
| |
| <div class="model"> |
| <h3>KeyBERT</h3> |
| <span class="tag">Keyword Extraction</span> |
| <p>Built on top of all-MiniLM-L6-v2, KeyBERT extracts the most semantically representative keywords from each topic cluster using MMR (Maximal Marginal Relevance) to balance relevance and diversity. Keywords label clusters, drive the network graph, and surface emerging issues.</p> |
| <div class="endpoint">Used internally across all clustering endpoints</div> |
| </div> |
| |
| <hr> |
| <h2>Status</h2> |
| <p>β
All models loaded and ready.</p> |
| <div class="endpoint">GET /health β returns {"status": "ok"}</div> |
| </body> |
| </html> |
| """, 200 |
|
|
|
|
| |
| @app.route("/health", methods=["GET"]) |
| def health(): |
| return jsonify({"status": "ok"}) |
|
|
|
|
| @app.route("/roberta-base-sentiment", methods=["POST"]) |
| def roberta_base_sentiment(): |
| data = request.json |
| comments = data.get("data", []) |
|
|
| if not comments: |
| return jsonify({"status": "error", "message": "No comments provided"}), 400 |
|
|
| labels = run_sentiment_batch(comments) |
| analysis = labels_to_breakdown(labels) |
|
|
| |
| analysis = {k: round(v, 2) for k, v in analysis.items()} |
| return jsonify({"status": "success", "results": analysis}) |
|
|
|
|
| @app.route("/roberta-base-go", methods=["POST"]) |
| def roberta_base_go(): |
| data = request.get_json() |
| comments = data.get("data", []) |
|
|
| if not comments: |
| return jsonify({"status": "error", "message": "No comments provided"}), 400 |
|
|
| results = emotion_classifier( |
| comments, |
| truncation=True, |
| max_length=MAX_TOKEN_LENGTH, |
| batch_size=SENTIMENT_BATCH_SIZE, |
| ) |
| if isinstance(results[0], dict): |
| results = [results] |
|
|
| totals = {} |
| for comment_scores in results: |
| for item in comment_scores: |
| label = item["label"] |
| if label == "neutral": |
| continue |
| totals[label] = totals.get(label, 0.0) + item["score"] |
|
|
| sorted_emotions = sorted(totals.items(), key=lambda x: x[1], reverse=True)[:5] |
| total_score = sum(score for _, score in sorted_emotions) |
| results_json = [ |
| {"label": label.capitalize(), "value": round(score / total_score * 100, 2)} |
| for label, score in sorted_emotions |
| ] |
|
|
| return jsonify({"status": "success", "results": results_json}) |
|
|
|
|
| @app.route("/all-MiniLM-L6-v2", methods=["POST"]) |
| def all_MiniLM_L6_v2(): |
| data = request.get_json() |
| raw_docs = data.get("data", []) |
|
|
| if not raw_docs: |
| return jsonify({"status": "error", "message": "No comments provided"}), 400 |
|
|
| embeddings = embedder.encode(raw_docs, show_progress_bar=False, batch_size=64) |
| reduced = reduce_embeddings(embeddings, n_components=5) |
|
|
| |
| try: |
| import umap |
| if len(raw_docs) >= 50: |
| reducer_2d = umap.UMAP( |
| n_components=2, random_state=42, min_dist=0.3, metric="cosine" |
| ) |
| coords_2d = reducer_2d.fit_transform(embeddings) |
| else: |
| raise ImportError("Small dataset β PCA fallback") |
| except (ImportError, Exception): |
| from sklearn.decomposition import PCA |
| coords_2d = PCA(n_components=2, random_state=42).fit_transform(embeddings) |
|
|
| labels, cluster_keywords = cluster_and_label(raw_docs, embeddings, reduced) |
|
|
| x_vals = coords_2d[:, 0] |
| y_vals = coords_2d[:, 1] |
| x_min, x_max = x_vals.min(), x_vals.max() |
| y_min, y_max = y_vals.min(), y_vals.max() |
|
|
| def normalize_coord(val, vmin, vmax): |
| if vmax == vmin: |
| return 0.5 |
| return 0.1 + (val - vmin) / (vmax - vmin) * 0.8 |
|
|
| output = {} |
| for i, label in enumerate(sorted(set(labels)), start=1): |
| if label == -1: |
| continue |
|
|
| indices = [j for j, l in enumerate(labels) if l == label] |
| cluster_docs = [raw_docs[j] for j in indices] |
| kw_labels = cluster_keywords.get(label, []) |
| theme_name = kw_labels[0].title() if kw_labels else f"Theme {i}" |
|
|
| cluster_x = [normalize_coord(coords_2d[j, 0], x_min, x_max) for j in indices] |
| cluster_y = [normalize_coord(coords_2d[j, 1], y_min, y_max) for j in indices] |
| centroid_x = sum(cluster_x) / len(cluster_x) |
| centroid_y = sum(cluster_y) / len(cluster_y) |
|
|
| output[theme_name.lower()] = { |
| "comments": len(cluster_docs), |
| "keywords": kw_labels, |
| "x": round(float(centroid_x), 4), |
| "y": round(float(centroid_y), 4), |
| } |
|
|
| return jsonify({"status": "success", "results": output}) |
|
|
|
|
| @app.route("/roberta-base-sentiment-SOT", methods=["POST"]) |
| def roberta_base_sentiment_SOT(): |
| body = request.get_json() |
| comments = body.get("data", []) |
|
|
| if not comments: |
| return jsonify({"status": "error", "message": "No comments provided"}), 400 |
|
|
| texts = [c["text"] for c in comments] |
| dates = [c["date"] for c in comments] |
|
|
| labels_out = run_sentiment_batch(texts) |
|
|
| labeled = [ |
| {"date": date, "label": label} |
| for date, label in zip(dates, labels_out) |
| ] |
|
|
| return jsonify({"status": "success", "results": labeled}) |
|
|
|
|
| @app.route("/all-MiniLM-L6-v2-ETO", methods=["POST"]) |
| def all_MiniLM_L6_v2_ETO(): |
| body = request.get_json() |
| comments = body.get("data", []) |
|
|
| if not comments: |
| return jsonify({"status": "error", "message": "No comments provided"}), 400 |
|
|
| texts = [c["text"] for c in comments] |
| dates = [c["date"] for c in comments] |
|
|
| embeddings = embedder.encode(texts, show_progress_bar=False, batch_size=64) |
|
|
| k = find_optimal_k(embeddings) |
| kmeans = KMeans(n_clusters=k, random_state=42, n_init=10) |
| cluster_ids = kmeans.fit_predict(embeddings).tolist() |
| cluster_ids = merge_small_clusters(cluster_ids, k, embeddings, kmeans) |
|
|
| unique_ids = sorted(set(cluster_ids)) |
| id_remap = {old: new for new, old in enumerate(unique_ids)} |
| cluster_ids = [id_remap[c] for c in cluster_ids] |
| k = len(unique_ids) |
|
|
| cluster_labels = {} |
| for cluster_id in range(k): |
| cluster_comments = [texts[i] for i, l in enumerate(cluster_ids) if l == cluster_id] |
| if len(cluster_comments) < 2: |
| cluster_labels[cluster_id] = cluster_comments[0][:30] |
| continue |
| cleaned = [re.sub(r'[^\w\s]|[\d]', ' ', c) for c in cluster_comments] |
| combined_text = ' '.join(expand_contractions(c) for c in cleaned) |
| keywords = kw_model.extract_keywords( |
| combined_text, |
| keyphrase_ngram_range=(1, 1), |
| stop_words='english', |
| use_mmr=True, |
| diversity=0.5, |
| top_n=1, |
| ) |
| cluster_labels[cluster_id] = keywords[0][0] if keywords else f"theme_{cluster_id}" |
|
|
| counts = defaultdict(lambda: defaultdict(int)) |
| for i, date_str in enumerate(dates): |
| keyword = cluster_labels[cluster_ids[i]] |
| counts[keyword][date_str] += 1 |
|
|
| trend_data = [] |
| for keyword, date_counts in counts.items(): |
| for date, count in sorted(date_counts.items()): |
| trend_data.append({ |
| "time": date, |
| "count": count, |
| "keyword": keyword, |
| }) |
|
|
| return jsonify({"status": "success", "results": trend_data}) |
|
|
|
|
| @app.route("/roberta-base-sentimentCO", methods=["POST"]) |
| def roberta_base_sentiment_CO(): |
| data = request.json |
| comments = data.get("data", []) |
|
|
| if not comments: |
| return jsonify({"status": "error", "message": "No comments provided"}), 400 |
|
|
| |
| all_texts = [item["text"] for item in comments] |
| sent_labels = run_sentiment_batch(all_texts) |
|
|
| |
| for item, label in zip(comments, sent_labels): |
| item["_sentiment"] = label |
|
|
| |
| country_groups = defaultdict(list) |
| for item in comments: |
| country_groups[item["country"]].append(item) |
|
|
| def sentiment_breakdown_from_items(items_subset): |
| return labels_to_breakdown([i["_sentiment"] for i in items_subset]) |
|
|
| output = [] |
| for country, items in country_groups.items(): |
| male_items = [i for i in items if i["gender"].lower() == "male"] |
| female_items = [i for i in items if i["gender"].lower() == "female"] |
|
|
| output.append({ |
| "country": country, |
| "commentsNo": len(items), |
| "maleSentiment": sentiment_breakdown_from_items(male_items), |
| "femaleSentiment": sentiment_breakdown_from_items(female_items), |
| }) |
|
|
| return jsonify({"status": "success", "results": output}) |
|
|
|
|
| @app.route("/all-MiniLM-L6-v2-G", methods=["POST"]) |
| def all_MiniLM_L6_v2_G(): |
| data = request.get_json() |
| raw_docs = data.get("data", []) |
|
|
| if not raw_docs: |
| return jsonify({"status": "error", "message": "No comments provided"}), 400 |
|
|
| embeddings = embedder.encode(raw_docs, show_progress_bar=False, batch_size=64) |
| reduced = reduce_embeddings(embeddings, n_components=5) |
| labels, cluster_keywords = cluster_and_label(raw_docs, embeddings, reduced) |
|
|
| all_keywords_per_doc = [ |
| cluster_keywords.get(labels[i], []) |
| for i in range(len(raw_docs)) |
| ] |
|
|
| node_counts = defaultdict(int) |
| edge_counts = defaultdict(int) |
|
|
| for kws in all_keywords_per_doc: |
| unique_kws = list(set(kws)) |
| for kw in unique_kws: |
| node_counts[kw] += 1 |
| for i in range(len(unique_kws)): |
| for j in range(i + 1, len(unique_kws)): |
| pair = tuple(sorted([unique_kws[i], unique_kws[j]])) |
| edge_counts[pair] += 1 |
|
|
| top_n_nodes = 16 |
| top_keywords = sorted(node_counts.items(), key=lambda x: x[1], reverse=True)[:top_n_nodes] |
| top_kw_set = {kw for kw, _ in top_keywords} |
|
|
| min_cooccurrence = 1 |
| edges = [] |
| max_count = max(edge_counts.values()) if edge_counts else 1 |
|
|
| for (kw1, kw2), count in edge_counts.items(): |
| if kw1 not in top_kw_set or kw2 not in top_kw_set: |
| continue |
| if count < min_cooccurrence: |
| continue |
| edges.append({ |
| "source": kw1, |
| "target": kw2, |
| "relation": "co-occurs with", |
| "weight": round(count / max_count, 4), |
| }) |
|
|
| max_node_count = top_keywords[0][1] if top_keywords else 1 |
| connected_kws = {e["source"] for e in edges} | {e["target"] for e in edges} |
| nodes = [ |
| { |
| "id": kw, |
| "label": kw, |
| "frequency": count, |
| "weight": round(count / max_node_count, 4), |
| } |
| for kw, count in top_keywords |
| if kw in connected_kws |
| ] |
|
|
| return jsonify({"status": "success", "results": {"nodes": nodes, "edges": edges}}) |
|
|
|
|
| @app.route("/all-MiniLM-L6-v2-COR", methods=["POST"]) |
| def all_MiniLM_L6_v2_COR(): |
| body = request.get_json() |
| comments = body.get("data", []) |
|
|
| if not comments: |
| return jsonify({"status": "error", "message": "No comments provided"}), 400 |
|
|
| def safe_int(val): |
| try: |
| return int(val) |
| except (ValueError, TypeError): |
| return None |
|
|
| comments = [{**c, "age": safe_int(c.get("age"))} for c in comments] |
|
|
| all_texts = [c["text"] for c in comments] |
|
|
| |
| all_embeddings = embedder.encode(all_texts, show_progress_bar=False, batch_size=64) |
|
|
| |
| |
| |
| |
| all_sent_labels = run_sentiment_batch(all_texts) |
|
|
| |
| buckets = {label: [] for label, _, _ in AGE_BUCKETS} |
| bucket_emb = {label: [] for label, _, _ in AGE_BUCKETS} |
| bucket_sent = {label: [] for label, _, _ in AGE_BUCKETS} |
|
|
| for idx, c in enumerate(comments): |
| age = c["age"] |
| if age is None: |
| continue |
| for label, lo, hi in AGE_BUCKETS: |
| if lo <= age <= hi: |
| buckets[label].append(c["text"]) |
| bucket_emb[label].append(all_embeddings[idx]) |
| bucket_sent[label].append(all_sent_labels[idx]) |
| break |
|
|
| results = [] |
|
|
| for label, _, _ in AGE_BUCKETS: |
| texts = buckets[label] |
| if not texts: |
| continue |
|
|
| embeddings = np.array(bucket_emb[label]) |
| s_labels = bucket_sent[label] |
|
|
| |
| counts = {"positive": 0, "neutral": 0, "negative": 0} |
| for sl in s_labels: |
| counts[sl] += 1 |
| total = len(texts) |
| sentiment = {k: round(v / total * 100, 1) for k, v in counts.items()} |
| dominant = max(counts, key=counts.get) |
|
|
| |
| n_comp = min(5, len(texts) - 1) |
| try: |
| import umap |
| if len(texts) >= 50: |
| reducer = umap.UMAP( |
| n_components=n_comp, |
| random_state=42, |
| min_dist=0.0, |
| metric="cosine", |
| ) |
| reduced = reducer.fit_transform(embeddings) |
| else: |
| raise ImportError("Small bucket β PCA fallback") |
| except (ImportError, Exception): |
| from sklearn.decomposition import PCA |
| n = min(n_comp, len(texts) - 1) |
| reduced = PCA(n_components=n, random_state=42).fit_transform(embeddings) |
|
|
| try: |
| import hdbscan |
| clusterer = hdbscan.HDBSCAN( |
| min_cluster_size=2, |
| min_samples=1, |
| metric="euclidean", |
| cluster_selection_method="leaf", |
| ) |
| cluster_labels = clusterer.fit_predict(reduced) |
| except (ImportError, Exception): |
| from sklearn.preprocessing import normalize |
| k = min(3, len(texts)) |
| cluster_labels = KMeans( |
| n_clusters=k, random_state=42, n_init=10 |
| ).fit_predict(normalize(embeddings)) |
|
|
| |
| |
| text_to_sent = {texts[i]: s_labels[i] for i in range(len(texts))} |
|
|
| all_aspects = [] |
| for cluster_id in sorted(set(cluster_labels)): |
| if cluster_id == -1: |
| continue |
|
|
| indices = [j for j, l in enumerate(cluster_labels) if l == cluster_id] |
| cluster_texts = [texts[j] for j in indices] |
| expanded = " ".join(expand_contractions(t) for t in cluster_texts) |
|
|
| keywords = kw_model.extract_keywords( |
| expanded, |
| keyphrase_ngram_range=(1, 2), |
| stop_words="english", |
| top_n=10, |
| use_mmr=True, |
| diversity=0.6, |
| ) |
| cleaned = clean_keywords([kw for kw, _ in keywords])[:3] |
|
|
| for aspect in cleaned: |
| related = [t for t in cluster_texts if aspect.lower() in t.lower()] |
| if not related: |
| related = cluster_texts |
|
|
| |
| a_counts = {"positive": 0, "neutral": 0, "negative": 0} |
| for t in related: |
| lbl = text_to_sent.get(t, "neutral") |
| if lbl in a_counts: |
| a_counts[lbl] += 1 |
| a_total = len(related) |
|
|
| all_aspects.append({ |
| "aspect": aspect, |
| "sentiment": max(a_counts, key=a_counts.get), |
| "breakdown": { |
| k: round(v / a_total * 100, 1) |
| for k, v in a_counts.items() |
| }, |
| "mentionCount": len(related), |
| }) |
|
|
| seen = {} |
| for a in all_aspects: |
| key = a["aspect"].lower() |
| if key not in seen or a["mentionCount"] > seen[key]["mentionCount"]: |
| seen[key] = a |
| deduped_aspects = list(seen.values())[:5] |
|
|
| results.append({ |
| "ageGroup": label, |
| "commentCount": total, |
| "dominantSentiment": dominant, |
| "sentimentBreakdown": sentiment, |
| "aspects": deduped_aspects, |
| }) |
|
|
| return jsonify({"status": "success", "results": results}) |
|
|
|
|
| @app.route("/roberta-base-sentiment-SCORES", methods=["POST"]) |
| def roberta_base_sentiment_scores(): |
| data = request.json |
| comments = data.get("data", []) |
|
|
| if not comments: |
| return jsonify({"status": "error", "message": "No comments provided"}), 400 |
|
|
| |
| results = sentiment_classifier( |
| comments, |
| top_k=None, |
| truncation=True, |
| max_length=MAX_TOKEN_LENGTH, |
| batch_size=SENTIMENT_BATCH_SIZE, |
| ) |
|
|
| output = [] |
| for i, comment_scores in enumerate(results): |
| neg_score = next( |
| (s["score"] for s in comment_scores if s["label"].lower() == "negative"), |
| 0.0, |
| ) |
| output.append({ |
| "id": i, |
| "text": comments[i], |
| "score": round(neg_score, 4), |
| }) |
|
|
| return jsonify({"status": "success", "results": output}) |
|
|
|
|
| if __name__ == "__main__": |
| app.run(host="0.0.0.0", port=5000, debug=False) |