"""Semantic clustering engine. Groups by MEANING, not by shared words. Free. The keyword engine (analyze_local.py) has a real flaw: it matches vocabulary, so "I love the ads" and "I hate the ads" look identical to it, and "the app is sluggish" and "it takes forever to load" look unrelated. This engine converts each piece of feedback into a vector that encodes what it MEANS, then groups vectors that sit near each other. Complaints phrased in completely different words end up together; opposite opinions about the same topic pull apart. Model: potion-base-8M via model2vec. ~60MB, downloaded once, then runs offline on your machine. Free, no account, no API key, no per-run cost. Quality controls live in quality.py: the cluster count is chosen rather than guessed, near-duplicate clusters get merged, and oversized catch-all clusters get flagged instead of quietly presented as findings. """ import numpy as np import quality MODEL_NAME = "minishlab/potion-base-8M" _MODEL_CACHE = {} def embed(texts): """Turn text into vectors. Model is loaded once and reused.""" from model2vec import StaticModel if MODEL_NAME not in _MODEL_CACHE: print(f" loading embedding model ({MODEL_NAME})") print(" first run downloads ~60MB, after that it is offline and instant") _MODEL_CACHE[MODEL_NAME] = StaticModel.from_pretrained(MODEL_NAME) vectors = np.asarray(_MODEL_CACHE[MODEL_NAME].encode(list(texts))) norms = np.linalg.norm(vectors, axis=1, keepdims=True) norms[norms == 0] = 1 return vectors / norms def _describe(cluster_indices, texts, unit, centre, records): """Build a human-readable theme from a group of records.""" distances = np.linalg.norm(unit[cluster_indices] - centre, axis=1) central = cluster_indices[np.argsort(distances)][:3] exemplars = [texts[i] for i in central] # Sentiment split comes from star ratings, not from guessing at wording. # This is what stops "I like the ads" being counted as a complaint. ratings = [ int(records[i]["rating"]) for i in cluster_indices if str(records[i].get("rating", "")).isdigit() ] negative = sum(1 for r in ratings if r <= 3) positive = sum(1 for r in ratings if r >= 4) headline = exemplars[0][:70].rstrip() if len(exemplars[0]) > 70: headline += "…" return { "name": headline, "definition": "Most representative feedback in this group: " + " | ".join(e[:120] for e in exemplars), "exemplars": exemplars, "sentiment": { "negative": negative, "positive": positive, "unrated": len(cluster_indices) - len(ratings), }, } def cluster_semantic(records, n_themes=None, seed=42, merge_threshold=0.88): """Group records by meaning. n_themes=None picks the cluster count automatically. Pass a number to force it. Returns (themes, records). """ from sklearn.cluster import KMeans texts = [r["text"] for r in records] unit = embed(texts) if n_themes is None: n_themes, score = quality.choose_k(unit, seed=seed) print(f" chose {n_themes} clusters automatically (silhouette {score:.3f})") else: n_themes = min(n_themes, len(records)) model = KMeans(n_clusters=n_themes, random_state=seed, n_init=10) labels = model.fit_predict(unit) labels, _, merged = quality.merge_similar( labels, model.cluster_centers_, threshold=merge_threshold ) if merged: print(f" merged {merged} near-duplicate clusters") themes = [] for cluster_id in sorted(set(labels)): members = np.where(labels == cluster_id)[0] centre = unit[members].mean(axis=0) theme = _describe(members, texts, unit, centre, records) theme["_id"] = int(cluster_id) themes.append(theme) for record, label in zip(records, labels): record["theme"] = themes[[t["_id"] for t in themes].index(label)]["name"] return themes, records