"""Cluster quality controls. Three problems showed up in the first semantic run, all of them real and all of them fixable without an API: 1. Ads dominated ~40% of complaints, so KMeans split one theme into three near-identical clusters. 2. One cluster became a catch-all for everything that fit nowhere. 3. Non-English reviews grouped by LANGUAGE rather than by topic. This module handles all three, plus picks the number of clusters instead of making you guess. """ import numpy as np # Function words that appear in almost any English sentence. A review with very # few of them is probably not English. Crude, dependency-free, good enough to # stop Spanish reviews forming their own "theme". ENGLISH_MARKERS = { "the", "and", "is", "it", "to", "a", "of", "in", "you", "i", "for", "not", "this", "that", "with", "but", "on", "have", "are", "be", "my", "was", "so", "can", "if", "just", "when", "all", "they", "we", "there", "no", } def looks_english(text, threshold=0.10): """True if enough of the words are common English function words.""" words = [w for w in text.lower().split() if w.isalpha()] if len(words) < 4: return True # too short to judge; keep it rather than silently drop it hits = sum(1 for w in words if w in ENGLISH_MARKERS) return (hits / len(words)) >= threshold def filter_english(records): kept = [r for r in records if looks_english(r["text"])] return kept, len(records) - len(kept) def choose_k(vectors, k_min=4, k_max=12, seed=42): """Pick the number of clusters that separates the data best. Silhouette score measures how tightly grouped each cluster is versus how far apart the clusters sit. Higher is better. Instead of guessing 8, we try every k in range and keep the winner. """ from sklearn.cluster import KMeans from sklearn.metrics import silhouette_score k_max = min(k_max, len(vectors) - 1) best_k, best_score = k_min, -1.0 for k in range(k_min, k_max + 1): labels = KMeans(n_clusters=k, random_state=seed, n_init=10).fit_predict(vectors) if len(set(labels)) < 2: continue score = silhouette_score(vectors, labels, metric="euclidean") if score > best_score: best_k, best_score = k, score return best_k, best_score def merge_similar(labels, centroids, threshold=0.88): """Merge clusters whose centres point in nearly the same direction. This is what collapses "too many ads", "adds. too many adds" and "ads after every song" back into one theme instead of three. Returns (new_labels, merge_map, n_merged). """ n = len(centroids) norms = np.linalg.norm(centroids, axis=1, keepdims=True) norms[norms == 0] = 1 unit = centroids / norms similarity = unit @ unit.T # Union-find: walk pairs, join anything above the threshold. parent = list(range(n)) def find(x): while parent[x] != x: parent[x] = parent[parent[x]] x = parent[x] return x def union(a, b): ra, rb = find(a), find(b) if ra != rb: parent[max(ra, rb)] = min(ra, rb) for i in range(n): for j in range(i + 1, n): if similarity[i, j] >= threshold: union(i, j) roots = sorted({find(i) for i in range(n)}) remap = {root: new for new, root in enumerate(roots)} merge_map = {i: remap[find(i)] for i in range(n)} new_labels = np.array([merge_map[l] for l in labels]) return new_labels, merge_map, n - len(roots) def flag_catchall(clusters, share_threshold=0.30): """Mark any cluster big enough to be a dumping ground rather than a theme. Not removed, just labelled, so the report can say so out loud instead of quietly presenting noise as a finding. """ for c in clusters: c["is_catchall"] = c["share"] >= (share_threshold * 100) return clusters