"""The decision layer. Scores opportunities, picks one, and argues for it. Everything upstream of this file surfaces and ranks. This file commits. It scores every cluster on five factors, weights them, picks a winner, and writes out the reasoning including the case against its own recommendation. Why a scoring model rather than asking a language model for an opinion: a model that says "I recommend improving onboarding because it seems impactful" gives you nothing to argue with. Here every factor is visible and every weight is tunable in the config. If you disagree with the recommendation, you can point at exactly which number you'd change. That is a better artifact for a PM than a confident paragraph. Factors: reach how many people raise it corroboration how many independent channels raise it severity how angry they are (star ratings, weighted by engagement) okr_fit how closely it relates to the active objective confidence how cleanly it is a complaint rather than mixed sentiment """ import numpy as np DEFAULT_WEIGHTS = { "reach": 0.25, "corroboration": 0.25, "severity": 0.20, "okr_fit": 0.20, "confidence": 0.10, } def _normalize(values): """Scale a list of numbers to 0-1. Flat lists become all 0.5.""" values = np.asarray(values, dtype=float) low, high = values.min(), values.max() if high - low < 1e-9: return np.full_like(values, 0.5) return (values - low) / (high - low) def _okr_alignment(clusters, okr_text): """How semantically close each cluster is to the stated objective. Uses the same embedding model as the clustering step, so "free tier restrictions" scores high against an OKR about conversion without anyone hard-coding that relationship. Falls back to word overlap if the model is unavailable. """ texts = [c["definition"] + " " + c["name"] for c in clusters] try: import analyze_semantic vectors = analyze_semantic.embed(texts + [okr_text]) okr_vector = vectors[-1] return [float(v @ okr_vector) for v in vectors[:-1]] except Exception: okr_words = {w.lower().strip(".,") for w in okr_text.split() if len(w) > 3} scores = [] for text in texts: words = {w.lower().strip(".,") for w in text.split() if len(w) > 3} overlap = len(words & okr_words) / max(len(okr_words), 1) scores.append(overlap) return scores def score(clusters, okr_text, weights=None): """Attach factor scores and a weighted total to every cluster.""" weights = {**DEFAULT_WEIGHTS, **(weights or {})} reach = _normalize([c["count"] for c in clusters]) corroboration = _normalize([c["source_spread"] for c in clusters]) severity = _normalize([np.log1p(c["total_engagement"]) for c in clusters]) okr_fit = _normalize(_okr_alignment(clusters, okr_text)) confidence = [] for c in clusters: s = c.get("sentiment") or {} negative, positive = s.get("negative", 0), s.get("positive", 0) rated = negative + positive # A cluster that is 90% complaints is a clearer signal than one that is # half praise. Unrated-only clusters sit in the middle at 0.5. confidence.append(negative / rated if rated else 0.5) confidence = np.asarray(confidence) # A catch-all cluster is an artefact, not a finding. Penalise it hard rather # than letting sheer size win the argument. for i, c in enumerate(clusters): factors = { "reach": float(reach[i]), "corroboration": float(corroboration[i]), "severity": float(severity[i]), "okr_fit": float(okr_fit[i]), "confidence": float(confidence[i]), } total = sum(factors[k] * weights[k] for k in weights) if c.get("is_catchall"): total *= 0.5 c["penalised"] = True c["factors"] = factors c["score"] = round(total, 3) return sorted(clusters, key=lambda c: -c["score"]) def _strongest_factor(cluster): return max(cluster["factors"].items(), key=lambda kv: kv[1]) def _weakest_factor(cluster): return min(cluster["factors"].items(), key=lambda kv: kv[1]) FACTOR_PHRASES = { "reach": "the number of people raising it", "corroboration": "how many separate channels raise it", "severity": "how strongly people react to it", "okr_fit": "how closely it maps to the current objective", "confidence": "how cleanly it reads as a complaint rather than mixed feedback", } def recommend(clusters, okr_text, weights=None, top_n=3): """Return a markdown decision memo arguing for the top-scoring opportunity.""" ranked = score(clusters, okr_text, weights) weights = {**DEFAULT_WEIGHTS, **(weights or {})} # Never recommend a cluster already flagged as a catch-all. Its size is an # artefact of the clustering, not evidence that many people care. It stays # in the ranking so you can see it, but it cannot win the argument. candidates = [c for c in ranked if not c.get("is_catchall")] or ranked winner = candidates[0] runners = [c for c in ranked if c is not winner][:top_n] best_name, best_value = _strongest_factor(winner) worst_name, worst_value = _weakest_factor(winner) sentiment = winner.get("sentiment") or {} sources = ", ".join( f"{k.replace('_', ' ')} ({v})" for k, v in sorted(winner["by_source"].items()) ) lines = [ "# Recommendation", "", f"**Objective in play:** {okr_text}", "", f"## Build this next: {winner['name']}", "", f"Score {winner['score']} out of a possible 1.0, ahead of " f"{runners[0]['name'][:55]} at {runners[0]['score']}." if runners else f"Score {winner['score']}.", "", "### Why", "", f"- Raised in {winner['count']} pieces of feedback, " f"{winner['share']}% of everything analyzed", f"- Appears across {winner['source_spread']} independent sources: {sources}", f"- Sentiment split: {sentiment.get('negative', 0)} complaining, " f"{sentiment.get('positive', 0)} praising, " f"{sentiment.get('unrated', 0)} unrated", f"- Strongest factor is {FACTOR_PHRASES[best_name]} " f"(scored {best_value:.2f})", "", "Representative feedback:", "", ] for example in winner.get("examples", [])[:3]: lines.append(f"> [{example['source'].replace('_', ' ')}] {example['text']}") lines.append("") lines += ["### Why not the alternatives", ""] for r in runners: rw_name, rw_value = _weakest_factor(r) note = ( "flagged as a catch-all cluster, so its size is an artefact rather " "than a finding" if r.get("penalised") else f"weakest on {FACTOR_PHRASES[rw_name]} ({rw_value:.2f})" ) lines.append( f"- **{r['name'][:60]}** (score {r['score']}, {r['count']} mentions, " f"{r['source_spread']} sources): {note}" ) lines.append("") lines += [ "### The strongest case against this recommendation", "", f"- This scores weakest on {FACTOR_PHRASES[worst_name]} " f"({worst_value:.2f}). If that factor matters more than the weights " f"currently say, the ranking changes.", ] if sentiment.get("positive", 0) > 0: lines.append( f"- {sentiment['positive']} people mentioned this topic approvingly. " f"Some of what looks like a problem may be the product working as " f"intended for a different segment." ) lines += [ "- All of this is self-reported public feedback. People who write " "reviews are not a representative sample of people who use the product.", "- Nothing here measures how much it would cost to fix, or whether a " "fix is technically possible. This ranks problems, not solutions.", "", "### What would change the answer", "", ] for name, weight in sorted(weights.items(), key=lambda kv: -kv[1]): lines.append( f"- **{name}** is weighted at {weight:.0%}. " f"This recommendation scored {winner['factors'][name]:.2f} on it." ) lines += [ "", "Change the weights in the config and re-run. If the recommendation " "survives a reasonable range of weightings, it is a robust pick. If it " "flips easily, the top two are effectively tied and the decision needs " "evidence this data cannot provide.", "", "### How you would know if this was the right call", "", "Define the experiment before building: what metric should move, by how " "much, and how long before you call it. A recommendation you cannot " "falsify is an opinion.", "", "---", "", "*Generated by the scoring model in decide.py. Every factor and weight " "is visible and tunable. Disagree with the output by changing a number, " "not by overruling a black box.*", "", ] return "\n".join(lines), ranked