CourseRAG · Module 13: Advanced and Emerging Topics · part 72 of 82
Part 72 · Module 13: Advanced and Emerging Topics

Topic 3: Feedback Loops

8 min read·21 Sept 2026

module13/feedback.py

python
# module13/feedback.py
"""Mining query logs for failures, turning them into data, and improving without silent regression."""
from __future__ import annotations

import re
from collections import Counter, defaultdict
from dataclasses import dataclass, field
from datetime import datetime

import numpy as np

# ---------------------------------------------------------------- 1. failure signals in logs
FAILURE_SIGNALS = {
    "zero_results": "retrieval returned nothing: vocabulary gap or over-strict filter",
    "low_score": "best score below the abstention threshold: likely a corpus gap",
    "refused": "the system abstained: corpus gap, or the threshold is too strict",
    "reformulation": "the user rewrote the question within the session: the first answer missed",
    "repeat_question": "the same question again later: the answer did not stick",
    "thumbs_down": "explicit rejection",
    "escalation": "handed to a human: the strongest negative signal",
    "no_engagement": "no click, copy, or dwell: weak but numerous",
}


@dataclass
class QueryEvent:
    session_id: str
    at: str
    query: str
    retrieved_ids: list[str] = field(default_factory=list)
    top_score: float = 0.0
    refused: bool = False
    thumbs: str | None = None
    escalated: bool = False
    clicked_id: str | None = None
    dwell_seconds: float = 0.0


def normalized(text: str) -> str:
    return " ".join(re.findall(r"[a-z0-9]+", text.lower()))


def is_reformulation(first: str, second: str, min_overlap: float = 0.3) -> bool:
    """A follow-up that reuses much of the first question is usually a retry, not a new topic."""
    a, b = set(normalized(first).split()), set(normalized(second).split())
    if not a or not b or a == b:
        return bool(a and a == b)
    return len(a & b) / len(a | b) >= min_overlap


def mine_failures(events: list[QueryEvent], min_score: float = 0.45) -> list[dict]:
    """Every failure signal in one pass, grouped per event, newest signals win."""
    by_session: dict[str, list[QueryEvent]] = defaultdict(list)
    for event in sorted(events, key=lambda e: (e.session_id, e.at)):
        by_session[event.session_id].append(event)

    failures = []
    for session, session_events in by_session.items():
        for index, event in enumerate(session_events):
            signals = []
            if not event.retrieved_ids:
                signals.append("zero_results")
            if event.retrieved_ids and event.top_score < min_score:
                signals.append("low_score")
            if event.refused:
                signals.append("refused")
            if event.thumbs == "down":
                signals.append("thumbs_down")
            if event.escalated:
                signals.append("escalation")
            following = session_events[index + 1] if index + 1 < len(session_events) else None
            if following and is_reformulation(event.query, following.query):
                signals.append("reformulation")
            if not event.clicked_id and event.dwell_seconds < 3 and not event.refused:
                signals.append("no_engagement")
            if signals:
                failures.append({"session_id": session, "query": event.query, "signals": signals,
                                 "top_score": event.top_score, "retrieved_ids": event.retrieved_ids})
    return failures


def cluster_failures(failures: list[dict], embedder, clusters: int = 4) -> list[dict]:
    """Group failing questions so you fix themes, not individual tickets."""
    queries = [f["query"] for f in failures]
    if len(queries) < 2:
        return [{"size": len(queries), "examples": queries, "common_signals": []}]
    from sklearn.cluster import KMeans
    vectors = np.vstack([embedder.embed_query(q) for q in queries])
    labels = KMeans(n_clusters=min(clusters, len(queries)), n_init=10, random_state=0).fit_predict(vectors)
    grouped = defaultdict(list)
    for failure, label in zip(failures, labels):
        grouped[int(label)].append(failure)
    output = []
    for group in grouped.values():
        signals = Counter(s for f in group for s in f["signals"])
        output.append({"size": len(group), "examples": [f["query"] for f in group[:3]],
                       "common_signals": signals.most_common(3),
                       "mean_top_score": round(float(np.mean([f["top_score"] for f in group])), 3)})
    return sorted(output, key=lambda row: -row["size"])


def triage(cluster: dict) -> str:
    """Route a failure cluster to the team that can actually fix it."""
    signals = dict(cluster["common_signals"])
    if signals.get("zero_results") or signals.get("low_score"):
        return "corpus: write or import the missing document"
    if signals.get("reformulation"):
        return "query understanding: vocabulary, synonyms, rewriting (Module 7)"
    if signals.get("thumbs_down") or signals.get("escalation"):
        return "generation or ranking: read the traces and attribute the stage (Module 11)"
    if signals.get("refused"):
        return "threshold calibration, or a genuine corpus gap (Module 8)"
    return "review manually"


# ---------------------------------------------------------------- 2. feedback into training data
def feedback_to_examples(events: list[QueryEvent], text_of: dict[str, str]) -> dict:
    """Split interactions into positives, negatives, and evaluation cases."""
    from module13.finetune import TrainingPair

    positives, negatives, eval_cases = [], [], []
    for event in events:
        if event.clicked_id and event.dwell_seconds >= 10 and event.thumbs != "down":
            if event.clicked_id in text_of:
                weight = 0.8 if event.thumbs == "up" else 0.5
                positives.append(TrainingPair(event.query, text_of[event.clicked_id],
                                              positive_id=event.clicked_id, weight=weight,
                                              source="feedback"))
        if event.thumbs == "down" and event.retrieved_ids:
            negatives.append({"query": event.query, "rejected_ids": event.retrieved_ids[:3]})
        if event.escalated or event.refused:
            eval_cases.append({"question": event.query, "kind": "unanswered_in_production",
                               "note": "escalated" if event.escalated else "refused"})
    return {"positives": positives, "negatives": negatives, "eval_cases": eval_cases}


BIASES = {
    "position": "users click what is shown first: down-weight top positions or randomize slightly",
    "presentation": "richer snippets get clicked more: compare like with like",
    "survivorship": "you only see feedback on what you retrieved: mine abstentions too",
    "popularity": "training on clicks reinforces what already ranks: hold out a random slice",
    "vocal minority": "thumbs are rare and skewed: weight them, do not worship them",
}


# ---------------------------------------------------------------- 3. improving without regressing
@dataclass
class ImprovementGate:
    """A change ships only if it wins overall AND harms no protected segment."""
    metric: str = "mrr"
    minimum_lift: float = 0.01
    protected_segments: tuple = ("code", "policy", "follow-up")
    max_segment_drop: float = 0.02


def check_for_silent_regression(baseline: dict[str, list[float]], candidate: dict[str, list[float]],
                                gate: ImprovementGate) -> dict:
    """Overall gains often hide a segment that got worse. Check every segment separately."""
    from module11.metrics import paired_bootstrap

    all_base = [v for values in baseline.values() for v in values]
    all_cand = [v for values in candidate.values() for v in values]
    overall = paired_bootstrap(all_base, all_cand)
    segments = {}
    for name, base_values in baseline.items():
        cand_values = candidate.get(name, [])
        if not cand_values:
            continue
        change = float(np.mean(cand_values) - np.mean(base_values))
        segments[name] = {"change": round(change, 3),
                          "protected": name in gate.protected_segments,
                          "regressed": change < -gate.max_segment_drop}
    blocking = [name for name, row in segments.items() if row["regressed"] and row["protected"]]
    return {"overall_lift": round(overall["lift"], 3), "significant": overall["significant"],
            "segments": segments, "blocked_by": blocking,
            "ship": bool(overall["lift"] >= gate.minimum_lift and overall["significant"] and not blocking)}


CONTINUOUS_IMPROVEMENT_LOOP = [
    "1. Mine failures from last week's logs and cluster them",
    "2. Triage each cluster to a stage owner (corpus, retrieval, generation)",
    "3. Add the clearest cases to the evaluation set (Module 11), labelled by hand",
    "4. Make one change (Module 11's discipline: one variable)",
    "5. Evaluate offline: overall lift AND per-segment regression",
    "6. Shadow or canary on live traffic with guardrail metrics (Module 12)",
    "7. Ship behind a flag, watch the dashboards, keep rollback ready",
    "8. Feed the new failures back in: the loop never ends",
]

The rest of this course is yours to keep

This course is bought on its own, once, and stays readable afterwards, including the parts added to it later.