Topic 3: Feedback Loops
8 min read·21 Sept 2026
module13/feedback.py
python
# module13/feedback.py
"""Mining query logs for failures, turning them into data, and improving without silent regression."""
from __future__ import annotations
import re
from collections import Counter, defaultdict
from dataclasses import dataclass, field
from datetime import datetime
import numpy as np
# ---------------------------------------------------------------- 1. failure signals in logs
FAILURE_SIGNALS = {
"zero_results": "retrieval returned nothing: vocabulary gap or over-strict filter",
"low_score": "best score below the abstention threshold: likely a corpus gap",
"refused": "the system abstained: corpus gap, or the threshold is too strict",
"reformulation": "the user rewrote the question within the session: the first answer missed",
"repeat_question": "the same question again later: the answer did not stick",
"thumbs_down": "explicit rejection",
"escalation": "handed to a human: the strongest negative signal",
"no_engagement": "no click, copy, or dwell: weak but numerous",
}
@dataclass
class QueryEvent:
session_id: str
at: str
query: str
retrieved_ids: list[str] = field(default_factory=list)
top_score: float = 0.0
refused: bool = False
thumbs: str | None = None
escalated: bool = False
clicked_id: str | None = None
dwell_seconds: float = 0.0
def normalized(text: str) -> str:
return " ".join(re.findall(r"[a-z0-9]+", text.lower()))
def is_reformulation(first: str, second: str, min_overlap: float = 0.3) -> bool:
"""A follow-up that reuses much of the first question is usually a retry, not a new topic."""
a, b = set(normalized(first).split()), set(normalized(second).split())
if not a or not b or a == b:
return bool(a and a == b)
return len(a & b) / len(a | b) >= min_overlap
def mine_failures(events: list[QueryEvent], min_score: float = 0.45) -> list[dict]:
"""Every failure signal in one pass, grouped per event, newest signals win."""
by_session: dict[str, list[QueryEvent]] = defaultdict(list)
for event in sorted(events, key=lambda e: (e.session_id, e.at)):
by_session[event.session_id].append(event)
failures = []
for session, session_events in by_session.items():
for index, event in enumerate(session_events):
signals = []
if not event.retrieved_ids:
signals.append("zero_results")
if event.retrieved_ids and event.top_score < min_score:
signals.append("low_score")
if event.refused:
signals.append("refused")
if event.thumbs == "down":
signals.append("thumbs_down")
if event.escalated:
signals.append("escalation")
following = session_events[index + 1] if index + 1 < len(session_events) else None
if following and is_reformulation(event.query, following.query):
signals.append("reformulation")
if not event.clicked_id and event.dwell_seconds < 3 and not event.refused:
signals.append("no_engagement")
if signals:
failures.append({"session_id": session, "query": event.query, "signals": signals,
"top_score": event.top_score, "retrieved_ids": event.retrieved_ids})
return failures
def cluster_failures(failures: list[dict], embedder, clusters: int = 4) -> list[dict]:
"""Group failing questions so you fix themes, not individual tickets."""
queries = [f["query"] for f in failures]
if len(queries) < 2:
return [{"size": len(queries), "examples": queries, "common_signals": []}]
from sklearn.cluster import KMeans
vectors = np.vstack([embedder.embed_query(q) for q in queries])
labels = KMeans(n_clusters=min(clusters, len(queries)), n_init=10, random_state=0).fit_predict(vectors)
grouped = defaultdict(list)
for failure, label in zip(failures, labels):
grouped[int(label)].append(failure)
output = []
for group in grouped.values():
signals = Counter(s for f in group for s in f["signals"])
output.append({"size": len(group), "examples": [f["query"] for f in group[:3]],
"common_signals": signals.most_common(3),
"mean_top_score": round(float(np.mean([f["top_score"] for f in group])), 3)})
return sorted(output, key=lambda row: -row["size"])
def triage(cluster: dict) -> str:
"""Route a failure cluster to the team that can actually fix it."""
signals = dict(cluster["common_signals"])
if signals.get("zero_results") or signals.get("low_score"):
return "corpus: write or import the missing document"
if signals.get("reformulation"):
return "query understanding: vocabulary, synonyms, rewriting (Module 7)"
if signals.get("thumbs_down") or signals.get("escalation"):
return "generation or ranking: read the traces and attribute the stage (Module 11)"
if signals.get("refused"):
return "threshold calibration, or a genuine corpus gap (Module 8)"
return "review manually"
# ---------------------------------------------------------------- 2. feedback into training data
def feedback_to_examples(events: list[QueryEvent], text_of: dict[str, str]) -> dict:
"""Split interactions into positives, negatives, and evaluation cases."""
from module13.finetune import TrainingPair
positives, negatives, eval_cases = [], [], []
for event in events:
if event.clicked_id and event.dwell_seconds >= 10 and event.thumbs != "down":
if event.clicked_id in text_of:
weight = 0.8 if event.thumbs == "up" else 0.5
positives.append(TrainingPair(event.query, text_of[event.clicked_id],
positive_id=event.clicked_id, weight=weight,
source="feedback"))
if event.thumbs == "down" and event.retrieved_ids:
negatives.append({"query": event.query, "rejected_ids": event.retrieved_ids[:3]})
if event.escalated or event.refused:
eval_cases.append({"question": event.query, "kind": "unanswered_in_production",
"note": "escalated" if event.escalated else "refused"})
return {"positives": positives, "negatives": negatives, "eval_cases": eval_cases}
BIASES = {
"position": "users click what is shown first: down-weight top positions or randomize slightly",
"presentation": "richer snippets get clicked more: compare like with like",
"survivorship": "you only see feedback on what you retrieved: mine abstentions too",
"popularity": "training on clicks reinforces what already ranks: hold out a random slice",
"vocal minority": "thumbs are rare and skewed: weight them, do not worship them",
}
# ---------------------------------------------------------------- 3. improving without regressing
@dataclass
class ImprovementGate:
"""A change ships only if it wins overall AND harms no protected segment."""
metric: str = "mrr"
minimum_lift: float = 0.01
protected_segments: tuple = ("code", "policy", "follow-up")
max_segment_drop: float = 0.02
def check_for_silent_regression(baseline: dict[str, list[float]], candidate: dict[str, list[float]],
gate: ImprovementGate) -> dict:
"""Overall gains often hide a segment that got worse. Check every segment separately."""
from module11.metrics import paired_bootstrap
all_base = [v for values in baseline.values() for v in values]
all_cand = [v for values in candidate.values() for v in values]
overall = paired_bootstrap(all_base, all_cand)
segments = {}
for name, base_values in baseline.items():
cand_values = candidate.get(name, [])
if not cand_values:
continue
change = float(np.mean(cand_values) - np.mean(base_values))
segments[name] = {"change": round(change, 3),
"protected": name in gate.protected_segments,
"regressed": change < -gate.max_segment_drop}
blocking = [name for name, row in segments.items() if row["regressed"] and row["protected"]]
return {"overall_lift": round(overall["lift"], 3), "significant": overall["significant"],
"segments": segments, "blocked_by": blocking,
"ship": bool(overall["lift"] >= gate.minimum_lift and overall["significant"] and not blocking)}
CONTINUOUS_IMPROVEMENT_LOOP = [
"1. Mine failures from last week's logs and cluster them",
"2. Triage each cluster to a stage owner (corpus, retrieval, generation)",
"3. Add the clearest cases to the evaluation set (Module 11), labelled by hand",
"4. Make one change (Module 11's discipline: one variable)",
"5. Evaluate offline: overall lift AND per-segment regression",
"6. Shadow or canary on live traffic with guardrail metrics (Module 12)",
"7. Ship behind a flag, watch the dashboards, keep rollback ready",
"8. Feed the new failures back in: the loop never ends",
]