| """analysis/tone_drift.py — multi-quarter transcript drift signals for the Analyst Edge layer. |
| |
| Pure Python + sentence-transformers, zero LLM calls. Extends the textdiff |
| approach (filings) to earnings-call transcripts across the last N quarters: |
| |
| 1. tone_trend — hedge/certainty word rate trajectory in management speech |
| 2. topic_arc — lexicon term frequency rising/falling across 3+ calls |
| 3. recurring_evasion — analyst question asked on 2+ calls, answered evasively |
| 4. topic_fade — topic prominent in prior prepared remarks, absent now |
| |
| Usage: |
| from analysis.tone_drift import compute |
| signals = compute("NVDA") |
| """ |
| from __future__ import annotations |
|
|
| import re |
| from collections import Counter |
|
|
| from analysis.signals import QuarterDelta |
| from analysis.textdiff import ( |
| _LEXICON, |
| _KPI_PATTERNS, |
| _detect_trend, |
| _embed, |
| _find_context_sentence, |
| _split_sentences, |
| _truncate, |
| ) |
| from analysis.transcript_parse import ParsedCall, parse_call |
| from storage.sections_db import _transcript_sort_key, get_recent_transcripts |
|
|
| |
| |
| |
|
|
| |
| _SPOKEN_HEDGES = [ |
| "i think", "we believe", "sort of", "kind of", "we'll see", |
| "hard to say", "too early", "it depends", "uncertain", "cautious", |
| "headwind", "challenge", "moderate", "soften", "roughly", |
| "somewhat", "a bit of", "remains to be seen", |
| ] |
|
|
| _CERTAINTY_WORDS = [ |
| "we will", "confident", "strong", "record", "robust", |
| "momentum", "accelerat", "very pleased", "outstanding", "exceed", |
| "ahead of plan", "better than expected", |
| ] |
|
|
| |
| _DEFLECTIONS = [ |
| "we don't guide", "we do not guide", "not going to guide", |
| "too early to say", "too early to tell", "as i said", "as we said", |
| "we don't disclose", "we do not disclose", "won't break out", |
| "don't break out", "not going to get into", "stay tuned", |
| "more to come", "we'll see how", |
| ] |
|
|
| |
| _TRANSCRIPT_EXTRA_TERMS: list[tuple[str, str]] = [ |
| (r"\bdemand\b", "demand"), |
| (r"\binventory\b", "inventory"), |
| (r"\bpricing\b", "pricing"), |
| (r"\bvisibility\b", "visibility"), |
| (r"\bsupply\b", "supply"), |
| (r"\bchina\b", "China"), |
| (r"\bbacklog\b", "backlog"), |
| ] |
|
|
| _EVASION_SIM_THRESHOLD = 0.60 |
| _MIN_QUESTION_WORDS = 15 |
| _MAX_QUESTIONS_PER_CALL = 30 |
| _QUESTION_EMBED_WORDS = 120 |
| _DIGIT_RATIO_THRESHOLD = 0.005 |
| _SHORT_ANSWER_WORDS = 80 |
|
|
| _MAX_TOTAL = 6 |
|
|
| _STOPWORDS = frozenset( |
| "a an and are as at be but by can could for from has have how i if in is it " |
| "just like me my of on or our so that the then there this to was we what when " |
| "which will with would you your about more very really them they those these " |
| "going get got want wanted maybe think know kind sort little also any do does " |
| "your guys thanks thank question congrats curious wondering color give us " |
| "side year years quarter quarters talk talked talking look looking lot bit " |
| "say said see seeing help understand grew growing mentioned should".split() |
| ) |
|
|
| |
| |
| |
|
|
|
|
| def _word_count(text: str) -> int: |
| return len(text.split()) |
|
|
|
|
| def _phrase_rate(text: str, phrases: list[str]) -> int: |
| """Occurrences of any phrase per 10k words, rounded to int.""" |
| words = _word_count(text) |
| if words == 0: |
| return 0 |
| lower = text.lower() |
| hits = sum(lower.count(p) for p in phrases) |
| return round(hits / words * 10_000) |
|
|
|
|
| def _rep_sentence(text: str, phrases: list[str]) -> str: |
| """Shortest quotable sentence containing one of the phrases.""" |
| matches = [ |
| s for s in _split_sentences(text) |
| if any(p in s.lower() for p in phrases) |
| ] |
| if not matches: |
| return "" |
| return _truncate(min(matches, key=lambda s: len(s.split())), 60) |
|
|
|
|
| def _question_topic(question: str) -> str: |
| """2-3 most frequent non-stopword tokens as a compact topic label.""" |
| tokens = re.findall(r"[a-z]{3,}", question.lower()) |
| counts = Counter(t for t in tokens if t not in _STOPWORDS) |
| return " / ".join(t for t, _ in counts.most_common(3)) |
|
|
|
|
| def _is_evasive(answer: str) -> tuple[bool, str]: |
| """(evasive?, deflection phrase found or '').""" |
| lower = answer.lower() |
| for phrase in _DEFLECTIONS: |
| if phrase in lower: |
| return True, phrase |
| words = answer.split() |
| if words and len(words) < _SHORT_ANSWER_WORDS: |
| digit_tokens = sum(1 for w in words if any(c.isdigit() for c in w)) |
| if digit_tokens / len(words) < _DIGIT_RATIO_THRESHOLD: |
| return True, "" |
| return False, "" |
|
|
|
|
| |
| |
| |
|
|
| def compute_tone_trend(calls: list[ParsedCall]) -> list[QuarterDelta]: |
| """Hedge/certainty word rate trajectory across 3+ calls (management speech only).""" |
| usable = [c for c in calls if _word_count(c.management_text) >= 500] |
| if len(usable) < 3: |
| return [] |
|
|
| deltas: list[QuarterDelta] = [] |
| series_specs = [ |
| (_SPOKEN_HEDGES, "hedging language", "hedge-word", |
| {"rising": "more cautious", "falling": "more confident"}), |
| (_CERTAINTY_WORDS, "confidence language", "certainty-word", |
| {"rising": "more confident", "falling": "more cautious"}), |
| ] |
|
|
| for phrases, term, rate_label, direction_map in series_specs: |
| rates = [_phrase_rate(c.management_text, phrases) for c in usable] |
| trend = _detect_trend(rates) |
| if trend is None: |
| continue |
|
|
| direction, run_str, _ = trend.split() |
| run_quarters = int(run_str) |
| reading = direction_map[direction] |
| first, last = rates[0], rates[-1] |
| net_change = abs(last - first) / (first or 1) |
| sig = "HIGH" if (run_quarters >= 4 or net_change >= 0.5) else "MEDIUM" |
|
|
| trajectory = "→".join(str(r) for r in rates) |
| metric = ( |
| f"{rate_label} rate {trajectory} per 10k words over " |
| f"{usable[0].period}→{usable[-1].period} ({trend}) → {reading}" |
| ) |
| deltas.append(QuarterDelta( |
| kind="tone_trend", |
| period_from=usable[0].period, |
| period_to=usable[-1].period, |
| before_text=_rep_sentence(usable[0].management_text, phrases), |
| after_text=_rep_sentence(usable[-1].management_text, phrases), |
| computed_metric=metric, |
| source="transcript", |
| significance=sig, |
| term=term, |
| )) |
|
|
| return deltas |
|
|
|
|
| |
| |
| |
|
|
| def compute_topic_arcs( |
| calls: list[ParsedCall], |
| raw_texts: list[str], |
| ) -> list[QuarterDelta]: |
| """Lexicon terms whose mention count rises/falls monotonically across 3+ calls.""" |
| if len(calls) < 3: |
| return [] |
|
|
| deltas: list[QuarterDelta] = [] |
| for pattern, label in _LEXICON + _TRANSCRIPT_EXTRA_TERMS: |
| counts = [len(re.findall(pattern, t, re.IGNORECASE)) for t in raw_texts] |
| if max(counts) < 3: |
| continue |
| trend = _detect_trend(counts) |
| if trend is None: |
| continue |
|
|
| run_quarters = int(trend.split()[1]) |
| metric = ( |
| f"{counts[0]}→{counts[-1]} mentions over " |
| f"{calls[0].period}→{calls[-1].period} ({trend})" |
| ) |
| deltas.append(QuarterDelta( |
| kind="topic_arc", |
| period_from=calls[0].period, |
| period_to=calls[-1].period, |
| before_text=_find_context_sentence(raw_texts[0], pattern) if counts[0] else "", |
| after_text=_find_context_sentence(raw_texts[-1], pattern) if counts[-1] else "", |
| computed_metric=metric, |
| source="transcript", |
| significance="HIGH" if run_quarters >= 4 else "MEDIUM", |
| term=label, |
| )) |
|
|
| deltas.sort(key=lambda d: {"HIGH": 0, "MEDIUM": 1}.get(d.significance, 2)) |
| return deltas[:3] |
|
|
|
|
| |
| |
| |
|
|
| def compute_recurring_evasions(calls: list[ParsedCall]) -> list[QuarterDelta]: |
| """Analyst question topics raised on 2+ calls where answers stay non-quantitative.""" |
| qa_calls = [c for c in calls if c.qa] |
| if len(qa_calls) < 2: |
| return [] |
|
|
| latest = qa_calls[-1] |
| priors = qa_calls[:-1] |
|
|
| def _select(call: ParsedCall) -> list: |
| picked = [x for x in call.qa if _word_count(x.question) >= _MIN_QUESTION_WORDS] |
| return picked[:_MAX_QUESTIONS_PER_CALL] |
|
|
| latest_qs = _select(latest) |
| prior_qs: list[tuple[str, object]] = [] |
| for call in priors: |
| prior_qs.extend((call.period, x) for x in _select(call)) |
| if not latest_qs or not prior_qs: |
| return [] |
|
|
| texts = ( |
| [_truncate(x.question, _QUESTION_EMBED_WORDS) for x in latest_qs] |
| + [_truncate(x.question, _QUESTION_EMBED_WORDS) for _, x in prior_qs] |
| ) |
| vecs = _embed(texts) |
| latest_vecs = vecs[: len(latest_qs)] |
| prior_vecs = vecs[len(latest_qs):] |
| sim = latest_vecs @ prior_vecs.T |
|
|
| deltas: list[QuarterDelta] = [] |
| used_prior: set[int] = set() |
|
|
| for li, lx in enumerate(latest_qs): |
| matched = [ |
| pi for pi in range(len(prior_qs)) |
| if pi not in used_prior and sim[li, pi] >= _EVASION_SIM_THRESHOLD |
| ] |
| if not matched: |
| continue |
|
|
| |
| cluster = [(prior_qs[pi][0], prior_qs[pi][1]) for pi in matched] |
| cluster.append((latest.period, lx)) |
| periods = sorted({p for p, _ in cluster}, key=_transcript_sort_key) |
| if len(periods) < 2: |
| continue |
|
|
| evasive_flags = [_is_evasive(x.answer) for _, x in cluster] |
| n_evasive = sum(1 for flag, _ in evasive_flags if flag) |
| |
| |
| if n_evasive < 2 or n_evasive * 2 < len(cluster): |
| continue |
| used_prior.update(matched) |
|
|
| deflection = next((p for flag, p in evasive_flags if flag and p), "") |
| deflection_str = f" (deflection: '{deflection}')" if deflection else "" |
| metric = ( |
| f"asked in {', '.join(periods)}; " |
| f"{n_evasive}/{len(cluster)} answers non-quantitative{deflection_str}" |
| ) |
|
|
| earliest_period, earliest_x = cluster[0] |
| latest_evasive = next( |
| (x for (_, x), (flag, _) in zip(reversed(cluster), reversed(evasive_flags)) if flag), |
| lx, |
| ) |
| answer_quote = ( |
| _rep_sentence(latest_evasive.answer, [deflection]) if deflection else "" |
| ) or _truncate(latest_evasive.answer, 60) |
|
|
| deltas.append(QuarterDelta( |
| kind="recurring_evasion", |
| period_from=periods[0], |
| period_to=latest.period, |
| before_text=_truncate(f"{earliest_x.analyst}: {earliest_x.question}", 60), |
| after_text=answer_quote, |
| computed_metric=metric, |
| source="transcript", |
| significance="HIGH" if len(periods) >= 3 else "MEDIUM", |
| term=_question_topic(lx.question), |
| )) |
|
|
| deltas.sort(key=lambda d: {"HIGH": 0, "MEDIUM": 1}.get(d.significance, 2)) |
| return deltas[:2] |
|
|
|
|
| |
| |
| |
|
|
| def compute_topic_fades(calls: list[ParsedCall]) -> list[QuarterDelta]: |
| """Topic with 2+ mentions in 2+ prior prepared remarks, absent from the latest.""" |
| usable = [c for c in calls if _word_count(c.prepared_text) >= 300] |
| if len(usable) < 3: |
| return [] |
|
|
| latest = usable[-1] |
| priors = usable[:-1] |
|
|
| deltas: list[QuarterDelta] = [] |
| seen_labels: set[str] = set() |
| for pattern, label in _KPI_PATTERNS + _TRANSCRIPT_EXTRA_TERMS: |
| if label in seen_labels: |
| continue |
| seen_labels.add(label) |
| prior_counts = [ |
| (c, len(re.findall(pattern, c.prepared_text, re.IGNORECASE))) |
| for c in priors |
| ] |
| prominent = [(c, n) for c, n in prior_counts if n >= 2] |
| if len(prominent) < 2: |
| continue |
| if re.search(pattern, latest.prepared_text, re.IGNORECASE): |
| continue |
|
|
| periods_str = " and ".join(c.period for c, _ in prominent) |
| counts_str = ", ".join(str(n) for _, n in prominent) |
| last_prominent = prominent[-1][0] |
| deltas.append(QuarterDelta( |
| kind="topic_fade", |
| period_from=prominent[0][0].period, |
| period_to=latest.period, |
| before_text=_find_context_sentence(last_prominent.prepared_text, pattern), |
| after_text="", |
| computed_metric=( |
| f"'{label}' in prepared remarks of {periods_str} " |
| f"({counts_str} mentions), absent in {latest.period}" |
| ), |
| source="transcript", |
| significance="MEDIUM", |
| term=label, |
| )) |
|
|
| return deltas[:2] |
|
|
|
|
| |
| |
| |
|
|
| def compute(ticker: str, n: int = 4) -> list[QuarterDelta]: |
| """Compute all transcript drift signals for a ticker. |
| |
| Examines the last *n* non-empty transcripts. Returns an empty list if |
| fewer than 2 are available or on any error — never raises. |
| """ |
| try: |
| return _compute_inner(ticker, n) |
| except Exception as exc: |
| import sys |
| print(f"[tone_drift] Error computing deltas for {ticker}: {exc}", file=sys.stderr) |
| return [] |
|
|
|
|
| def _compute_inner(ticker: str, n: int) -> list[QuarterDelta]: |
| transcripts = get_recent_transcripts(ticker.upper(), n=n) |
| if len(transcripts) < 2: |
| return [] |
|
|
| raw_texts = [text for _, text in transcripts] |
| calls = [parse_call(period, text) for period, text in transcripts] |
|
|
| all_deltas: list[QuarterDelta] = [] |
| all_deltas.extend(compute_tone_trend(calls)) |
| all_deltas.extend(compute_topic_arcs(calls, raw_texts)) |
| all_deltas.extend(compute_recurring_evasions(calls)) |
| all_deltas.extend(compute_topic_fades(calls)) |
|
|
| |
| seen: set[tuple[str, str]] = set() |
| deduped: list[QuarterDelta] = [] |
| order = {"HIGH": 0, "MEDIUM": 1, "LOW": 2} |
| all_deltas.sort(key=lambda d: (order[d.significance], d.kind)) |
| for d in all_deltas: |
| key = (d.kind, d.term) |
| if key not in seen: |
| seen.add(key) |
| deduped.append(d) |
| return deduped[:_MAX_TOTAL] |
|
|