Spaces:
Sleeping
Sleeping
File size: 11,614 Bytes
e6496c0 5b9b478 e6496c0 5b9b478 e6496c0 1bc4598 e6496c0 5b9b478 e6496c0 5b9b478 1bc4598 e6496c0 1bc4598 e6496c0 5b9b478 e6496c0 1bc4598 e6496c0 1bc4598 e6496c0 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 | """analysis/transcript_parse.py β structural parsing of flattened earnings-call text.
Pure regex, zero embeddings, zero LLM. Operates on the Alpha Vantage transcript
format produced by ingestion/transcript.py: one line per speaker segment,
either "Speaker: content" or "Speaker (Title): content".
Reconstructs:
- prepared remarks vs Q&A boundary
- speaker roles (operator / management / analyst)
- analyst question β management answer exchanges
Degrades gracefully: if the structure cannot be recovered, the full text is
treated as management speech and the Q&A exchange list is empty.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
# A speaker's role routinely carries its own parentheses β the common form is
# "Jane Roe (Analyst (Example Bank)):". A title pattern that stops at the first
# closing bracket never reaches the colon on those lines, so the turn was not
# recognised as a speaker change at all and got appended to whoever spoke last.
# One level of nesting is enough for every format observed.
_TITLE = r"(?:[^()]|\([^()]*\))"
# "Speaker: content" or "Speaker (Title): content" at line start.
_SPEAKER_RE = re.compile(
rf"^([A-Z][\w.\-' ]{{0,60}}?)(?:\s*\(({_TITLE}{{1,80}})\))?:\s+(.*)$"
)
# Marks the transition from prepared remarks to analyst Q&A.
_QA_BOUNDARY_RE = re.compile(
r"question-and-answer session"
r"|question and answer session"
r"|ready to (?:start|begin) the q\s*&\s*a"
r"|open (?:up )?the (?:call|line|floor)s? for questions"
r"|poll (?:the audience|the lines?|for) ?(?:for questions|questions)?"
r"|(?:take|taking) (?:your|the first) questions?"
r"|first question comes? from",
re.IGNORECASE,
)
# Operator hand-offs that name the analyst and their firm.
_ANALYST_INTRO_RE = re.compile(
r"(?:line of|comes? from(?: the line of)?)\s+([A-Z][\w.\-' ]+?)\s+(?:with|from|at)\s+",
)
_MIN_SEGMENTS = 5
@dataclass
class Segment:
speaker: str
text: str
role: str = "unknown" # operator | management | analyst | unknown
@dataclass
class QAExchange:
analyst: str
question: str
answer: str
@dataclass
class ParsedCall:
period: str
prepared_text: str # management prepared remarks (pre-Q&A)
management_text: str # prepared remarks + all answers
qa: list[QAExchange] = field(default_factory=list)
n_segments: int = 0
# Answers alone, so a caller can tell what management volunteered from
# what analysts made it address. Empty when no Q&A boundary was found.
answers_text: str = ""
# Every voice on the call, in order of first appearance. Term extraction
# needs them: a name announced at each hand-off is not a subject.
speakers: list[str] = field(default_factory=list)
def _last_name(name: str) -> str:
parts = name.strip().rstrip(".").split()
return parts[-1].lower() if parts else ""
def _split_segments(text: str) -> list[Segment]:
segments: list[Segment] = []
for line in text.splitlines():
line = line.strip()
if not line:
continue
m = _SPEAKER_RE.match(line)
if m:
title = m.group(2) or ""
speaker = m.group(1).strip()
for seg in _split_inline_speakers(speaker, title, m.group(3).strip()):
segments.append(seg)
elif segments:
segments[-1].text += " " + line
return segments
# Some providers put a whole exchange on one line: the operator's hand-off and
# the analyst's question share it, as in
#
# Operator: Our first question comes from Jane Roe with Example Bank.
# Jane Roe (Analyst (Example Bank)): I have two, one for...
#
# arriving as a single line. Matching only at line start then buried every
# analyst turn inside the operator's segment, so the call parsed to zero
# questions despite being fully structured. Requiring a multi-word capitalised
# name followed by a parenthesised role keeps ordinary prose ("the ratio (as
# defined): ...") from being mistaken for a speaker change.
_INLINE_SPEAKER_RE = re.compile(
rf"(?<=\s)((?:[A-Z][\w.\-']*\s){{1,3}}[A-Z][\w.\-']*)\s*\(({_TITLE}{{1,80}})\):\s+"
)
def _split_inline_speakers(speaker: str, title: str, body: str) -> list[Segment]:
"""Split one line into every speaker turn it actually contains."""
def _make(name: str, role_title: str, content: str) -> Segment:
seg = Segment(speaker=name.strip(), text=content.strip())
if role_title and re.search(r"analyst", role_title, re.IGNORECASE):
seg.role = "analyst"
return seg
matches = list(_INLINE_SPEAKER_RE.finditer(body))
if not matches:
return [_make(speaker, title, body)]
out = [_make(speaker, title, body[: matches[0].start()])]
for index, match in enumerate(matches):
end = matches[index + 1].start() if index + 1 < len(matches) else len(body)
out.append(_make(match.group(1), match.group(2), body[match.end():end]))
# A hand-off line can leave the operator with nothing but the introduction;
# keep it only when it carries text of its own.
return [seg for seg in out if seg.text]
def _ordered_speakers(segments: list[Segment]) -> list[str]:
"""Distinct speaker names, in order of first appearance."""
seen: list[str] = []
for seg in segments:
name = seg.speaker.strip()
if name and name not in seen:
seen.append(name)
return seen
def parse_call(period: str, text: str) -> ParsedCall:
"""Parse one flattened transcript into roles and Q&A exchanges.
Never raises. Falls back to a Q&A-free ParsedCall whose prepared/management
text is the full transcript when structure cannot be recovered.
"""
segments = _split_segments(text or "")
if len(segments) < _MIN_SEGMENTS:
full = (text or "").strip()
return ParsedCall(
period=period, prepared_text=full, management_text=full,
qa=[], n_segments=len(segments),
speakers=_ordered_speakers(segments),
)
# Locate the prepared-remarks β Q&A boundary. Operator intros often
# announce "there will be a question-and-answer session" before anyone
# has spoken β only accept a boundary once a non-operator segment exists.
qa_start: int | None = None
for i, seg in enumerate(segments):
if not _QA_BOUNDARY_RE.search(seg.text):
continue
has_speech_before = any(
s.speaker.lower() != "operator" for s in segments[:i]
)
if has_speech_before:
qa_start = i
break
# Structural fallback when no announcement phrase matched.
#
# Transcript providers word the hand-off differently, and some drop the
# announcement entirely β Apple's calls are a standing example, where the
# host moves straight to the first analyst. The section is still plainly
# there in the structure: an operator turn, then a voice that has not
# spoken yet. That new voice is the first analyst, so the operator turn
# before it is the boundary. Requiring prior speech keeps an operator's
# opening housekeeping from being mistaken for the Q&A.
if qa_start is None:
heard: set[str] = set()
for i, seg in enumerate(segments):
if seg.speaker.lower() != "operator":
heard.add(_last_name(seg.speaker))
continue
if not heard:
continue
following = next(
(s for s in segments[i + 1:] if s.speaker.lower() != "operator"),
None,
)
if following is not None and _last_name(following.speaker) not in heard:
qa_start = i
break
# Last resort: a call with no operator at all.
#
# Some issuers run their own Q&A β an investor-relations host reads the
# questions and there is never an "Operator" turn to key off. Tesla's calls
# are the standing example. What still holds is the shape: prepared remarks
# from a handful of known voices, then a voice that has not been heard yet
# asking something. The question mark is what separates that from a second
# executive joining the prepared remarks.
if qa_start is None:
heard = set()
for i, seg in enumerate(segments):
name = _last_name(seg.speaker)
# Two prior voices β host plus at least one executive, i.e. the
# prepared section is genuinely under way.
if len(heard) >= 2 and name not in heard and "?" in seg.text:
qa_start = i
break
heard.add(name)
# Analyst roster from Operator hand-offs (anywhere in the call).
roster: set[str] = set()
for seg in segments:
if seg.speaker.lower() == "operator":
seg.role = "operator"
for m in _ANALYST_INTRO_RE.finditer(seg.text):
roster.add(_last_name(m.group(1)))
# Management = non-operator speakers heard before the Q&A boundary.
pre_qa_end = qa_start if qa_start is not None else len(segments)
management: set[str] = {
_last_name(seg.speaker)
for seg in segments[:pre_qa_end]
if seg.role not in ("operator", "analyst") and seg.speaker
}
# Assign roles.
prev_role = ""
for i, seg in enumerate(segments):
if seg.role in ("operator", "analyst"):
prev_role = seg.role
continue
key = _last_name(seg.speaker)
if key in management:
seg.role = "management"
elif key in roster:
seg.role = "analyst"
elif qa_start is not None and i > qa_start:
# Unknown speaker in Q&A: question-shaped or operator hand-off β analyst.
if seg.text.rstrip().endswith("?") or prev_role == "operator":
seg.role = "analyst"
else:
seg.role = "management"
else:
seg.role = "management"
prev_role = seg.role
prepared_parts = [
seg.text for seg in segments[:pre_qa_end] if seg.role == "management"
]
management_parts = [seg.text for seg in segments if seg.role == "management"]
answer_parts = [
seg.text for seg in segments[pre_qa_end:] if seg.role == "management"
]
# Pair analyst turns with the management turns that follow them.
qa: list[QAExchange] = []
if qa_start is not None:
current: QAExchange | None = None
for seg in segments[qa_start:]:
if seg.role == "analyst":
if current is None or current.answer:
if current is not None and current.answer:
qa.append(current)
current = QAExchange(analyst=seg.speaker, question=seg.text, answer="")
else:
current.question += " " + seg.text
elif seg.role == "management" and current is not None:
current.answer = (current.answer + " " + seg.text).strip()
elif seg.role == "operator" and current is not None and current.answer:
qa.append(current)
current = None
if current is not None and current.answer:
qa.append(current)
return ParsedCall(
period=period,
prepared_text="\n".join(prepared_parts).strip(),
management_text="\n".join(management_parts).strip(),
qa=qa,
n_segments=len(segments),
answers_text="\n".join(answer_parts).strip(),
speakers=_ordered_speakers(segments),
)
|