File size: 11,614 Bytes
e6496c0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5b9b478
 
 
 
 
 
 
e6496c0
5b9b478
 
 
e6496c0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1bc4598
 
 
 
 
 
e6496c0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5b9b478
 
e6496c0
 
 
 
 
5b9b478
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1bc4598
 
 
 
 
 
 
 
 
 
e6496c0
 
 
 
 
 
 
 
 
 
 
 
1bc4598
e6496c0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5b9b478
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e6496c0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1bc4598
 
 
e6496c0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1bc4598
 
e6496c0
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
"""analysis/transcript_parse.py β€” structural parsing of flattened earnings-call text.

Pure regex, zero embeddings, zero LLM. Operates on the Alpha Vantage transcript
format produced by ingestion/transcript.py: one line per speaker segment,
either "Speaker: content" or "Speaker (Title): content".

Reconstructs:
  - prepared remarks vs Q&A boundary
  - speaker roles (operator / management / analyst)
  - analyst question β†’ management answer exchanges

Degrades gracefully: if the structure cannot be recovered, the full text is
treated as management speech and the Q&A exchange list is empty.
"""
from __future__ import annotations

import re
from dataclasses import dataclass, field

# A speaker's role routinely carries its own parentheses β€” the common form is
# "Jane Roe (Analyst (Example Bank)):". A title pattern that stops at the first
# closing bracket never reaches the colon on those lines, so the turn was not
# recognised as a speaker change at all and got appended to whoever spoke last.
# One level of nesting is enough for every format observed.
_TITLE = r"(?:[^()]|\([^()]*\))"

# "Speaker: content" or "Speaker (Title): content" at line start.
_SPEAKER_RE = re.compile(
    rf"^([A-Z][\w.\-' ]{{0,60}}?)(?:\s*\(({_TITLE}{{1,80}})\))?:\s+(.*)$"
)

# Marks the transition from prepared remarks to analyst Q&A.
_QA_BOUNDARY_RE = re.compile(
    r"question-and-answer session"
    r"|question and answer session"
    r"|ready to (?:start|begin) the q\s*&\s*a"
    r"|open (?:up )?the (?:call|line|floor)s? for questions"
    r"|poll (?:the audience|the lines?|for) ?(?:for questions|questions)?"
    r"|(?:take|taking) (?:your|the first) questions?"
    r"|first question comes? from",
    re.IGNORECASE,
)

# Operator hand-offs that name the analyst and their firm.
_ANALYST_INTRO_RE = re.compile(
    r"(?:line of|comes? from(?: the line of)?)\s+([A-Z][\w.\-' ]+?)\s+(?:with|from|at)\s+",
)

_MIN_SEGMENTS = 5


@dataclass
class Segment:
    speaker: str
    text: str
    role: str = "unknown"  # operator | management | analyst | unknown


@dataclass
class QAExchange:
    analyst: str
    question: str
    answer: str


@dataclass
class ParsedCall:
    period: str
    prepared_text: str       # management prepared remarks (pre-Q&A)
    management_text: str     # prepared remarks + all answers
    qa: list[QAExchange] = field(default_factory=list)
    n_segments: int = 0
    # Answers alone, so a caller can tell what management volunteered from
    # what analysts made it address. Empty when no Q&A boundary was found.
    answers_text: str = ""
    # Every voice on the call, in order of first appearance. Term extraction
    # needs them: a name announced at each hand-off is not a subject.
    speakers: list[str] = field(default_factory=list)


def _last_name(name: str) -> str:
    parts = name.strip().rstrip(".").split()
    return parts[-1].lower() if parts else ""


def _split_segments(text: str) -> list[Segment]:
    segments: list[Segment] = []
    for line in text.splitlines():
        line = line.strip()
        if not line:
            continue
        m = _SPEAKER_RE.match(line)
        if m:
            title = m.group(2) or ""
            speaker = m.group(1).strip()
            for seg in _split_inline_speakers(speaker, title, m.group(3).strip()):
                segments.append(seg)
        elif segments:
            segments[-1].text += " " + line
    return segments


# Some providers put a whole exchange on one line: the operator's hand-off and
# the analyst's question share it, as in
#
#   Operator: Our first question comes from Jane Roe with Example Bank.
#   Jane Roe (Analyst (Example Bank)): I have two, one for...
#
# arriving as a single line. Matching only at line start then buried every
# analyst turn inside the operator's segment, so the call parsed to zero
# questions despite being fully structured. Requiring a multi-word capitalised
# name followed by a parenthesised role keeps ordinary prose ("the ratio (as
# defined): ...") from being mistaken for a speaker change.
_INLINE_SPEAKER_RE = re.compile(
    rf"(?<=\s)((?:[A-Z][\w.\-']*\s){{1,3}}[A-Z][\w.\-']*)\s*\(({_TITLE}{{1,80}})\):\s+"
)


def _split_inline_speakers(speaker: str, title: str, body: str) -> list[Segment]:
    """Split one line into every speaker turn it actually contains."""

    def _make(name: str, role_title: str, content: str) -> Segment:
        seg = Segment(speaker=name.strip(), text=content.strip())
        if role_title and re.search(r"analyst", role_title, re.IGNORECASE):
            seg.role = "analyst"
        return seg

    matches = list(_INLINE_SPEAKER_RE.finditer(body))
    if not matches:
        return [_make(speaker, title, body)]

    out = [_make(speaker, title, body[: matches[0].start()])]
    for index, match in enumerate(matches):
        end = matches[index + 1].start() if index + 1 < len(matches) else len(body)
        out.append(_make(match.group(1), match.group(2), body[match.end():end]))
    # A hand-off line can leave the operator with nothing but the introduction;
    # keep it only when it carries text of its own.
    return [seg for seg in out if seg.text]


def _ordered_speakers(segments: list[Segment]) -> list[str]:
    """Distinct speaker names, in order of first appearance."""
    seen: list[str] = []
    for seg in segments:
        name = seg.speaker.strip()
        if name and name not in seen:
            seen.append(name)
    return seen


def parse_call(period: str, text: str) -> ParsedCall:
    """Parse one flattened transcript into roles and Q&A exchanges.

    Never raises. Falls back to a Q&A-free ParsedCall whose prepared/management
    text is the full transcript when structure cannot be recovered.
    """
    segments = _split_segments(text or "")
    if len(segments) < _MIN_SEGMENTS:
        full = (text or "").strip()
        return ParsedCall(
            period=period, prepared_text=full, management_text=full,
            qa=[], n_segments=len(segments),
            speakers=_ordered_speakers(segments),
        )

    # Locate the prepared-remarks β†’ Q&A boundary. Operator intros often
    # announce "there will be a question-and-answer session" before anyone
    # has spoken β€” only accept a boundary once a non-operator segment exists.
    qa_start: int | None = None
    for i, seg in enumerate(segments):
        if not _QA_BOUNDARY_RE.search(seg.text):
            continue
        has_speech_before = any(
            s.speaker.lower() != "operator" for s in segments[:i]
        )
        if has_speech_before:
            qa_start = i
            break

    # Structural fallback when no announcement phrase matched.
    #
    # Transcript providers word the hand-off differently, and some drop the
    # announcement entirely β€” Apple's calls are a standing example, where the
    # host moves straight to the first analyst. The section is still plainly
    # there in the structure: an operator turn, then a voice that has not
    # spoken yet. That new voice is the first analyst, so the operator turn
    # before it is the boundary. Requiring prior speech keeps an operator's
    # opening housekeeping from being mistaken for the Q&A.
    if qa_start is None:
        heard: set[str] = set()
        for i, seg in enumerate(segments):
            if seg.speaker.lower() != "operator":
                heard.add(_last_name(seg.speaker))
                continue
            if not heard:
                continue
            following = next(
                (s for s in segments[i + 1:] if s.speaker.lower() != "operator"),
                None,
            )
            if following is not None and _last_name(following.speaker) not in heard:
                qa_start = i
                break

    # Last resort: a call with no operator at all.
    #
    # Some issuers run their own Q&A β€” an investor-relations host reads the
    # questions and there is never an "Operator" turn to key off. Tesla's calls
    # are the standing example. What still holds is the shape: prepared remarks
    # from a handful of known voices, then a voice that has not been heard yet
    # asking something. The question mark is what separates that from a second
    # executive joining the prepared remarks.
    if qa_start is None:
        heard = set()
        for i, seg in enumerate(segments):
            name = _last_name(seg.speaker)
            # Two prior voices β‰ˆ host plus at least one executive, i.e. the
            # prepared section is genuinely under way.
            if len(heard) >= 2 and name not in heard and "?" in seg.text:
                qa_start = i
                break
            heard.add(name)

    # Analyst roster from Operator hand-offs (anywhere in the call).
    roster: set[str] = set()
    for seg in segments:
        if seg.speaker.lower() == "operator":
            seg.role = "operator"
            for m in _ANALYST_INTRO_RE.finditer(seg.text):
                roster.add(_last_name(m.group(1)))

    # Management = non-operator speakers heard before the Q&A boundary.
    pre_qa_end = qa_start if qa_start is not None else len(segments)
    management: set[str] = {
        _last_name(seg.speaker)
        for seg in segments[:pre_qa_end]
        if seg.role not in ("operator", "analyst") and seg.speaker
    }

    # Assign roles.
    prev_role = ""
    for i, seg in enumerate(segments):
        if seg.role in ("operator", "analyst"):
            prev_role = seg.role
            continue
        key = _last_name(seg.speaker)
        if key in management:
            seg.role = "management"
        elif key in roster:
            seg.role = "analyst"
        elif qa_start is not None and i > qa_start:
            # Unknown speaker in Q&A: question-shaped or operator hand-off β†’ analyst.
            if seg.text.rstrip().endswith("?") or prev_role == "operator":
                seg.role = "analyst"
            else:
                seg.role = "management"
        else:
            seg.role = "management"
        prev_role = seg.role

    prepared_parts = [
        seg.text for seg in segments[:pre_qa_end] if seg.role == "management"
    ]
    management_parts = [seg.text for seg in segments if seg.role == "management"]
    answer_parts = [
        seg.text for seg in segments[pre_qa_end:] if seg.role == "management"
    ]

    # Pair analyst turns with the management turns that follow them.
    qa: list[QAExchange] = []
    if qa_start is not None:
        current: QAExchange | None = None
        for seg in segments[qa_start:]:
            if seg.role == "analyst":
                if current is None or current.answer:
                    if current is not None and current.answer:
                        qa.append(current)
                    current = QAExchange(analyst=seg.speaker, question=seg.text, answer="")
                else:
                    current.question += " " + seg.text
            elif seg.role == "management" and current is not None:
                current.answer = (current.answer + " " + seg.text).strip()
            elif seg.role == "operator" and current is not None and current.answer:
                qa.append(current)
                current = None
        if current is not None and current.answer:
            qa.append(current)

    return ParsedCall(
        period=period,
        prepared_text="\n".join(prepared_parts).strip(),
        management_text="\n".join(management_parts).strip(),
        qa=qa,
        n_segments=len(segments),
        answers_text="\n".join(answer_parts).strip(),
        speakers=_ordered_speakers(segments),
    )