lamossta commited on
Commit
0150291
·
1 Parent(s): b12b2c4

data analysis notebooks

Browse files
notebooks/data_augmentation_analysis.ipynb ADDED
The diff for this file is too large to render. See raw diff
 
notebooks/data_preprocessing_analysis.ipynb ADDED
@@ -0,0 +1,1103 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "markdown",
5
+ "metadata": {},
6
+ "source": [
7
+ "# Data Preprocessing Analysis\n",
8
+ "\n",
9
+ "Data hygiene checks for entity-level sentiment classification.\n",
10
+ "\n",
11
+ "**Structure**\n",
12
+ "1. Load Data\n",
13
+ "2. Dataset Overview\n",
14
+ "3. Duplicate Texts\n",
15
+ "4. Entity Deduplication Pipeline\n",
16
+ "5. Label Validity\n",
17
+ "6. Position Text Mismatches\n",
18
+ "7. HTML Tags\n",
19
+ "8. Preprocessing Summary"
20
+ ],
21
+ "id": "8cdbbbb3574744ab"
22
+ },
23
+ {
24
+ "cell_type": "code",
25
+ "metadata": {
26
+ "ExecuteTime": {
27
+ "end_time": "2026-04-18T19:53:18.500521Z",
28
+ "start_time": "2026-04-18T19:53:18.495644Z"
29
+ }
30
+ },
31
+ "source": [
32
+ "import os\n",
33
+ "import json\n",
34
+ "import warnings\n",
35
+ "from collections import Counter\n",
36
+ "\n",
37
+ "import numpy as np\n",
38
+ "import pandas as pd\n",
39
+ "import matplotlib.pyplot as plt\n",
40
+ "import matplotlib.patches as mpatches\n",
41
+ "import seaborn as sns\n",
42
+ "\n",
43
+ "warnings.filterwarnings(\"ignore\")\n",
44
+ "sns.set_theme(style=\"whitegrid\", palette=\"muted\")\n",
45
+ "plt.rcParams[\"figure.dpi\"] = 120\n",
46
+ "plt.rcParams[\"axes.titlesize\"] = 13\n",
47
+ "plt.rcParams[\"axes.labelsize\"] = 11\n",
48
+ "\n",
49
+ "VALID_LABELS = {\"positive\", \"neutral\", \"negative\"}\n",
50
+ "LABEL_ORDER = [\"negative\", \"neutral\", \"positive\"]\n",
51
+ "LABEL_COLORS = {\"positive\": \"#4CAF50\", \"neutral\": \"#90A4AE\", \"negative\": \"#EF5350\"}"
52
+ ],
53
+ "id": "32a9ac93d2dec54",
54
+ "outputs": [],
55
+ "execution_count": 13
56
+ },
57
+ {
58
+ "cell_type": "markdown",
59
+ "metadata": {},
60
+ "source": [
61
+ "## 1. Load Data"
62
+ ],
63
+ "id": "5eabfa4c06905b4"
64
+ },
65
+ {
66
+ "cell_type": "code",
67
+ "metadata": {
68
+ "ExecuteTime": {
69
+ "end_time": "2026-04-18T19:53:18.574483Z",
70
+ "start_time": "2026-04-18T19:53:18.515917Z"
71
+ }
72
+ },
73
+ "source": [
74
+ "data_path = os.path.join(\"..\", \"data\", \"data_raw.json\")\n",
75
+ "with open(data_path, \"r\", encoding=\"utf-8\") as f:\n",
76
+ " data_raw = json.load(f)\n",
77
+ "\n",
78
+ "# Flat article-level dataframe\n",
79
+ "df_articles = pd.DataFrame([\n",
80
+ " {\"id\": s[\"id\"], \"text\": s[\"text\"], \"n_entities\": len(s[\"entities\"])}\n",
81
+ " for s in data_raw\n",
82
+ "])\n",
83
+ "\n",
84
+ "# Flat entity-level dataframe\n",
85
+ "df_entities = pd.DataFrame([\n",
86
+ " {\n",
87
+ " \"sample_id\": s[\"id\"],\n",
88
+ " \"entity_id\": e[\"entity_id\"],\n",
89
+ " \"entity_text\": e[\"entity_text\"],\n",
90
+ " \"entity_type\": e[\"entity_type\"],\n",
91
+ " \"label\": e[\"label\"],\n",
92
+ " \"n_positions\": len(e[\"positions\"]),\n",
93
+ " }\n",
94
+ " for s in data_raw\n",
95
+ " for e in s[\"entities\"]\n",
96
+ "])\n",
97
+ "\n",
98
+ "print(f\"Articles : {len(df_articles):,}\")\n",
99
+ "print(f\"Entities : {len(df_entities):,}\")\n",
100
+ "print(f\"Avg entities per article: {len(df_entities)/len(df_articles):.2f}\")\n",
101
+ "df_articles.head()"
102
+ ],
103
+ "id": "463eb928d9be91b0",
104
+ "outputs": [
105
+ {
106
+ "name": "stdout",
107
+ "output_type": "stream",
108
+ "text": [
109
+ "Articles : 1,637\n",
110
+ "Entities : 12,627\n",
111
+ "Avg entities per article: 7.71\n"
112
+ ]
113
+ },
114
+ {
115
+ "data": {
116
+ "text/plain": [
117
+ " id text n_entities\n",
118
+ "0 0 Amazon sold unauthorized mole removers, and th... 4\n",
119
+ "1 1 Russia Announces Response Measures to New US S... 11\n",
120
+ "2 2 India's richest man takes on Amazon, Walmart i... 12\n",
121
+ "3 3 Govt may impose anti-dumping duty on chemical ... 10\n",
122
+ "4 4 Apollo Go: AI-Powered Autonomous Ride-Hailing ... 13"
123
+ ],
124
+ "text/html": [
125
+ "<div>\n",
126
+ "<style scoped>\n",
127
+ " .dataframe tbody tr th:only-of-type {\n",
128
+ " vertical-align: middle;\n",
129
+ " }\n",
130
+ "\n",
131
+ " .dataframe tbody tr th {\n",
132
+ " vertical-align: top;\n",
133
+ " }\n",
134
+ "\n",
135
+ " .dataframe thead th {\n",
136
+ " text-align: right;\n",
137
+ " }\n",
138
+ "</style>\n",
139
+ "<table border=\"1\" class=\"dataframe\">\n",
140
+ " <thead>\n",
141
+ " <tr style=\"text-align: right;\">\n",
142
+ " <th></th>\n",
143
+ " <th>id</th>\n",
144
+ " <th>text</th>\n",
145
+ " <th>n_entities</th>\n",
146
+ " </tr>\n",
147
+ " </thead>\n",
148
+ " <tbody>\n",
149
+ " <tr>\n",
150
+ " <th>0</th>\n",
151
+ " <td>0</td>\n",
152
+ " <td>Amazon sold unauthorized mole removers, and th...</td>\n",
153
+ " <td>4</td>\n",
154
+ " </tr>\n",
155
+ " <tr>\n",
156
+ " <th>1</th>\n",
157
+ " <td>1</td>\n",
158
+ " <td>Russia Announces Response Measures to New US S...</td>\n",
159
+ " <td>11</td>\n",
160
+ " </tr>\n",
161
+ " <tr>\n",
162
+ " <th>2</th>\n",
163
+ " <td>2</td>\n",
164
+ " <td>India's richest man takes on Amazon, Walmart i...</td>\n",
165
+ " <td>12</td>\n",
166
+ " </tr>\n",
167
+ " <tr>\n",
168
+ " <th>3</th>\n",
169
+ " <td>3</td>\n",
170
+ " <td>Govt may impose anti-dumping duty on chemical ...</td>\n",
171
+ " <td>10</td>\n",
172
+ " </tr>\n",
173
+ " <tr>\n",
174
+ " <th>4</th>\n",
175
+ " <td>4</td>\n",
176
+ " <td>Apollo Go: AI-Powered Autonomous Ride-Hailing ...</td>\n",
177
+ " <td>13</td>\n",
178
+ " </tr>\n",
179
+ " </tbody>\n",
180
+ "</table>\n",
181
+ "</div>"
182
+ ]
183
+ },
184
+ "execution_count": 14,
185
+ "metadata": {},
186
+ "output_type": "execute_result"
187
+ }
188
+ ],
189
+ "execution_count": 14
190
+ },
191
+ {
192
+ "cell_type": "markdown",
193
+ "metadata": {},
194
+ "source": [
195
+ "## 2. Dataset Overview"
196
+ ],
197
+ "id": "6812878d46314b80"
198
+ },
199
+ {
200
+ "cell_type": "code",
201
+ "metadata": {
202
+ "ExecuteTime": {
203
+ "end_time": "2026-04-18T19:53:18.601499Z",
204
+ "start_time": "2026-04-18T19:53:18.597184Z"
205
+ }
206
+ },
207
+ "source": [
208
+ "print(\"=\" * 45)\n",
209
+ "print(\"DATASET OVERVIEW\")\n",
210
+ "print(\"=\" * 45)\n",
211
+ "print(f\" Total samples : {len(data_raw):,}\")\n",
212
+ "print(f\" Total entities : {len(df_entities):,}\")\n",
213
+ "print(f\" Unique entity texts : {df_entities['entity_text'].nunique():,}\")\n",
214
+ "print(f\" Unique entity types : {df_entities['entity_type'].nunique()}\")\n",
215
+ "print()\n",
216
+ "print(\"Entity type breakdown:\")\n",
217
+ "for etype, cnt in df_entities[\"entity_type\"].value_counts().items():\n",
218
+ " print(f\" {etype:12s}: {cnt:6,} ({cnt / len(df_entities) * 100:.1f}%)\")"
219
+ ],
220
+ "id": "c7f3929c5b35b77c",
221
+ "outputs": [
222
+ {
223
+ "name": "stdout",
224
+ "output_type": "stream",
225
+ "text": [
226
+ "=============================================\n",
227
+ "DATASET OVERVIEW\n",
228
+ "=============================================\n",
229
+ " Total samples : 1,637\n",
230
+ " Total entities : 12,627\n",
231
+ " Unique entity texts : 5,815\n",
232
+ " Unique entity types : 2\n",
233
+ "\n",
234
+ "Entity type breakdown:\n",
235
+ " company : 10,351 (82.0%)\n",
236
+ " location : 2,276 (18.0%)\n"
237
+ ]
238
+ }
239
+ ],
240
+ "execution_count": 15
241
+ },
242
+ {
243
+ "cell_type": "markdown",
244
+ "metadata": {},
245
+ "source": [
246
+ "## 3. Duplicate Article Texts\n",
247
+ "\n",
248
+ "Two samples sharing the same article text would allow the model to see identical context\n",
249
+ "in both training and evaluation, artificially inflating metrics."
250
+ ],
251
+ "id": "7623b7ffc328b9a7"
252
+ },
253
+ {
254
+ "cell_type": "code",
255
+ "metadata": {
256
+ "ExecuteTime": {
257
+ "end_time": "2026-04-18T19:53:18.626539Z",
258
+ "start_time": "2026-04-18T19:53:18.621684Z"
259
+ }
260
+ },
261
+ "source": [
262
+ "dup_mask = df_articles.duplicated(\"text\", keep=False)\n",
263
+ "dup_articles = df_articles[dup_mask]\n",
264
+ "\n",
265
+ "print(f\"Duplicate article texts: {len(dup_articles)}\")\n",
266
+ "\n",
267
+ "if len(dup_articles) > 0:\n",
268
+ " print(dup_articles.head(10))"
269
+ ],
270
+ "id": "17313e6496ece22",
271
+ "outputs": [
272
+ {
273
+ "name": "stdout",
274
+ "output_type": "stream",
275
+ "text": [
276
+ "Duplicate article texts: 16\n",
277
+ " id text n_entities\n",
278
+ "48 48 Elon Musk buys Twitter for $44B and will priva... 13\n",
279
+ "254 254 Judge tosses Biden threats case against northe... 12\n",
280
+ "265 265 Dozens feared dead as Russian shell hits Ukrai... 9\n",
281
+ "322 322 US weekly jobless claims unexpectedly fall\\nWA... 4\n",
282
+ "393 393 Yemen's rebels launch drone and missile strike... 9\n",
283
+ "610 610 Dozens feared dead as Russian shell hits Ukrai... 8\n",
284
+ "667 667 Elon Musk buys Twitter for $44B and will priva... 14\n",
285
+ "975 975 Mercado Pago anuncia 'conta que mais rende no ... 3\n",
286
+ "1019 1019 Yemen's rebels launch drone and missile strike... 8\n",
287
+ "1166 1166 US weekly jobless claims unexpectedly fall\\nWA... 4\n"
288
+ ]
289
+ }
290
+ ],
291
+ "execution_count": 16
292
+ },
293
+ {
294
+ "cell_type": "markdown",
295
+ "metadata": {},
296
+ "source": [
297
+ "## 4. Entity Deduplication Pipeline\n",
298
+ "\n",
299
+ "Within each sample, multiple entity records can refer to the same real-world entity\n",
300
+ "with overlapping or identical position spans. The pipeline below merges and cleans\n",
301
+ "these in four steps:\n",
302
+ "\n",
303
+ "1. **Merge entity records** by `(entity_text, label)` → all positions for that entity in one list\n",
304
+ "2. **Deduplicate exact-duplicate positions** (same offset + length)\n",
305
+ "3. **Resolve overlapping positions** (same offset, different length → keep longest)\n",
306
+ "4. **Handle partial overlaps** (rare edge cases)"
307
+ ],
308
+ "id": "612bfc4a2968aa02"
309
+ },
310
+ {
311
+ "cell_type": "markdown",
312
+ "metadata": {},
313
+ "source": [
314
+ "### Step 1 — Merge entity records by `(entity_text, label)`\n",
315
+ "\n",
316
+ "Multiple entity records with the same `entity_text` and `label` inside one sample\n",
317
+ "are merged into a single entity with a unified position list. Conflicting labels\n",
318
+ "for the same entity text are flagged as annotation errors."
319
+ ],
320
+ "id": "7765b7af17c9b50d"
321
+ },
322
+ {
323
+ "cell_type": "code",
324
+ "metadata": {
325
+ "ExecuteTime": {
326
+ "end_time": "2026-04-18T19:53:18.656417Z",
327
+ "start_time": "2026-04-18T19:53:18.650897Z"
328
+ }
329
+ },
330
+ "source": [
331
+ "merge_stats = {\"merged_entities\": 0, \"different_label\": 0, \"affected_samples\": set()}\n",
332
+ "conflict_examples = []\n",
333
+ "\n",
334
+ "for s in data_raw:\n",
335
+ " seen = {}\n",
336
+ " for e in s[\"entities\"]:\n",
337
+ " key = e[\"entity_text\"].lower()\n",
338
+ " if key in seen:\n",
339
+ " merge_stats[\"affected_samples\"].add(s[\"id\"])\n",
340
+ " if seen[key] == e[\"label\"]:\n",
341
+ " merge_stats[\"merged_entities\"] += 1\n",
342
+ " else:\n",
343
+ " merge_stats[\"different_label\"] += 1\n",
344
+ " conflict_examples.append({\n",
345
+ " \"sample_id\": s[\"id\"],\n",
346
+ " \"entity_text\": e[\"entity_text\"],\n",
347
+ " \"labels\": [seen[key], e[\"label\"]],\n",
348
+ " })\n",
349
+ " else:\n",
350
+ " seen[key] = e[\"label\"]\n",
351
+ "\n",
352
+ "total_merged = merge_stats[\"merged_entities\"] + merge_stats[\"different_label\"]\n",
353
+ "print(\"Step 1: Merge entity records by (entity_text, label)\")\n",
354
+ "print(\"=\" * 55)\n",
355
+ "print(f\"Duplicate (entity_text, sample) pairs: {total_merged:,}\")\n",
356
+ "print(f\"Same label: {merge_stats['merged_entities']:,}\")\n",
357
+ "print(f\"Different_labels: {merge_stats['different_label']:,}\")\n",
358
+ "print(f\"Affected samples: {len(merge_stats['affected_samples']):,}\")\n",
359
+ "print()\n",
360
+ "\n",
361
+ "if merge_stats[\"different_label\"] == 0:\n",
362
+ " print(\"All duplicates have consistent labels.\")\n",
363
+ " print(f\"Merging will consolidate {total_merged:,} redundant entity records ({total_merged / len(df_entities) * 100:.1f}% of total).\")\n",
364
+ "else:\n",
365
+ " print(f\"{merge_stats['different_label']} different-label duplicates.\")\n",
366
+ " print(pd.DataFrame(conflict_examples).head())"
367
+ ],
368
+ "id": "e58872333d152f15",
369
+ "outputs": [
370
+ {
371
+ "name": "stdout",
372
+ "output_type": "stream",
373
+ "text": [
374
+ "Step 1: Merge entity records by (entity_text, label)\n",
375
+ "=======================================================\n",
376
+ "Duplicate (entity_text, sample) pairs: 2,014\n",
377
+ "Same label: 2,014\n",
378
+ "Different_labels: 0\n",
379
+ "Affected samples: 838\n",
380
+ "\n",
381
+ "All duplicates have consistent labels.\n",
382
+ "Merging will consolidate 2,014 redundant entity records (15.9% of total).\n"
383
+ ]
384
+ }
385
+ ],
386
+ "execution_count": 17
387
+ },
388
+ {
389
+ "cell_type": "markdown",
390
+ "metadata": {},
391
+ "source": [
392
+ "### Step 2 — Deduplicate exact-duplicate positions\n",
393
+ "\n",
394
+ "After merging, a single entity may carry duplicate position entries (identical offset\n",
395
+ "and length). These are removed, keeping one copy per unique span."
396
+ ],
397
+ "id": "b0eb6a25890d61b6"
398
+ },
399
+ {
400
+ "cell_type": "code",
401
+ "metadata": {
402
+ "ExecuteTime": {
403
+ "end_time": "2026-04-18T19:53:18.694440Z",
404
+ "start_time": "2026-04-18T19:53:18.675611Z"
405
+ }
406
+ },
407
+ "source": [
408
+ "exact_dup_positions = []\n",
409
+ "\n",
410
+ "for s in data_raw:\n",
411
+ " merged = {}\n",
412
+ " for e in s[\"entities\"]:\n",
413
+ " key = (e[\"entity_text\"].lower(), e[\"label\"])\n",
414
+ " if key not in merged:\n",
415
+ " merged[key] = {\"entity_text\": e[\"entity_text\"], \"label\": e[\"label\"], \"positions\": []}\n",
416
+ " merged[key][\"positions\"].extend(e[\"positions\"])\n",
417
+ "\n",
418
+ " for key, ent in merged.items():\n",
419
+ " seen_spans = set()\n",
420
+ " for p in ent[\"positions\"]:\n",
421
+ " span = (p[\"offset\"], p[\"length\"])\n",
422
+ " if span in seen_spans:\n",
423
+ " exact_dup_positions.append({\n",
424
+ " \"sample_id\": s[\"id\"],\n",
425
+ " \"entity_text\": ent[\"entity_text\"],\n",
426
+ " \"position_text\": p[\"position_text\"],\n",
427
+ " \"offset\": p[\"offset\"],\n",
428
+ " \"length\": p[\"length\"],\n",
429
+ " })\n",
430
+ " else:\n",
431
+ " seen_spans.add(span)\n",
432
+ "\n",
433
+ "print(\"Step 2: Deduplicate exact-duplicate positions\")\n",
434
+ "print(\"=\" * 55)\n",
435
+ "print(f\"Exact-duplicate positions found: {len(exact_dup_positions):,}\")\n",
436
+ "if exact_dup_positions:\n",
437
+ " df_exact_dup = pd.DataFrame(exact_dup_positions)\n",
438
+ " n_samples = df_exact_dup[\"sample_id\"].nunique()\n",
439
+ " print(f\"Across {n_samples:,} sample(s)\")\n",
440
+ " print()\n",
441
+ " print(df_exact_dup.head(20).to_string(index=False))\n",
442
+ "else:\n",
443
+ " print(\"No exact-duplicate positions after merging.\")"
444
+ ],
445
+ "id": "d5acf8bf642c044",
446
+ "outputs": [
447
+ {
448
+ "name": "stdout",
449
+ "output_type": "stream",
450
+ "text": [
451
+ "Step 2: Deduplicate exact-duplicate positions\n",
452
+ "=======================================================\n",
453
+ "Exact-duplicate positions found: 9,634\n",
454
+ "Across 938 sample(s)\n",
455
+ "\n",
456
+ " sample_id entity_text position_text offset length\n",
457
+ " 0 Verge Verge 2082 5\n",
458
+ " 1 SolarWinds SolarWinds 271 10\n",
459
+ " 1 SolarWinds SolarWinds 2767 10\n",
460
+ " 2 Grofers Grofers 2184 7\n",
461
+ " 3 Reliance industries Ltd Reliance industries Ltd 70 23\n",
462
+ " 3 Reliance industries Ltd Reliance industries Ltd 630 23\n",
463
+ " 3 Reliance industries Ltd Reliance industries Ltd 70 23\n",
464
+ " 3 Reliance industries Ltd Reliance 70 8\n",
465
+ " 3 Reliance industries Ltd Reliance industries Ltd 630 23\n",
466
+ " 3 Reliance industries Ltd Reliance 630 8\n",
467
+ " 3 India Glycols Limited India Glycols Limited 924 21\n",
468
+ " 3 India Glycols Limited India Glycols Limited 924 21\n",
469
+ " 4 Baidu Baidu 459 5\n",
470
+ " 4 Baidu Baidu 980 5\n",
471
+ " 4 Baidu Baidu 1289 5\n",
472
+ " 4 Baidu Baidu 2915 5\n",
473
+ " 4 Baidu Baidu 4180 5\n",
474
+ " 4 global industries global industries 175 17\n",
475
+ " 5 Nokian Tyres plc Nokian Tyres plc 51 16\n",
476
+ " 5 Nokian Tyres plc Nokian Tyres plc 51 16\n"
477
+ ]
478
+ }
479
+ ],
480
+ "execution_count": 18
481
+ },
482
+ {
483
+ "cell_type": "markdown",
484
+ "metadata": {},
485
+ "source": [
486
+ "### Step 3 — Resolve overlapping positions (same offset, different length)\n",
487
+ "\n",
488
+ "Two positions for the same entity starting at the same offset but with different\n",
489
+ "lengths represent the same mention captured at different granularity.\n",
490
+ "Resolution: keep the longest span (it captures the full entity mention)."
491
+ ],
492
+ "id": "31dd5cc13f612c0b"
493
+ },
494
+ {
495
+ "cell_type": "code",
496
+ "metadata": {
497
+ "ExecuteTime": {
498
+ "end_time": "2026-04-18T19:53:18.721435Z",
499
+ "start_time": "2026-04-18T19:53:18.703478Z"
500
+ }
501
+ },
502
+ "source": [
503
+ "same_offset_diff_len = []\n",
504
+ "\n",
505
+ "for s in data_raw:\n",
506
+ " merged = {}\n",
507
+ " for e in s[\"entities\"]:\n",
508
+ " key = (e[\"entity_text\"].lower(), e[\"label\"])\n",
509
+ " if key not in merged:\n",
510
+ " merged[key] = {\"entity_text\": e[\"entity_text\"], \"positions\": []}\n",
511
+ " merged[key][\"positions\"].extend(e[\"positions\"])\n",
512
+ "\n",
513
+ " for key, ent in merged.items():\n",
514
+ " # Deduplicate exact spans first\n",
515
+ " unique_positions = {}\n",
516
+ " for p in ent[\"positions\"]:\n",
517
+ " span = (p[\"offset\"], p[\"length\"])\n",
518
+ " if span not in unique_positions:\n",
519
+ " unique_positions[span] = p\n",
520
+ "\n",
521
+ " # Group by offset\n",
522
+ " by_offset = {}\n",
523
+ " for (off, length), p in unique_positions.items():\n",
524
+ " by_offset.setdefault(off, []).append(p)\n",
525
+ "\n",
526
+ " for off, positions in by_offset.items():\n",
527
+ " if len(positions) > 1:\n",
528
+ " positions.sort(key=lambda p: p[\"length\"], reverse=True)\n",
529
+ " keep = positions[0]\n",
530
+ " for discard in positions[1:]:\n",
531
+ " same_offset_diff_len.append({\n",
532
+ " \"sample_id\": s[\"id\"],\n",
533
+ " \"entity_text\": ent[\"entity_text\"],\n",
534
+ " \"keep_text\": keep[\"position_text\"],\n",
535
+ " \"keep_span\": f'[{keep[\"offset\"]}:{keep[\"offset\"]+keep[\"length\"]})',\n",
536
+ " \"discard_text\": discard[\"position_text\"],\n",
537
+ " \"discard_span\": f'[{discard[\"offset\"]}:{discard[\"offset\"]+discard[\"length\"]})',\n",
538
+ " })\n",
539
+ "\n",
540
+ "print(\"Step 3: Resolve same-offset, different-length positions\")\n",
541
+ "print(\"=\" * 55)\n",
542
+ "print(f\"Cases found: {len(same_offset_diff_len):,}\")\n",
543
+ "if same_offset_diff_len:\n",
544
+ " df_same_off = pd.DataFrame(same_offset_diff_len)\n",
545
+ " n_samples = df_same_off[\"sample_id\"].nunique()\n",
546
+ " print(f\"Across {n_samples:,} sample(s)\")\n",
547
+ " print()\n",
548
+ " print(df_same_off.head(10).to_string(index=False))\n",
549
+ "else:\n",
550
+ " print(\"No same-offset length conflicts after deduplication.\")"
551
+ ],
552
+ "id": "2ec34f05729dd40c",
553
+ "outputs": [
554
+ {
555
+ "name": "stdout",
556
+ "output_type": "stream",
557
+ "text": [
558
+ "Step 3: Resolve same-offset, different-length positions\n",
559
+ "=======================================================\n",
560
+ "Cases found: 805\n",
561
+ "Across 377 sample(s)\n",
562
+ "\n",
563
+ " sample_id entity_text keep_text keep_span discard_text discard_span\n",
564
+ " 3 Reliance industries Ltd Reliance industries Ltd [70:93) Reliance [70:78)\n",
565
+ " 3 Reliance industries Ltd Reliance industries Ltd [630:653) Reliance [630:638)\n",
566
+ " 7 Charles Schwab Charles Schwab Corporation [171:197) Charles Schwab [171:185)\n",
567
+ " 7 Charles Schwab Charles Schwab's [823:839) Charles Schwab [823:837)\n",
568
+ " 7 Charles Schwab Charles Schwab's [1208:1224) Charles Schwab [1208:1222)\n",
569
+ " 7 Charles Schwab Charles Schwab's [1404:1420) Charles Schwab [1404:1418)\n",
570
+ " 10 Cracker Barrel Cracker Barrel Old Country Store, Inc [130:167) Cracker Barrel Old Country Store [130:162)\n",
571
+ " 10 Cracker Barrel Cracker Barrel Old Country Store, Inc [130:167) Cracker Barrel [130:144)\n",
572
+ " 14 American Water Works Association American Water Works Association [1299:1331) American Water [1299:1313)\n",
573
+ " 15 Sainsbury's Sainsbury's [232:243) Sainsbury [232:241)\n"
574
+ ]
575
+ }
576
+ ],
577
+ "execution_count": 19
578
+ },
579
+ {
580
+ "cell_type": "markdown",
581
+ "metadata": {},
582
+ "source": [
583
+ "### Step 4 — Handle partial overlaps\n",
584
+ "\n",
585
+ "Remaining overlaps where two positions for the same entity share some characters\n",
586
+ "but start at different offsets. These are rare and may indicate tokenisation\n",
587
+ "differences in the original annotation."
588
+ ],
589
+ "id": "abdbd0fa2842058c"
590
+ },
591
+ {
592
+ "cell_type": "code",
593
+ "metadata": {
594
+ "ExecuteTime": {
595
+ "end_time": "2026-04-18T19:53:18.748741Z",
596
+ "start_time": "2026-04-18T19:53:18.726218Z"
597
+ }
598
+ },
599
+ "source": [
600
+ "partial_overlaps = []\n",
601
+ "\n",
602
+ "for s in data_raw:\n",
603
+ " merged = {}\n",
604
+ " for e in s[\"entities\"]:\n",
605
+ " key = (e[\"entity_text\"].lower(), e[\"label\"])\n",
606
+ " if key not in merged:\n",
607
+ " merged[key] = {\"entity_text\": e[\"entity_text\"], \"positions\": []}\n",
608
+ " merged[key][\"positions\"].extend(e[\"positions\"])\n",
609
+ "\n",
610
+ " for key, ent in merged.items():\n",
611
+ " # Deduplicate exact spans\n",
612
+ " unique_positions = {}\n",
613
+ " for p in ent[\"positions\"]:\n",
614
+ " span = (p[\"offset\"], p[\"length\"])\n",
615
+ " if span not in unique_positions:\n",
616
+ " unique_positions[span] = p\n",
617
+ "\n",
618
+ " # Resolve same-offset conflicts (keep longest)\n",
619
+ " by_offset = {}\n",
620
+ " for (off, length), p in unique_positions.items():\n",
621
+ " if off not in by_offset or length > by_offset[off][\"length\"]:\n",
622
+ " by_offset[off] = p\n",
623
+ " resolved = sorted(by_offset.values(), key=lambda p: p[\"offset\"])\n",
624
+ "\n",
625
+ " # Check remaining partial overlaps\n",
626
+ " for i in range(len(resolved)):\n",
627
+ " for j in range(i + 1, len(resolved)):\n",
628
+ " si_off = resolved[i][\"offset\"]\n",
629
+ " si_end = si_off + resolved[i][\"length\"]\n",
630
+ " sj_off = resolved[j][\"offset\"]\n",
631
+ " sj_end = sj_off + resolved[j][\"length\"]\n",
632
+ " if sj_off >= si_end:\n",
633
+ " break\n",
634
+ " partial_overlaps.append({\n",
635
+ " \"sample_id\": s[\"id\"],\n",
636
+ " \"entity_text\": ent[\"entity_text\"],\n",
637
+ " \"position_text_a\": resolved[i][\"position_text\"],\n",
638
+ " \"position_text_b\": resolved[j][\"position_text\"],\n",
639
+ " \"span_a\": f'[{si_off}:{si_end})',\n",
640
+ " \"span_b\": f'[{sj_off}:{sj_end})',\n",
641
+ " \"overlap_chars\": si_end - sj_off,\n",
642
+ " })\n",
643
+ "\n",
644
+ "print(\"Step 4: Handle partial overlaps\")\n",
645
+ "print(\"=\" * 55)\n",
646
+ "print(f\"Partial overlaps remaining: {len(partial_overlaps):,}\")\n",
647
+ "if partial_overlaps:\n",
648
+ " df_partial = pd.DataFrame(partial_overlaps)\n",
649
+ " n_samples = df_partial[\"sample_id\"].nunique()\n",
650
+ " print(f\"Across {n_samples:,} sample(s)\")\n",
651
+ " print()\n",
652
+ " print(df_partial.to_string(index=False))\n",
653
+ "else:\n",
654
+ " print(\"No partial overlaps remain.\")"
655
+ ],
656
+ "id": "af1fb86847fd23b3",
657
+ "outputs": [
658
+ {
659
+ "name": "stdout",
660
+ "output_type": "stream",
661
+ "text": [
662
+ "Step 4: Handle partial overlaps\n",
663
+ "=======================================================\n",
664
+ "Partial overlaps remaining: 19\n",
665
+ "Across 11 sample(s)\n",
666
+ "\n",
667
+ " sample_id entity_text position_text_a position_text_b span_a span_b overlap_chars\n",
668
+ " 35 The Walt Disney Company The Walt Disney Company Walt Disney Company's [0:23) [4:25) 19\n",
669
+ " 35 The Walt Disney Company The Walt Disney Company Walt Disney Company's [136:159) [140:161) 19\n",
670
+ " 35 The Walt Disney Company The Walt Disney Company Walt Disney Company's [462:485) [466:487) 19\n",
671
+ " 147 Shell Royal Dutch Shell's Shell [154:173) [166:171) 7\n",
672
+ " 147 Shell Royal Dutch Shell Shell [479:496) [491:496) 5\n",
673
+ " 397 the Walt Disney Company the Walt Disney Company Walt Disney Company [427:450) [431:450) 19\n",
674
+ " 397 Disney Walt Disney Co Disney [135:149) [140:146) 9\n",
675
+ " 397 Disney Walt Disney Disney [916:927) [921:927) 6\n",
676
+ " 537 Royal Dutch Shell Royal Dutch Shell Shell [991:1008) [1003:1008) 5\n",
677
+ " 548 Shell Royal Dutch Shell PLC Shell [1340:1361) [1352:1357) 9\n",
678
+ " 886 Disney Walt Disney Co Disney [70:84) [75:81) 9\n",
679
+ " 925 Royal Dutch Shell Royal Dutch Shell Shell [2720:2737) [2732:2737) 5\n",
680
+ " 938 Disney Walt Disney Disney [192:203) [197:203) 6\n",
681
+ " 1050 Disney The Walt Disney Company Walt Disney Company [61:84) [65:84) 19\n",
682
+ " 1050 Disney The Walt Disney Company Disney [61:84) [70:76) 14\n",
683
+ " 1050 Disney Walt Disney Company Disney [65:84) [70:76) 14\n",
684
+ " 1050 The Walt Disney Company The Walt Disney Company Walt Disney Company [61:84) [65:84) 19\n",
685
+ " 1400 Royal Dutch Shell Plc Royal Dutch Shell Plc Shell [2864:2885) [2876:2881) 9\n",
686
+ " 1410 Vale PT Vale Indonesia Vale [241:258) [244:248) 14\n"
687
+ ]
688
+ }
689
+ ],
690
+ "execution_count": 20
691
+ },
692
+ {
693
+ "cell_type": "markdown",
694
+ "metadata": {},
695
+ "source": [
696
+ "## 5. Label Validity\n",
697
+ "\n",
698
+ "The schema defines three valid labels: `positive`, `neutral`, `negative`.\n",
699
+ "Any other value is an annotation error. We check for remappable near-matches\n",
700
+ "(e.g. `very positive`) that can be salvaged rather than dropped."
701
+ ],
702
+ "id": "fee7fd592186f064"
703
+ },
704
+ {
705
+ "cell_type": "code",
706
+ "metadata": {
707
+ "ExecuteTime": {
708
+ "end_time": "2026-04-18T19:53:18.765731Z",
709
+ "start_time": "2026-04-18T19:53:18.762683Z"
710
+ }
711
+ },
712
+ "source": [
713
+ "label_counts = df_entities[\"label\"].value_counts()\n",
714
+ "invalid = label_counts[~label_counts.index.isin(VALID_LABELS)]\n",
715
+ "\n",
716
+ "print(\"All label values and counts:\")\n",
717
+ "print(label_counts.to_frame(\"count\").to_string())\n",
718
+ "print()\n",
719
+ "\n",
720
+ "if len(invalid) == 0:\n",
721
+ " print(\"All labels are valid.\")"
722
+ ],
723
+ "id": "123db84848d0be7f",
724
+ "outputs": [
725
+ {
726
+ "name": "stdout",
727
+ "output_type": "stream",
728
+ "text": [
729
+ "All label values and counts:\n",
730
+ " count\n",
731
+ "label \n",
732
+ "neutral 8400\n",
733
+ "negative 2148\n",
734
+ "positive 2078\n",
735
+ "very positive 1\n",
736
+ "\n"
737
+ ]
738
+ }
739
+ ],
740
+ "execution_count": 21
741
+ },
742
+ {
743
+ "cell_type": "markdown",
744
+ "metadata": {},
745
+ "source": [
746
+ "## 6. Position Text Mismatches\n",
747
+ "\n",
748
+ "Each position stores `position_text`, `offset`, and `length`. The actual span in\n",
749
+ "the article is `text[offset : offset+length]`. We check whether these agree.\n",
750
+ "\n",
751
+ "- **Case-only mismatch** (`BRUSSELS` vs `Brussels`): offset is correct, stored string differs → fix by overwriting.\n",
752
+ "- **Content mismatch**: offset itself is wrong → requires manual review."
753
+ ],
754
+ "id": "62d9dd3ff74f13c6"
755
+ },
756
+ {
757
+ "cell_type": "code",
758
+ "metadata": {
759
+ "ExecuteTime": {
760
+ "end_time": "2026-04-18T19:53:18.785834Z",
761
+ "start_time": "2026-04-18T19:53:18.775763Z"
762
+ }
763
+ },
764
+ "source": [
765
+ "case_mismatches, content_mismatches = [], []\n",
766
+ "\n",
767
+ "for s in data_raw:\n",
768
+ " txt = s[\"text\"]\n",
769
+ " for e in s[\"entities\"]:\n",
770
+ " for p in e[\"positions\"]:\n",
771
+ " end = p[\"offset\"] + p[\"length\"]\n",
772
+ " actual = txt[p[\"offset\"]:end]\n",
773
+ " stored = p[\"position_text\"]\n",
774
+ "\n",
775
+ " if actual != stored:\n",
776
+ " row = {\n",
777
+ " \"sample_id\": s[\"id\"],\n",
778
+ " \"entity_text\": e[\"entity_text\"],\n",
779
+ " \"stored\": stored,\n",
780
+ " \"actual\": actual,\n",
781
+ " }\n",
782
+ "\n",
783
+ " if actual.lower() == stored.lower():\n",
784
+ " case_mismatches.append(row)\n",
785
+ " else:\n",
786
+ " content_mismatches.append(row)\n",
787
+ "\n",
788
+ "print(f\"Case-only mismatches : {len(case_mismatches):,}\")\n",
789
+ "print(f\"Content mismatches : {len(content_mismatches):,}\")\n",
790
+ "print()\n",
791
+ "\n",
792
+ "if case_mismatches:\n",
793
+ " print(\"Sample case mismatches:\")\n",
794
+ " print(pd.DataFrame(case_mismatches).head()[[\"sample_id\", \"entity_text\", \"stored\", \"actual\"]].to_string(index=False))\n",
795
+ " print()\n",
796
+ " print(\"All case mismatches: overwrite position_text with actual span from text.\")\n",
797
+ "\n",
798
+ "if content_mismatches:\n",
799
+ " print(\"Content mismatches require manual review:\")\n",
800
+ " print(pd.DataFrame(content_mismatches).head().to_string(index=False))\n",
801
+ "else:\n",
802
+ " print(\"No content mismatches.\")"
803
+ ],
804
+ "id": "ccbe961ef0ec6248",
805
+ "outputs": [
806
+ {
807
+ "name": "stdout",
808
+ "output_type": "stream",
809
+ "text": [
810
+ "Case-only mismatches : 224\n",
811
+ "Content mismatches : 0\n",
812
+ "\n",
813
+ "Sample case mismatches:\n",
814
+ " sample_id entity_text stored actual\n",
815
+ " 12 BRUSSELS BRUSSELS Brussels\n",
816
+ " 12 BRUSSELS BRUSSELS Brussels\n",
817
+ " 12 BRUSSELS BRUSSELS Brussels\n",
818
+ " 12 BRUSSELS BRUSSELS Brussels\n",
819
+ " 12 BRUSSELS BRUSSELS Brussels\n",
820
+ "\n",
821
+ "All case mismatches: overwrite position_text with actual span from text.\n",
822
+ "No content mismatches.\n"
823
+ ]
824
+ }
825
+ ],
826
+ "execution_count": 22
827
+ },
828
+ {
829
+ "cell_type": "markdown",
830
+ "metadata": {},
831
+ "source": [
832
+ "## 7. HTML Tags\n",
833
+ "\n",
834
+ "Raw text may contain residual HTML tags from web scraping (e.g. `<b>`, `<a href=...>`, `&amp;`).\n",
835
+ "These add noise to the model input and should be stripped during preprocessing."
836
+ ],
837
+ "id": "93ccd719e8730f47"
838
+ },
839
+ {
840
+ "cell_type": "code",
841
+ "metadata": {
842
+ "ExecuteTime": {
843
+ "end_time": "2026-04-18T19:53:18.838764Z",
844
+ "start_time": "2026-04-18T19:53:18.795458Z"
845
+ }
846
+ },
847
+ "source": [
848
+ "from bs4 import BeautifulSoup\n",
849
+ "import html as html_lib\n",
850
+ "\n",
851
+ "articles_with_tags = []\n",
852
+ "articles_with_entities = []\n",
853
+ "tag_counts = Counter()\n",
854
+ "entity_counts = Counter()\n",
855
+ "\n",
856
+ "for s in data_raw:\n",
857
+ " txt = s[\"text\"]\n",
858
+ "\n",
859
+ " soup = BeautifulSoup(txt, \"html.parser\")\n",
860
+ " tags = soup.find_all()\n",
861
+ " if tags:\n",
862
+ " tag_names = [str(t) for t in tags]\n",
863
+ " articles_with_tags.append({\"id\": s[\"id\"], \"tags\": tag_names, \"count\": len(tag_names)})\n",
864
+ " for t in tags:\n",
865
+ " tag_counts[t.name] += 1\n",
866
+ "\n",
867
+ " decoded = html_lib.unescape(txt)\n",
868
+ " if decoded != txt:\n",
869
+ " diff_chars = [\n",
870
+ " txt[i:i+20] for i in range(len(txt))\n",
871
+ " if i < len(decoded) and txt[i] != decoded[i]\n",
872
+ " ]\n",
873
+ " entities_found = []\n",
874
+ " i = 0\n",
875
+ " while i < len(txt):\n",
876
+ " if txt[i] == \"&\":\n",
877
+ " end = txt.find(\";\", i)\n",
878
+ " if end != -1:\n",
879
+ " candidate = txt[i:end+1]\n",
880
+ " if html_lib.unescape(candidate) != candidate:\n",
881
+ " entities_found.append(candidate)\n",
882
+ " i = end + 1\n",
883
+ " continue\n",
884
+ " i += 1\n",
885
+ " if entities_found:\n",
886
+ " articles_with_entities.append({\"id\": s[\"id\"], \"entities\": entities_found, \"count\": len(entities_found)})\n",
887
+ " for e in entities_found:\n",
888
+ " entity_counts[e] += 1\n",
889
+ "\n",
890
+ "print(f\"Articles containing HTML tags : {len(articles_with_tags):,} / {len(data_raw):,}\")\n",
891
+ "print(f\"Articles containing HTML entities : {len(articles_with_entities):,} / {len(data_raw):,}\")\n",
892
+ "print()\n",
893
+ "\n",
894
+ "if tag_counts:\n",
895
+ " print(\"Most common HTML tags:\")\n",
896
+ " for tag, cnt in tag_counts.most_common(10):\n",
897
+ " print(f\" {tag:30s}: {cnt:,}\")\n",
898
+ "else:\n",
899
+ " print(\"\\u2713 No HTML tags found in article texts.\")\n",
900
+ "\n",
901
+ "print()\n",
902
+ "if entity_counts:\n",
903
+ " print(\"Most common HTML entities:\")\n",
904
+ " for ent, cnt in entity_counts.most_common(10):\n",
905
+ " print(f\" {ent:30s}: {cnt:,}\")\n",
906
+ "else:\n",
907
+ " print(\"\\u2713 No HTML entities found in article texts.\")\n",
908
+ "\n",
909
+ "if articles_with_tags:\n",
910
+ " print()\n",
911
+ " print(\"Sample articles with HTML tags:\")\n",
912
+ " df_html = pd.DataFrame(articles_with_tags).sort_values(\"count\", ascending=False)\n",
913
+ " print(df_html.head(10).to_string(index=False))"
914
+ ],
915
+ "id": "51fe8e19c409a770",
916
+ "outputs": [
917
+ {
918
+ "name": "stdout",
919
+ "output_type": "stream",
920
+ "text": [
921
+ "Articles containing HTML tags : 2 / 1,637\n",
922
+ "Articles containing HTML entities : 6 / 1,637\n",
923
+ "\n",
924
+ "Most common HTML tags:\n",
925
+ " azn.l : 1\n",
926
+ " jnj.n : 1\n",
927
+ " sasy.pa : 1\n",
928
+ " mrna.o : 1\n",
929
+ " cvac.o : 1\n",
930
+ " pfe.n : 1\n",
931
+ " bntx.o : 1\n",
932
+ " rena.pa : 1\n",
933
+ "\n",
934
+ "Most common HTML entities:\n",
935
+ " &CloseCurlyQuote; : 13\n",
936
+ " &CloseCurlyDoubleQuote; : 10\n",
937
+ " &sol; : 2\n",
938
+ " &euro; : 2\n",
939
+ " &mdash; : 1\n",
940
+ " &P Global raises EDB credit rating to 'AA'\n",
941
+ "S&amp;: 1\n",
942
+ " &colon; : 1\n",
943
+ "\n",
944
+ "Sample articles with HTML tags:\n",
945
+ " id tags count\n",
946
+ "1046 [<azn.l> to secure at least 300 million doses of its potential COVID-19 vaccine, a spokesman said on Thursday. The deal covers development, liability and other costs faced by the vaccine maker. The EU has also secured an option to buy 100 million additional doses of the vaccine under development. The 27 EU states could buy it at a later stage, should the vaccine prove successful. The overall price they will pay to acquire the doses has not been revealed, but under an earlier deal struck in June with AstraZeneca by Germany, France, Italy and the Netherlands, all members of the EU, AstraZeneca agreed to sell 300 million doses for 750 million euros ($843 million). The EU deal completed the preliminary accord reached with the drug maker by the four countries, the Commission said in a statement. \"We cannot indicate at this stage the specific pricing per dose. However, a significant part of the overall costs are funded by a contribution from the overall ESI funding for vaccines,\" the commission spokesman said, referring to the 336 million euros paid through the bloc's so-called emergency support instrument. It is the first contract signed by the EU with a maker of potential COVID-19 vaccines. Brussels was previously said to be in advanced talks with Johnson &amp; Johnson <jnj.n>, Sanofi <sasy.pa>, Moderna <mrna.o> and CureVac <cvac.o> for their potential vaccines. EU officials told Reuters in July the bloc was also talking with Pfizer <pfe.n> and BionTech <bntx.o> for the shot they are developing together. The contract with AstraZeneca follows an advance purchase agreement signed by Brussels with the company earlier in August. Part of the money the EU pays for supply deals covers legal risks faced by vaccine makers if their shots have unexpected side effects. These risks are increased by the hastened process to develop a vaccine in the race against the COVID-19 pandemic. \"In order to compensate for such high risks taken by manufacturers, the Advanced Purchase Agreements provide for member states to indemnify the manufacturer for liabilities incurred under certain conditions,\" the commission said. \"Liability still remains with the companies,\" it added. This issue has been one of the stumbling blocs in talks with other vaccine makers, official told Reuters, as companies prefer to have a broader shield.</bntx.o></pfe.n></cvac.o></mrna.o></sasy.pa></jnj.n></azn.l>, <jnj.n>, Sanofi <sasy.pa>, Moderna <mrna.o> and CureVac <cvac.o> for their potential vaccines. EU officials told Reuters in July the bloc was also talking with Pfizer <pfe.n> and BionTech <bntx.o> for the shot they are developing together. The contract with AstraZeneca follows an advance purchase agreement signed by Brussels with the company earlier in August. Part of the money the EU pays for supply deals covers legal risks faced by vaccine makers if their shots have unexpected side effects. These risks are increased by the hastened process to develop a vaccine in the race against the COVID-19 pandemic. \"In order to compensate for such high risks taken by manufacturers, the Advanced Purchase Agreements provide for member states to indemnify the manufacturer for liabilities incurred under certain conditions,\" the commission said. \"Liability still remains with the companies,\" it added. This issue has been one of the stumbling blocs in talks with other vaccine makers, official told Reuters, as companies prefer to have a broader shield.</bntx.o></pfe.n></cvac.o></mrna.o></sasy.pa></jnj.n>, <sasy.pa>, Moderna <mrna.o> and CureVac <cvac.o> for their potential vaccines. EU officials told Reuters in July the bloc was also talking with Pfizer <pfe.n> and BionTech <bntx.o> for the shot they are developing together. The contract with AstraZeneca follows an advance purchase agreement signed by Brussels with the company earlier in August. Part of the money the EU pays for supply deals covers legal risks faced by vaccine makers if their shots have unexpected side effects. These risks are increased by the hastened process to develop a vaccine in the race against the COVID-19 pandemic. \"In order to compensate for such high risks taken by manufacturers, the Advanced Purchase Agreements provide for member states to indemnify the manufacturer for liabilities incurred under certain conditions,\" the commission said. \"Liability still remains with the companies,\" it added. This issue has been one of the stumbling blocs in talks with other vaccine makers, official told Reuters, as companies prefer to have a broader shield.</bntx.o></pfe.n></cvac.o></mrna.o></sasy.pa>, <mrna.o> and CureVac <cvac.o> for their potential vaccines. EU officials told Reuters in July the bloc was also talking with Pfizer <pfe.n> and BionTech <bntx.o> for the shot they are developing together. The contract with AstraZeneca follows an advance purchase agreement signed by Brussels with the company earlier in August. Part of the money the EU pays for supply deals covers legal risks faced by vaccine makers if their shots have unexpected side effects. These risks are increased by the hastened process to develop a vaccine in the race against the COVID-19 pandemic. \"In order to compensate for such high risks taken by manufacturers, the Advanced Purchase Agreements provide for member states to indemnify the manufacturer for liabilities incurred under certain conditions,\" the commission said. \"Liability still remains with the companies,\" it added. This issue has been one of the stumbling blocs in talks with other vaccine makers, official told Reuters, as companies prefer to have a broader shield.</bntx.o></pfe.n></cvac.o></mrna.o>, <cvac.o> for their potential vaccines. EU officials told Reuters in July the bloc was also talking with Pfizer <pfe.n> and BionTech <bntx.o> for the shot they are developing together. The contract with AstraZeneca follows an advance purchase agreement signed by Brussels with the company earlier in August. Part of the money the EU pays for supply deals covers legal risks faced by vaccine makers if their shots have unexpected side effects. These risks are increased by the hastened process to develop a vaccine in the race against the COVID-19 pandemic. \"In order to compensate for such high risks taken by manufacturers, the Advanced Purchase Agreements provide for member states to indemnify the manufacturer for liabilities incurred under certain conditions,\" the commission said. \"Liability still remains with the companies,\" it added. This issue has been one of the stumbling blocs in talks with other vaccine makers, official told Reuters, as companies prefer to have a broader shield.</bntx.o></pfe.n></cvac.o>, <pfe.n> and BionTech <bntx.o> for the shot they are developing together. The contract with AstraZeneca follows an advance purchase agreement signed by Brussels with the company earlier in August. Part of the money the EU pays for supply deals covers legal risks faced by vaccine makers if their shots have unexpected side effects. These risks are increased by the hastened process to develop a vaccine in the race against the COVID-19 pandemic. \"In order to compensate for such high risks taken by manufacturers, the Advanced Purchase Agreements provide for member states to indemnify the manufacturer for liabilities incurred under certain conditions,\" the commission said. \"Liability still remains with the companies,\" it added. This issue has been one of the stumbling blocs in talks with other vaccine makers, official told Reuters, as companies prefer to have a broader shield.</bntx.o></pfe.n>, <bntx.o> for the shot they are developing together. The contract with AstraZeneca follows an advance purchase agreement signed by Brussels with the company earlier in August. Part of the money the EU pays for supply deals covers legal risks faced by vaccine makers if their shots have unexpected side effects. These risks are increased by the hastened process to develop a vaccine in the race against the COVID-19 pandemic. \"In order to compensate for such high risks taken by manufacturers, the Advanced Purchase Agreements provide for member states to indemnify the manufacturer for liabilities incurred under certain conditions,\" the commission said. \"Liability still remains with the companies,\" it added. This issue has been one of the stumbling blocs in talks with other vaccine makers, official told Reuters, as companies prefer to have a broader shield.</bntx.o>] 7\n",
947
+ "1585 [<rena.pa> and thrown Nissan into disarray as it finds itself on course to book its lowest operating profit in 11 years. The sources said Nissan will likely kill loss-making variants for the Titan full-size pickup. Unprofitable variants include the single-cab and diesel versions. A planned shuttering of under-utilised production lines will most probably hit plants in emerging markets building Datsun and other small cars hardest, they added. \"We need to chart a recovery but the rot goes deep,\" one of the sources said of the many problems facing Nissan. The second source said all markets with factories except China were being looked at for possible reductions in production capacity. That source also said, however, that there were no plans to close an entire plant or withdraw completely from any country. In the United States, one of Nissan's biggest markets, the plan calls for fresh efforts to weed out the practice of buying market share by selling vehicles to rental car and other fleet operators at heavy discounts - a practice which destroyed profitability and undermined Nissan's brand image. \"We're trying to clean up what had happened in the past,\" one of the sources said, adding that under Ghosn, Nissan sought to meet sales objectives at any cost, including \"practically giving away cars\" to fleet customers. A team led by Jun Seki, a senior vice president and incoming vice chief operating officer, is expected to unveil the wide-ranging plan this month though some aspects are still being finalised, said the sources, who were not authorised to speak to media and declined to be identified. Nissan declined to comment. Seki is part of a new management team that will see Makoto Uchida, Nissan's head of China operations, take the helm - an appointment that is expected to take effect by Jan. 1. The new steps follow plans unveiled in July to cut headcount by 12,500 globally by early 2023 and which also flagged cuts to production capacity. At the time, then-CEO Hiroto Saikawa said 14 facilities would be affected. EMERGING MARKET WOES Overall, the plan's aim is to free up resources to focus more on the United States and China, the sources said. To that end it will roll back an aggressive expansionist strategy Ghosn set in motion under a five-year plan called Power 88 which aimed to raise profit margins and global market share to 8 percent by fiscal 2016 - goals which were never achieved. The Datsun brand - revived for emerging markets under Ghosn after being phased out in the 1980s - will likely bear the brunt of the restructuring. The models are manufactured in Indonesia, India and Russia. The sources said problems emerged after Nissan began deploying the no-frills cars in 2014 in small markets such as Indonesia, India, Russia and South Africa where it also sells vehicles under its mainstay Nissan brand. In Indonesia, for example, after a relatively good start, Datsun cars soon began eating into Nissan sales. \"We ended up pushing two mainstream brands in a market where you have a one or two percent market share. You cannot do that,\" one of the sources said, adding that there had been similar outcomes in India, South Africa and Russia. In its bigger markets, a steady supply of new or significantly redesigned models – starting with the redesigned Altima which was launched in the United States late last year – is expected to help Nissan reset the way it prices its vehicles. \"Still, it takes about a year to get any sort of tangible results,\" one of the sources said, adding that until then the Japanese automaker would continue to see sales by volume fall in the U.S. market. (Reporting by Norihiko Shirouzu; Additional reporting by Aditi Shah in New Delhi, Paul Lienert in Detroit and Naomi Tajitsu in Tokyo; Editing by Edwina Gibbs)</rena.pa>] 1\n"
948
+ ]
949
+ }
950
+ ],
951
+ "execution_count": 23
952
+ },
953
+ {
954
+ "cell_type": "markdown",
955
+ "metadata": {},
956
+ "source": [
957
+ "## 8. Preprocessing Summary\n",
958
+ "\n",
959
+ "All hygiene findings mapped to the preprocessing steps that resolve them."
960
+ ],
961
+ "id": "aac93d52102c5dcf"
962
+ },
963
+ {
964
+ "cell_type": "code",
965
+ "metadata": {
966
+ "ExecuteTime": {
967
+ "end_time": "2026-04-18T19:53:18.856223Z",
968
+ "start_time": "2026-04-18T19:53:18.851227Z"
969
+ }
970
+ },
971
+ "source": [
972
+ "summary = [\n",
973
+ " {\n",
974
+ " \"Issue\": \"Duplicate article texts\",\n",
975
+ " \"Count\": len(dup_articles),\n",
976
+ " \"Action\": \"Deduplicate at sample level before splitting\",\n",
977
+ " },\n",
978
+ " {\n",
979
+ " \"Issue\": \"Duplicate entities within sample (same label)\",\n",
980
+ " \"Count\": merge_stats[\"merged_entities\"],\n",
981
+ " \"Action\": \"Remove — redundant, no information loss\",\n",
982
+ " },\n",
983
+ " {\n",
984
+ " \"Issue\": \"Non-standard label ('very positive')\",\n",
985
+ " \"Count\": int((df_entities[\"label\"] == \"very positive\").sum()),\n",
986
+ " \"Action\": \"Remap to 'positive'\",\n",
987
+ " },\n",
988
+ " {\n",
989
+ " \"Issue\": \"position_text case mismatches (e.g. BRUSSELS vs Brussels)\",\n",
990
+ " \"Count\": len(case_mismatches),\n",
991
+ " \"Action\": \"Overwrite position_text with actual span from article text\",\n",
992
+ " },\n",
993
+ " {\n",
994
+ " \"Issue\": \"position_text content mismatches\",\n",
995
+ " \"Count\": len(content_mismatches),\n",
996
+ " \"Action\": \"Manual review; drop affected positions if unresolvable\",\n",
997
+ " },\n",
998
+ "]\n",
999
+ "\n",
1000
+ "df_summary = pd.DataFrame(summary)\n",
1001
+ "display(\n",
1002
+ "df_summary.style\n",
1003
+ ".set_table_styles([{\n",
1004
+ " \"selector\": \"th\",\n",
1005
+ " \"props\": [(\"font-weight\", \"bold\"), (\"background-color\", \"#ECEFF1\")]\n",
1006
+ "}])\n",
1007
+ ")\n",
1008
+ "\n",
1009
+ "print()\n",
1010
+ "total_removed = merge_stats[\"merged_entities\"] + int((df_entities[\"label\"] == \"very positive\").sum())\n",
1011
+ "print(f\"Entities before preprocessing : {len(df_entities):,}\")\n",
1012
+ "print(f\"Entities removed : {total_removed:,} ({total_removed / len(df_entities) * 100:.1f}%)\")\n",
1013
+ "print(f\"Entities after preprocessing : {len(df_entities) - total_removed:,}\")"
1014
+ ],
1015
+ "id": "77d44d31f2a37305",
1016
+ "outputs": [
1017
+ {
1018
+ "data": {
1019
+ "text/plain": [
1020
+ "<pandas.io.formats.style.Styler at 0x1141db4d0>"
1021
+ ],
1022
+ "text/html": [
1023
+ "<style type=\"text/css\">\n",
1024
+ "#T_b0e0e th {\n",
1025
+ " font-weight: bold;\n",
1026
+ " background-color: #ECEFF1;\n",
1027
+ "}\n",
1028
+ "</style>\n",
1029
+ "<table id=\"T_b0e0e\">\n",
1030
+ " <thead>\n",
1031
+ " <tr>\n",
1032
+ " <th class=\"blank level0\" >&nbsp;</th>\n",
1033
+ " <th id=\"T_b0e0e_level0_col0\" class=\"col_heading level0 col0\" >Issue</th>\n",
1034
+ " <th id=\"T_b0e0e_level0_col1\" class=\"col_heading level0 col1\" >Count</th>\n",
1035
+ " <th id=\"T_b0e0e_level0_col2\" class=\"col_heading level0 col2\" >Action</th>\n",
1036
+ " </tr>\n",
1037
+ " </thead>\n",
1038
+ " <tbody>\n",
1039
+ " <tr>\n",
1040
+ " <th id=\"T_b0e0e_level0_row0\" class=\"row_heading level0 row0\" >0</th>\n",
1041
+ " <td id=\"T_b0e0e_row0_col0\" class=\"data row0 col0\" >Duplicate article texts</td>\n",
1042
+ " <td id=\"T_b0e0e_row0_col1\" class=\"data row0 col1\" >16</td>\n",
1043
+ " <td id=\"T_b0e0e_row0_col2\" class=\"data row0 col2\" >Deduplicate at sample level before splitting</td>\n",
1044
+ " </tr>\n",
1045
+ " <tr>\n",
1046
+ " <th id=\"T_b0e0e_level0_row1\" class=\"row_heading level0 row1\" >1</th>\n",
1047
+ " <td id=\"T_b0e0e_row1_col0\" class=\"data row1 col0\" >Duplicate entities within sample (same label)</td>\n",
1048
+ " <td id=\"T_b0e0e_row1_col1\" class=\"data row1 col1\" >2014</td>\n",
1049
+ " <td id=\"T_b0e0e_row1_col2\" class=\"data row1 col2\" >Remove — redundant, no information loss</td>\n",
1050
+ " </tr>\n",
1051
+ " <tr>\n",
1052
+ " <th id=\"T_b0e0e_level0_row2\" class=\"row_heading level0 row2\" >2</th>\n",
1053
+ " <td id=\"T_b0e0e_row2_col0\" class=\"data row2 col0\" >Non-standard label ('very positive')</td>\n",
1054
+ " <td id=\"T_b0e0e_row2_col1\" class=\"data row2 col1\" >1</td>\n",
1055
+ " <td id=\"T_b0e0e_row2_col2\" class=\"data row2 col2\" >Remap to 'positive'</td>\n",
1056
+ " </tr>\n",
1057
+ " <tr>\n",
1058
+ " <th id=\"T_b0e0e_level0_row3\" class=\"row_heading level0 row3\" >3</th>\n",
1059
+ " <td id=\"T_b0e0e_row3_col0\" class=\"data row3 col0\" >position_text case mismatches (e.g. BRUSSELS vs Brussels)</td>\n",
1060
+ " <td id=\"T_b0e0e_row3_col1\" class=\"data row3 col1\" >224</td>\n",
1061
+ " <td id=\"T_b0e0e_row3_col2\" class=\"data row3 col2\" >Overwrite position_text with actual span from article text</td>\n",
1062
+ " </tr>\n",
1063
+ " <tr>\n",
1064
+ " <th id=\"T_b0e0e_level0_row4\" class=\"row_heading level0 row4\" >4</th>\n",
1065
+ " <td id=\"T_b0e0e_row4_col0\" class=\"data row4 col0\" >position_text content mismatches</td>\n",
1066
+ " <td id=\"T_b0e0e_row4_col1\" class=\"data row4 col1\" >0</td>\n",
1067
+ " <td id=\"T_b0e0e_row4_col2\" class=\"data row4 col2\" >Manual review; drop affected positions if unresolvable</td>\n",
1068
+ " </tr>\n",
1069
+ " </tbody>\n",
1070
+ "</table>\n"
1071
+ ]
1072
+ },
1073
+ "metadata": {},
1074
+ "output_type": "display_data"
1075
+ },
1076
+ {
1077
+ "name": "stdout",
1078
+ "output_type": "stream",
1079
+ "text": [
1080
+ "\n",
1081
+ "Entities before preprocessing : 12,627\n",
1082
+ "Entities removed : 2,015 (16.0%)\n",
1083
+ "Entities after preprocessing : 10,612\n"
1084
+ ]
1085
+ }
1086
+ ],
1087
+ "execution_count": 24
1088
+ }
1089
+ ],
1090
+ "metadata": {
1091
+ "kernelspec": {
1092
+ "display_name": "Python 3",
1093
+ "language": "python",
1094
+ "name": "python3"
1095
+ },
1096
+ "language_info": {
1097
+ "name": "python",
1098
+ "version": "3.10.0"
1099
+ }
1100
+ },
1101
+ "nbformat": 4,
1102
+ "nbformat_minor": 5
1103
+ }
notebooks/data_splits_analysis.ipynb ADDED
The diff for this file is too large to render. See raw diff
 
notebooks/flatten_examples_analysis.ipynb ADDED
@@ -0,0 +1,568 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "markdown",
5
+ "id": "a1",
6
+ "metadata": {},
7
+ "source": [
8
+ "# Flatten to Examples — Inspection\n",
9
+ "\n",
10
+ "Visual walkthrough of what `flatten_to_examples` produces for each mode.\n",
11
+ "\n",
12
+ "**Structure**\n",
13
+ "1. Load Augmented Data\n",
14
+ "2. Marker Mode Examples\n",
15
+ "3. QA-M Mode Examples\n",
16
+ "4. QA-B Mode Examples\n",
17
+ "5. Example Counts Summary"
18
+ ]
19
+ },
20
+ {
21
+ "cell_type": "code",
22
+ "id": "a2",
23
+ "metadata": {
24
+ "ExecuteTime": {
25
+ "end_time": "2026-04-19T00:12:35.836199Z",
26
+ "start_time": "2026-04-19T00:12:35.833236Z"
27
+ }
28
+ },
29
+ "source": [
30
+ "import os\n",
31
+ "import sys\n",
32
+ "from collections import Counter\n",
33
+ "\n",
34
+ "import pandas as pd\n",
35
+ "\n",
36
+ "sys.path.insert(0, os.path.abspath(\"..\"))\n",
37
+ "\n",
38
+ "from src.models.dataset import load_data, flatten_to_examples\n",
39
+ "from src.schemas.labels import SENTIMENT_LABELS"
40
+ ],
41
+ "outputs": [],
42
+ "execution_count": 34
43
+ },
44
+ {
45
+ "cell_type": "markdown",
46
+ "id": "a3",
47
+ "metadata": {},
48
+ "source": [
49
+ "## 1. Load Augmented Data"
50
+ ]
51
+ },
52
+ {
53
+ "cell_type": "code",
54
+ "id": "a4",
55
+ "metadata": {
56
+ "ExecuteTime": {
57
+ "end_time": "2026-04-19T00:12:35.991309Z",
58
+ "start_time": "2026-04-19T00:12:35.845709Z"
59
+ }
60
+ },
61
+ "source": [
62
+ "samples = load_data(os.path.join(\"..\", \"data\", \"data_augmented_256.jsonl\"))\n",
63
+ "print(f\"Loaded {len(samples)} samples\")\n",
64
+ "print(f\"First sample has {len(samples[0]['entities'])} entities\")\n",
65
+ "print(f\"First entity has {len(samples[0]['entities'][0]['positions'])} positions\")"
66
+ ],
67
+ "outputs": [
68
+ {
69
+ "name": "stdout",
70
+ "output_type": "stream",
71
+ "text": [
72
+ "Loaded 1629 samples\n",
73
+ "First sample has 3 entities\n",
74
+ "First entity has 1 positions\n"
75
+ ]
76
+ }
77
+ ],
78
+ "execution_count": 35
79
+ },
80
+ {
81
+ "cell_type": "markdown",
82
+ "id": "a5",
83
+ "metadata": {},
84
+ "source": [
85
+ "### Raw structure of one entity (before flattening)"
86
+ ]
87
+ },
88
+ {
89
+ "cell_type": "code",
90
+ "id": "a6",
91
+ "metadata": {
92
+ "ExecuteTime": {
93
+ "end_time": "2026-04-19T00:12:36.000280Z",
94
+ "start_time": "2026-04-19T00:12:35.997489Z"
95
+ }
96
+ },
97
+ "source": [
98
+ "s = samples[0]\n",
99
+ "e = s[\"entities\"][0]\n",
100
+ "p = e[\"positions\"][0]\n",
101
+ "\n",
102
+ "print(f\"Sample ID: {s['id']}\")\n",
103
+ "print(f\"Entity: {e['entity_text']} ({e['entity_type']})\")\n",
104
+ "print(f\"Label: {e['label']}\")\n",
105
+ "print(f\"Positions: {len(e['positions'])}\")\n",
106
+ "print()\n",
107
+ "print(\"Position fields:\")\n",
108
+ "for key in p:\n",
109
+ " val = p[key]\n",
110
+ " if isinstance(val, str) and len(val) > 80:\n",
111
+ " val = val[:80] + \"...\"\n",
112
+ " print(f\" {key}: {val}\")"
113
+ ],
114
+ "outputs": [
115
+ {
116
+ "name": "stdout",
117
+ "output_type": "stream",
118
+ "text": [
119
+ "Sample ID: 0\n",
120
+ "Entity: Verge (company)\n",
121
+ "Label: neutral\n",
122
+ "Positions: 1\n",
123
+ "\n",
124
+ "Position fields:\n",
125
+ " position_text: Verge\n",
126
+ " offset: 2082\n",
127
+ " length: 5\n",
128
+ " entity_centered_window: if the companies don't comply with the law. The two mole and skin tag removal pr...\n",
129
+ " marker_text: if the companies don't comply with the law. The two mole and skin tag removal pr...\n",
130
+ " qa_m_question: What do you think of the sentiment of the company Verge ?\n",
131
+ " qa_b_hypotheses: {'negative': 'The polarity of the company Verge is negative .', 'neutral': 'The polarity of the company Verge is neutral .', 'positive': 'The polarity of the company Verge is positive .'}\n"
132
+ ]
133
+ }
134
+ ],
135
+ "execution_count": 36
136
+ },
137
+ {
138
+ "cell_type": "code",
139
+ "id": "99904016",
140
+ "source": "print(\"All positions for first few entities:\\n\")\nfor s in samples[:2]:\n print(f\"Sample {s['id']}:\")\n for e in s[\"entities\"]:\n print(f\" Entity: {e['entity_text']} ({e['entity_type']}) — label: {e.get('label', 'N/A')}\")\n for j, p in enumerate(e[\"positions\"]):\n print(f\" Position {j}: offset={p['offset']}, length={p['length']}, text=\\\"{p['position_text']}\\\"\")\n print()",
141
+ "metadata": {
142
+ "ExecuteTime": {
143
+ "end_time": "2026-04-19T00:12:36.009947Z",
144
+ "start_time": "2026-04-19T00:12:36.007896Z"
145
+ }
146
+ },
147
+ "outputs": [
148
+ {
149
+ "name": "stdout",
150
+ "output_type": "stream",
151
+ "text": [
152
+ "All positions for first few entities:\n",
153
+ "\n",
154
+ "Sample 0:\n",
155
+ " Entity: Verge (company) — label: neutral\n",
156
+ " Position 0: offset=2082, length=5, text=\"Verge\"\n",
157
+ " Entity: Amazon (company) — label: negative\n",
158
+ " Position 0: offset=0, length=6, text=\"Amazon\"\n",
159
+ " Position 1: offset=121, length=6, text=\"Amazon\"\n",
160
+ " Position 2: offset=449, length=6, text=\"Amazon\"\n",
161
+ " Position 3: offset=556, length=6, text=\"Amazon\"\n",
162
+ " Position 4: offset=1689, length=6, text=\"Amazon\"\n",
163
+ " Position 5: offset=1798, length=6, text=\"Amazon\"\n",
164
+ " Position 6: offset=1848, length=6, text=\"Amazon\"\n",
165
+ " Position 7: offset=2097, length=6, text=\"Amazon\"\n",
166
+ " Entity: US (location) — label: neutral\n",
167
+ " Position 0: offset=1201, length=2, text=\"US\"\n",
168
+ "\n",
169
+ "Sample 1:\n",
170
+ " Entity: Sputnik International (company) — label: neutral\n",
171
+ " Position 0: offset=79, length=21, text=\"Sputnik International\"\n",
172
+ " Entity: Kremlin (company) — label: neutral\n",
173
+ " Position 0: offset=378, length=7, text=\"Kremlin\"\n",
174
+ " Position 1: offset=1126, length=7, text=\"Kremlin\"\n",
175
+ " Entity: Russian (company) — label: neutral\n",
176
+ " Position 0: offset=336, length=7, text=\"Russian\"\n",
177
+ " Entity: Foreign Ministry (company) — label: neutral\n",
178
+ " Position 0: offset=344, length=16, text=\"Foreign Ministry\"\n",
179
+ " Position 1: offset=1024, length=16, text=\"Foreign Ministry\"\n",
180
+ " Position 2: offset=2293, length=16, text=\"Foreign Ministry\"\n",
181
+ " Entity: FBI (company) — label: neutral\n",
182
+ " Position 0: offset=1591, length=3, text=\"FBI\"\n",
183
+ " Entity: SolarWinds (company) — label: neutral\n",
184
+ " Position 0: offset=271, length=10, text=\"SolarWinds\"\n",
185
+ " Position 1: offset=2767, length=10, text=\"SolarWinds\"\n",
186
+ " Entity: CIA (company) — label: neutral\n",
187
+ " Position 0: offset=1993, length=3, text=\"CIA\"\n",
188
+ " Entity: Federal Bureau of Prisons (company) — label: neutral\n",
189
+ " Position 0: offset=1723, length=25, text=\"Federal Bureau of Prisons\"\n",
190
+ " Entity: Sputnik News (company) — label: neutral\n",
191
+ " Position 0: offset=55, length=12, text=\"Sputnik News\"\n",
192
+ "\n"
193
+ ]
194
+ }
195
+ ],
196
+ "execution_count": 37
197
+ },
198
+ {
199
+ "cell_type": "markdown",
200
+ "id": "a7",
201
+ "metadata": {},
202
+ "source": [
203
+ "## 2. Marker Mode Examples\n",
204
+ "\n",
205
+ "One example per position. `seg_a` = entity wrapped with `[E]...[/E]`, `seg_b` = None."
206
+ ]
207
+ },
208
+ {
209
+ "cell_type": "code",
210
+ "id": "a8",
211
+ "metadata": {
212
+ "ExecuteTime": {
213
+ "end_time": "2026-04-19T00:12:36.032677Z",
214
+ "start_time": "2026-04-19T00:12:36.018233Z"
215
+ }
216
+ },
217
+ "source": "marker_exs = flatten_to_examples(samples, mode=\"marker\")\nprint(f\"Total marker examples: {len(marker_exs)}\")\nprint()",
218
+ "outputs": [
219
+ {
220
+ "name": "stdout",
221
+ "output_type": "stream",
222
+ "text": [
223
+ "Total marker examples: 26776\n",
224
+ "\n"
225
+ ]
226
+ }
227
+ ],
228
+ "execution_count": 38
229
+ },
230
+ {
231
+ "cell_type": "code",
232
+ "id": "a9",
233
+ "metadata": {
234
+ "ExecuteTime": {
235
+ "end_time": "2026-04-19T00:12:36.040278Z",
236
+ "start_time": "2026-04-19T00:12:36.038083Z"
237
+ }
238
+ },
239
+ "source": [
240
+ "for i, ex in enumerate(marker_exs[:3]):\n",
241
+ " print(f\"--- Example {i} ---\")\n",
242
+ " print(f\" entity: {ex['entity_text']} ({ex['entity_type']})\")\n",
243
+ " print(f\" label: {SENTIMENT_LABELS.id2label[ex['label']]} (id={ex['label']})\")\n",
244
+ " print(f\" seg_a: {ex['seg_a']}...\")\n",
245
+ " print(f\" seg_b: {ex['seg_b']}\")\n",
246
+ " print()"
247
+ ],
248
+ "outputs": [
249
+ {
250
+ "name": "stdout",
251
+ "output_type": "stream",
252
+ "text": [
253
+ "--- Example 0 ---\n",
254
+ " entity: Verge (company)\n",
255
+ " label: neutral (id=1)\n",
256
+ " seg_a: if the companies don't comply with the law. The two mole and skin tag removal products appear to no longer be available on Amazon's website. But there are still multiple other mole and skin tag removal serums and creams for sale on Amazon, according to a search for \"mole remover.\" Amazon has received warnings from the FDA before. In 2021, the FDA sent the company an untitled letter (a step below a warning letter), saying that the sale of sexual enhancement and weight loss products violated the law. Source: The [E] Verge [/E] The post Amazon sold unauthorized mole removers, and the FDA isn't happy about it appeared first on Trend Fool ....\n",
257
+ " seg_b: None\n",
258
+ "\n",
259
+ "--- Example 1 ---\n",
260
+ " entity: Amazon (company)\n",
261
+ " label: negative (id=0)\n",
262
+ " seg_a: [E] Amazon [/E] sold unauthorized mole removers, and the FDA isn't happy about it Unauthorized mole and skin tag removers sold on Amazon put the company in the crosshairs of the Food and Drug Administration, which sent a warning letter to the retail giant this month asking that it remove the products from its website. There are no authorized over-the-counter drugs that remove moles or skin tags, the FDA said in its warning letter, which was addressed to Amazon CEO Andy Jassy. As part of its research, the agency says it bought two...\n",
263
+ " seg_b: None\n",
264
+ "\n",
265
+ "--- Example 2 ---\n",
266
+ " entity: Amazon (company)\n",
267
+ " label: negative (id=0)\n",
268
+ " seg_a: Amazon sold unauthorized mole removers, and the FDA isn't happy about it Unauthorized mole and skin tag removers sold on [E] Amazon [/E] put the company in the crosshairs of the Food and Drug Administration, which sent a warning letter to the retail giant this month asking that it remove the products from its website. There are no authorized over-the-counter drugs that remove moles or skin tags, the FDA said in its warning letter, which was addressed to Amazon CEO Andy Jassy. As part of its research, the agency says it bought two of the offending products on Amazon: the \"Deisana Skin Tag Remover, Mole Remover and Repair Gel Set\" and the \"Skincell...\n",
269
+ " seg_b: None\n",
270
+ "\n"
271
+ ]
272
+ }
273
+ ],
274
+ "execution_count": 39
275
+ },
276
+ {
277
+ "cell_type": "markdown",
278
+ "id": "a10",
279
+ "metadata": {},
280
+ "source": [
281
+ "## 3. QA-M Mode Examples\n",
282
+ "\n",
283
+ "One example per position. `seg_a` = context window, `seg_b` = question about the entity."
284
+ ]
285
+ },
286
+ {
287
+ "cell_type": "code",
288
+ "id": "a11",
289
+ "metadata": {
290
+ "ExecuteTime": {
291
+ "end_time": "2026-04-19T00:12:36.119437Z",
292
+ "start_time": "2026-04-19T00:12:36.049072Z"
293
+ }
294
+ },
295
+ "source": "qa_m_exs = flatten_to_examples(samples, mode=\"qa_m\")\nprint(f\"Total qa_m examples: {len(qa_m_exs)}\")\nprint()",
296
+ "outputs": [
297
+ {
298
+ "name": "stdout",
299
+ "output_type": "stream",
300
+ "text": [
301
+ "Total qa_m examples: 26776\n",
302
+ "\n"
303
+ ]
304
+ }
305
+ ],
306
+ "execution_count": 40
307
+ },
308
+ {
309
+ "cell_type": "code",
310
+ "id": "a12",
311
+ "metadata": {
312
+ "ExecuteTime": {
313
+ "end_time": "2026-04-19T00:12:36.130551Z",
314
+ "start_time": "2026-04-19T00:12:36.128443Z"
315
+ }
316
+ },
317
+ "source": [
318
+ "for i, ex in enumerate(qa_m_exs[:3]):\n",
319
+ " print(f\"--- Example {i} ---\")\n",
320
+ " print(f\" entity: {ex['entity_text']} ({ex['entity_type']})\")\n",
321
+ " print(f\" label: {SENTIMENT_LABELS.id2label[ex['label']]} (id={ex['label']})\")\n",
322
+ " print(f\" seg_a: {ex['seg_a']}...\")\n",
323
+ " print(f\" seg_b: {ex['seg_b']}\")\n",
324
+ " print()"
325
+ ],
326
+ "outputs": [
327
+ {
328
+ "name": "stdout",
329
+ "output_type": "stream",
330
+ "text": [
331
+ "--- Example 0 ---\n",
332
+ " entity: Verge (company)\n",
333
+ " label: neutral (id=1)\n",
334
+ " seg_a: if the companies don't comply with the law. The two mole and skin tag removal products appear to no longer be available on Amazon's website. But there are still multiple other mole and skin tag removal serums and creams for sale on Amazon, according to a search for \"mole remover.\" Amazon has received warnings from the FDA before. In 2021, the FDA sent the company an untitled letter (a step below a warning letter), saying that the sale of sexual enhancement and weight loss products violated the law. Source: The Verge The post Amazon sold unauthorized mole removers, and the FDA isn't happy about it appeared first on Trend Fool ....\n",
335
+ " seg_b: What do you think of the sentiment of the company Verge ?\n",
336
+ "\n",
337
+ "--- Example 1 ---\n",
338
+ " entity: Amazon (company)\n",
339
+ " label: negative (id=0)\n",
340
+ " seg_a: Amazon sold unauthorized mole removers, and the FDA isn't happy about it\n",
341
+ "Unauthorized mole and skin tag removers sold on Amazon put the company in the crosshairs of the Food and Drug Administration, which sent a warning letter to the retail giant this month asking that it remove the products from its website. There are no authorized over-the-counter drugs that remove moles or skin tags, the FDA said in its warning letter, which was addressed to Amazon CEO Andy Jassy. As part of its research, the agency says it bought two...\n",
342
+ " seg_b: What do you think of the sentiment of the company Amazon ?\n",
343
+ "\n",
344
+ "--- Example 2 ---\n",
345
+ " entity: Amazon (company)\n",
346
+ " label: negative (id=0)\n",
347
+ " seg_a: Amazon sold unauthorized mole removers, and the FDA isn't happy about it\n",
348
+ "Unauthorized mole and skin tag removers sold on Amazon put the company in the crosshairs of the Food and Drug Administration, which sent a warning letter to the retail giant this month asking that it remove the products from its website. There are no authorized over-the-counter drugs that remove moles or skin tags, the FDA said in its warning letter, which was addressed to Amazon CEO Andy Jassy. As part of its research, the agency says it bought two of the offending products on Amazon: the \"Deisana Skin Tag Remover, Mole Remover and Repair Gel Set\" and the \"Skincell...\n",
349
+ " seg_b: What do you think of the sentiment of the company Amazon ?\n",
350
+ "\n"
351
+ ]
352
+ }
353
+ ],
354
+ "execution_count": 41
355
+ },
356
+ {
357
+ "cell_type": "markdown",
358
+ "id": "a13",
359
+ "metadata": {},
360
+ "source": [
361
+ "## 4. QA-B Mode Examples\n",
362
+ "\n",
363
+ "**Three** examples per position — one per sentiment hypothesis. Labels are binary (1 = correct sentiment, 0 = incorrect). Triplets are always in order: negative, neutral, positive."
364
+ ]
365
+ },
366
+ {
367
+ "cell_type": "code",
368
+ "id": "a14",
369
+ "metadata": {
370
+ "ExecuteTime": {
371
+ "end_time": "2026-04-19T00:12:36.226209Z",
372
+ "start_time": "2026-04-19T00:12:36.198371Z"
373
+ }
374
+ },
375
+ "source": "qa_b_exs = flatten_to_examples(samples, mode=\"qa_b\")\nprint(f\"Total qa_b examples: {len(qa_b_exs)}\")\nprint(f\" (= {len(qa_b_exs) // 3} triplets x 3 sentiments)\")\nprint()",
376
+ "outputs": [
377
+ {
378
+ "name": "stdout",
379
+ "output_type": "stream",
380
+ "text": [
381
+ "Total qa_b examples: 80328\n",
382
+ " (= 26776 triplets x 3 sentiments)\n",
383
+ "\n"
384
+ ]
385
+ }
386
+ ],
387
+ "execution_count": 42
388
+ },
389
+ {
390
+ "cell_type": "code",
391
+ "id": "a15",
392
+ "metadata": {
393
+ "ExecuteTime": {
394
+ "end_time": "2026-04-19T00:12:36.236907Z",
395
+ "start_time": "2026-04-19T00:12:36.234978Z"
396
+ }
397
+ },
398
+ "source": [
399
+ "print(\"First triplet (3 examples for one entity-position pair):\")\n",
400
+ "print()\n",
401
+ "for i, ex in enumerate(qa_b_exs[:3]):\n",
402
+ " print(f\"--- Triplet example {i} ({ex['sentiment']}) ---\")\n",
403
+ " print(f\" entity: {ex['entity_text']} ({ex['entity_type']})\")\n",
404
+ " print(f\" sentiment: {ex['sentiment']}\")\n",
405
+ " print(f\" label: {ex['label']} ({'yes' if ex['label'] == 1 else 'no'})\")\n",
406
+ " print(f\" seg_a: {ex['seg_a'][:80]}...\")\n",
407
+ " print(f\" seg_b: {ex['seg_b']}\")\n",
408
+ " print()"
409
+ ],
410
+ "outputs": [
411
+ {
412
+ "name": "stdout",
413
+ "output_type": "stream",
414
+ "text": [
415
+ "First triplet (3 examples for one entity-position pair):\n",
416
+ "\n",
417
+ "--- Triplet example 0 (negative) ---\n",
418
+ " entity: Verge (company)\n",
419
+ " sentiment: negative\n",
420
+ " label: 0 (no)\n",
421
+ " seg_a: if the companies don't comply with the law. The two mole and skin tag removal pr...\n",
422
+ " seg_b: The polarity of the company Verge is negative .\n",
423
+ "\n",
424
+ "--- Triplet example 1 (neutral) ---\n",
425
+ " entity: Verge (company)\n",
426
+ " sentiment: neutral\n",
427
+ " label: 1 (yes)\n",
428
+ " seg_a: if the companies don't comply with the law. The two mole and skin tag removal pr...\n",
429
+ " seg_b: The polarity of the company Verge is neutral .\n",
430
+ "\n",
431
+ "--- Triplet example 2 (positive) ---\n",
432
+ " entity: Verge (company)\n",
433
+ " sentiment: positive\n",
434
+ " label: 0 (no)\n",
435
+ " seg_a: if the companies don't comply with the law. The two mole and skin tag removal pr...\n",
436
+ " seg_b: The polarity of the company Verge is positive .\n",
437
+ "\n"
438
+ ]
439
+ }
440
+ ],
441
+ "execution_count": 43
442
+ },
443
+ {
444
+ "cell_type": "code",
445
+ "id": "a16",
446
+ "metadata": {
447
+ "ExecuteTime": {
448
+ "end_time": "2026-04-19T00:12:36.251498Z",
449
+ "start_time": "2026-04-19T00:12:36.249261Z"
450
+ }
451
+ },
452
+ "source": [
453
+ "print(\"Second triplet (different entity/position):\")\n",
454
+ "print()\n",
455
+ "for i, ex in enumerate(qa_b_exs[3:6]):\n",
456
+ " print(f\"--- Triplet example {i} ({ex['sentiment']}) ---\")\n",
457
+ " print(f\" entity: {ex['entity_text']}\")\n",
458
+ " print(f\" sentiment: {ex['sentiment']}\")\n",
459
+ " print(f\" label: {ex['label']} ({'yes' if ex['label'] == 1 else 'no'})\")\n",
460
+ " print(f\" seg_b: {ex['seg_b']}\")\n",
461
+ " print()"
462
+ ],
463
+ "outputs": [
464
+ {
465
+ "name": "stdout",
466
+ "output_type": "stream",
467
+ "text": [
468
+ "Second triplet (different entity/position):\n",
469
+ "\n",
470
+ "--- Triplet example 0 (negative) ---\n",
471
+ " entity: Amazon\n",
472
+ " sentiment: negative\n",
473
+ " label: 1 (yes)\n",
474
+ " seg_b: The polarity of the company Amazon is negative .\n",
475
+ "\n",
476
+ "--- Triplet example 1 (neutral) ---\n",
477
+ " entity: Amazon\n",
478
+ " sentiment: neutral\n",
479
+ " label: 0 (no)\n",
480
+ " seg_b: The polarity of the company Amazon is neutral .\n",
481
+ "\n",
482
+ "--- Triplet example 2 (positive) ---\n",
483
+ " entity: Amazon\n",
484
+ " sentiment: positive\n",
485
+ " label: 0 (no)\n",
486
+ " seg_b: The polarity of the company Amazon is positive .\n",
487
+ "\n"
488
+ ]
489
+ }
490
+ ],
491
+ "execution_count": 44
492
+ },
493
+ {
494
+ "cell_type": "markdown",
495
+ "id": "a17",
496
+ "metadata": {},
497
+ "source": [
498
+ "## 5. Example Counts Summary"
499
+ ]
500
+ },
501
+ {
502
+ "cell_type": "code",
503
+ "id": "a18",
504
+ "metadata": {
505
+ "ExecuteTime": {
506
+ "end_time": "2026-04-19T00:12:36.270922Z",
507
+ "start_time": "2026-04-19T00:12:36.260121Z"
508
+ }
509
+ },
510
+ "source": [
511
+ "total_positions = sum(\n",
512
+ " len(p)\n",
513
+ " for s in samples\n",
514
+ " for e in s[\"entities\"]\n",
515
+ " for p in [e[\"positions\"]]\n",
516
+ ")\n",
517
+ "total_entities = sum(len(s[\"entities\"]) for s in samples)\n",
518
+ "\n",
519
+ "print(f\"Samples: {len(samples)}\")\n",
520
+ "print(f\"Entities: {total_entities}\")\n",
521
+ "print(f\"Positions: {total_positions}\")\n",
522
+ "print()\n",
523
+ "print(f\"Marker examples: {len(marker_exs):>6} (1 per position)\")\n",
524
+ "print(f\"QA-M examples: {len(qa_m_exs):>6} (1 per position)\")\n",
525
+ "print(f\"QA-B examples: {len(qa_b_exs):>6} (3 per position)\")\n",
526
+ "print()\n",
527
+ "print(\"Label distributions:\")\n",
528
+ "print(f\" Marker: {dict(Counter(SENTIMENT_LABELS.id2label[e['label']] for e in marker_exs))}\")\n",
529
+ "print(f\" QA-M: {dict(Counter(SENTIMENT_LABELS.id2label[e['label']] for e in qa_m_exs))}\")\n",
530
+ "print(f\" QA-B: yes={sum(1 for e in qa_b_exs if e['label']==1)} no={sum(1 for e in qa_b_exs if e['label']==0)}\")"
531
+ ],
532
+ "outputs": [
533
+ {
534
+ "name": "stdout",
535
+ "output_type": "stream",
536
+ "text": [
537
+ "Samples: 1629\n",
538
+ "Entities: 10550\n",
539
+ "Positions: 26776\n",
540
+ "\n",
541
+ "Marker examples: 26776 (1 per position)\n",
542
+ "QA-M examples: 26776 (1 per position)\n",
543
+ "QA-B examples: 80328 (3 per position)\n",
544
+ "\n",
545
+ "Label distributions:\n",
546
+ " Marker: {'neutral': 10905, 'negative': 7854, 'positive': 8017}\n",
547
+ " QA-M: {'neutral': 10905, 'negative': 7854, 'positive': 8017}\n",
548
+ " QA-B: yes=26776 no=53552\n"
549
+ ]
550
+ }
551
+ ],
552
+ "execution_count": 45
553
+ }
554
+ ],
555
+ "metadata": {
556
+ "kernelspec": {
557
+ "display_name": "Python 3",
558
+ "language": "python",
559
+ "name": "python3"
560
+ },
561
+ "language_info": {
562
+ "name": "python",
563
+ "version": "3.11.0"
564
+ }
565
+ },
566
+ "nbformat": 4,
567
+ "nbformat_minor": 5
568
+ }