Spaces:
Running
Running
File size: 5,502 Bytes
ee37d63 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 | """Conservative Taiwan stock subsector classification for hotspot analysis.
``twstock.codes.group`` remains the authoritative broad industry. This module
adds a smaller, auditable layer only where a code override or a narrow company
name rule is reliable. Unmatched stocks keep their original industry group.
"""
from __future__ import annotations
from typing import Any
TAXONOMY_SOURCE = "twstock.codes.group + conservative code/name rules"
TAXONOMY_LEVEL = "industry > subsector"
CODE_OVERRIDES: dict[str, str] = {
# Semiconductor
"2303": "晶圓代工",
"2330": "晶圓代工",
"5347": "晶圓代工",
"6770": "晶圓代工",
"2379": "IC 設計",
"2454": "IC 設計",
"3034": "IC 設計",
"3443": "IC 設計",
"3661": "IC 設計",
"5274": "IC 設計",
"6415": "IC 設計",
"2329": "封裝測試",
"2449": "封裝測試",
"3711": "封裝測試",
"6239": "封裝測試",
"6257": "封裝測試",
"8150": "封裝測試",
"2408": "記憶體",
"2344": "記憶體",
"2337": "記憶體",
"3260": "記憶體",
"8299": "記憶體",
# Computer and peripherals
"2382": "AI 伺服器",
"3231": "AI 伺服器",
"6669": "AI 伺服器",
"3706": "伺服器",
"8210": "伺服器",
"2324": "筆電與品牌電腦",
"2353": "筆電與品牌電腦",
"2356": "筆電與品牌電腦",
"2357": "筆電與品牌電腦",
"4938": "筆電與品牌電腦",
# Electronic components
"2313": "PCB",
"2368": "PCB",
"2383": "PCB",
"3037": "PCB",
"3044": "PCB",
"4958": "PCB",
"5439": "PCB",
"6191": "PCB",
"6274": "PCB",
"8046": "PCB",
"2327": "被動元件",
"2492": "被動元件",
"3026": "被動元件",
"6173": "被動元件",
"6449": "被動元件",
"2059": "機構件與散熱",
"3017": "機構件與散熱",
"3653": "機構件與散熱",
# Optoelectronics
"2409": "面板",
"3481": "面板",
"6116": "面板",
"3008": "光學鏡頭",
"3406": "光學鏡頭",
"8069": "電子紙",
# Communications
"2412": "電信營運",
"3045": "電信營運",
"4904": "電信營運",
"2345": "網通設備",
"3596": "網通設備",
"5388": "網通設備",
"6285": "網通設備",
# Shipping and transportation
"2603": "貨櫃航運",
"2609": "貨櫃航運",
"2615": "貨櫃航運",
"2605": "散裝航運",
"2606": "散裝航運",
"2612": "散裝航運",
"2637": "散裝航運",
"5608": "散裝航運",
"2610": "航空",
"2618": "航空",
"2646": "航空",
"6757": "航空",
# Finance
"2855": "證券期貨",
"5864": "證券期貨",
"6005": "證券期貨",
"6015": "證券期貨",
"6016": "證券期貨",
"6020": "證券期貨",
"6021": "證券期貨",
"6023": "證券期貨",
"6024": "證券期貨",
"6026": "證券期貨",
}
NAME_RULES_BY_GROUP: dict[str, tuple[tuple[str, tuple[str, ...]], ...]] = {
"金融保險業": (
("證券期貨", ("證", "期")),
("保險", ("保", "產", "壽")),
("銀行", ("銀", "銀行", "商銀")),
),
"航運業": (
("航空", ("航空", "航太", "華航", "亞航", "虎航")),
("物流運輸", ("宅配", "貨運", "高鐵", "大榮", "中菲行", "捷迅")),
("港埠倉儲", ("港", "櫃", "榮運")),
("造船", ("船", "漢翔", "龍德")),
),
"光電業": (
("光學鏡頭", ("光學", "大立光", "玉晶光", "今國光", "佳能")),
("太陽能", ("太陽", "元晶", "茂迪", "聯合再生", "國碩", "安集")),
("LED", ("LED", "億光", "佰鴻", "宏齊", "光鋐", "富采")),
),
"通信網路業": (
("電信營運", ("電信", "台灣大", "遠傳", "中華電")),
("網通設備", ("網通", "智邦", "中磊", "啟碁", "神準", "合勤", "正文", "明泰")),
("光通訊", ("光通", "聯亞", "上詮", "華星光", "光聖", "波若威")),
),
"生技醫療業": (
("醫療器材", ("醫材", "醫療", "醫", "眼", "視陽", "晶碩", "精華")),
("藥品與新藥", ("藥", "生達", "美時", "保瑞", "神隆", "東洋")),
("生技研發", ("生技", "基因", "疫苗", "細胞")),
),
}
def classify_subsector(code: Any, name: Any, group: Any) -> dict[str, Any]:
"""Return a conservative fine-grained classification with audit metadata."""
normalized_code = str(code or "").replace(".TW", "").replace(".TWO", "").strip()
normalized_name = str(name or "").strip()
normalized_group = str(group or "").strip() or "未分類"
label = CODE_OVERRIDES.get(normalized_code)
source = "code_override" if label else ""
if not label:
for candidate, keywords in NAME_RULES_BY_GROUP.get(normalized_group, ()):
if any(keyword and keyword in normalized_name for keyword in keywords):
label = candidate
source = "name_keyword"
break
if not label:
return {
"subsector": normalized_group,
"subsector_path": normalized_group,
"subsector_source": "industry_fallback",
"subsector_refined": False,
}
return {
"subsector": label,
"subsector_path": f"{normalized_group} > {label}",
"subsector_source": source,
"subsector_refined": True,
}
|