PowerMachine commited on
Commit
214d542
·
verified ·
1 Parent(s): 4147e80

Delete cnn_bigru/utils/xeon_runtime.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. cnn_bigru/utils/xeon_runtime.py +0 -176
cnn_bigru/utils/xeon_runtime.py DELETED
@@ -1,176 +0,0 @@
1
- """xeon_runtime.py — Ativação do runtime Intel Xeon (AVX512 + AMX + IPEX + OneDNN).
2
-
3
- Adaptado do projeto BiGRU_T_version (PowerMachine), com simplificações para
4
- rodar em CPU-only e degradar graciosamente quando IPEX/AMX não estiverem
5
- disponíveis. A ativação é SEMPRE realizada (mesmo que parcial), conforme
6
- requisito do usuário: "ativar xeon_runtime.py".
7
-
8
- Otimizações aplicadas:
9
- 1. OMP_NUM_THREADS / MKL_NUM_THREADS = núcleos físicos
10
- 2. KMP_AFFINITY=granularity=fine,compact
11
- 3. MKL_ENABLE_INSTRUCTIONS=AVX512 (se suportado)
12
- 4. ONEDNN_MAX_CPU_ISA=AMX_INT8 (se suportado)
13
- 5. DNNL_PRIMITIVE_CACHE_CAPACITY=1024
14
- 6. MKL_DYNAMIC=FALSE
15
- 7. IPEX (intel_extension_for_pytorch) — se disponível, ipex.optimize(model)
16
- 8. torch.set_float32_matmul_precision("high")
17
- 9. torch.backends.cudnn.benchmark = True (no-op em CPU)
18
-
19
- Uso:
20
- from cnn_bigru.utils.xeon_runtime import optimize_xeon_environment
21
- N_CORES = optimize_xeon_environment() # chamar ANTES de importar torch
22
- import torch
23
- """
24
- from __future__ import annotations
25
-
26
- import logging
27
- import os
28
- import platform
29
- from typing import Any, Optional
30
-
31
- logger = logging.getLogger(__name__)
32
-
33
- _NUCLEOS_ALOCADOS: Optional[int] = None
34
- _IPEX_AVAILABLE: Optional[bool] = None
35
- _AMX_CAPABLE: Optional[bool] = None
36
- _INIT_DONE: bool = False
37
-
38
-
39
- def _detect_physical_cores() -> int:
40
- """Detecta núcleos físicos respeitando cgroup limits."""
41
- try:
42
- import psutil # type: ignore
43
- n_phys = psutil.cpu_count(logical=False) or 1
44
- except ImportError:
45
- try:
46
- with open("/proc/cpuinfo", "r") as f:
47
- cores = set()
48
- for line in f:
49
- if line.startswith("core id"):
50
- cores.add(line.strip())
51
- n_phys = len(cores) or 1
52
- except OSError:
53
- n_phys = 1
54
- try:
55
- n_affine = len(os.sched_getaffinity(0))
56
- n_logical = os.cpu_count() or 1
57
- if n_affine < n_logical:
58
- n_phys = max(1, n_affine // 2)
59
- else:
60
- n_phys = min(n_phys, n_affine)
61
- except (AttributeError, OSError):
62
- pass
63
- return max(1, n_phys)
64
-
65
-
66
- def _read_cpu_flags() -> str:
67
- try:
68
- with open("/proc/cpuinfo", "r") as f:
69
- for line in f:
70
- if line.startswith("flags"):
71
- return line
72
- except OSError:
73
- pass
74
- return ""
75
-
76
-
77
- def _detect_amx() -> bool:
78
- flags = _read_cpu_flags()
79
- return "amx_int8" in flags and "amx_bf16" in flags
80
-
81
-
82
- def _detect_avx512() -> bool:
83
- flags = _read_cpu_flags()
84
- return "avx512f" in flags
85
-
86
-
87
- def optimize_xeon_environment(force_threads: Optional[int] = None) -> int:
88
- """Ativa todas as otimizações de CPU. Retorna o número de núcleos alocados.
89
-
90
- Deve ser chamada UMA VEZ, antes de importar torch, para que as variáveis
91
- de ambiente tenham efeito. Chamadas subsequentes são no-op (idempotente).
92
- """
93
- global _NUCLEOS_ALOCADOS, _IPEX_AVAILABLE, _AMX_CAPABLE, _INIT_DONE
94
- if _INIT_DONE:
95
- return _NUCLEOS_ALOCADOS or 1
96
-
97
- n_cores = force_threads or _detect_physical_cores()
98
- _NUCLEOS_ALOCADOS = n_cores
99
-
100
- # Threads
101
- os.environ.setdefault("OMP_NUM_THREADS", str(n_cores))
102
- os.environ.setdefault("MKL_NUM_THREADS", str(n_cores))
103
- os.environ.setdefault("OPENBLAS_NUM_THREADS", str(n_cores))
104
- os.environ.setdefault("NUMEXPR_NUM_THREADS", str(n_cores))
105
- os.environ["KMP_AFFINITY"] = "granularity=fine,compact"
106
- os.environ["MKL_DYNAMIC"] = "FALSE"
107
-
108
- # ISA detection
109
- if _detect_avx512():
110
- os.environ["MKL_ENABLE_INSTRUCTIONS"] = "AVX512"
111
- logger.info("AVX512 detectado e ativado para MKL")
112
- if _detect_amx():
113
- os.environ["ONEDNN_MAX_CPU_ISA"] = "AMX_INT8"
114
- os.environ["DNNL_PRIMITIVE_CACHE_CAPACITY"] = "1024"
115
- _AMX_CAPABLE = True
116
- logger.info("AMX_INT8 detectado e ativado para OneDNN")
117
- else:
118
- _AMX_CAPABLE = False
119
-
120
- # Tokenizers parallelism
121
- os.environ.setdefault("TOKENIZERS_PARALLELISM", "true")
122
-
123
- _INIT_DONE = True
124
- logger.info(
125
- "Xeon runtime ativado: cores=%d, avx512=%s, amx=%s, ipex=%s",
126
- n_cores,
127
- _detect_avx512(),
128
- bool(_AMX_CAPABLE),
129
- _check_ipex_available(),
130
- )
131
- return n_cores
132
-
133
-
134
- def _check_ipex_available() -> bool:
135
- global _IPEX_AVAILABLE
136
- if _IPEX_AVAILABLE is not None:
137
- return _IPEX_AVAILABLE
138
- try:
139
- import intel_extension_for_pytorch # noqa: F401
140
- _IPEX_AVAILABLE = True
141
- except ImportError:
142
- _IPEX_AVAILABLE = False
143
- return _IPEX_AVAILABLE
144
-
145
-
146
- def optimize_model_ipex(model: Any) -> Any:
147
- """Aplica ipex.optimize() no modelo, se IPEX estiver disponível."""
148
- if not _check_ipex_available():
149
- logger.info("IPEX indisponível — pulando ipex.optimize()")
150
- return model
151
- try:
152
- import intel_extension_for_pytorch as ipex # type: ignore
153
- model = ipex.optimize(model)
154
- logger.info("Modelo otimizado com IPEX")
155
- except Exception as e:
156
- logger.warning("Falha ao aplicar ipex.optimize(): %s", e)
157
- return model
158
-
159
-
160
- def get_runtime_info() -> dict:
161
- """Retorna informações sobre o runtime ativado."""
162
- return {
163
- "nucleos_alocados": _NUCLEOS_ALOCADOS,
164
- "avx512": _detect_avx512(),
165
- "amx_capable": bool(_AMX_CAPABLE),
166
- "ipex_available": _check_ipex_available(),
167
- "platform": platform.platform(),
168
- "init_done": _INIT_DONE,
169
- }
170
-
171
-
172
- __all__ = [
173
- "optimize_xeon_environment",
174
- "optimize_model_ipex",
175
- "get_runtime_info",
176
- ]