PythonSTB commited on
Commit
988e28a
·
verified ·
1 Parent(s): 32ebc50

Upload scikit-learn/Test_ScikitLearn.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. scikit-learn/Test_ScikitLearn.py +252 -0
scikit-learn/Test_ScikitLearn.py ADDED
@@ -0,0 +1,252 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ On-device verification for the cross-compiled scikit-learn wheel.
3
+
4
+ Run after installing:
5
+ pip install scikit_learn-1.7.1-cp312-cp312-android_24_x86_64.whl
6
+
7
+ Usage:
8
+ python Test_ScikitLearn.py [--quick]
9
+
10
+ Exit code 0 = everything required PASSed.
11
+ Requires: numpy, scipy, joblib, threadpoolctl at runtime.
12
+
13
+ Generated by RIMI
14
+ """
15
+ import sys
16
+
17
+ RESULTS = []
18
+
19
+
20
+ def test(name, fn):
21
+ try:
22
+ fn()
23
+ RESULTS.append((name, "PASS", None))
24
+ except NotImplementedError as exc:
25
+ RESULTS.append((name, "SKIP", str(exc)))
26
+ except Exception as exc:
27
+ RESULTS.append((name, "FAIL", "%s: %s" % (type(exc).__name__, exc)))
28
+ print(" ! %s -> %s: %s" % (name, type(exc).__name__, exc))
29
+
30
+
31
+ def section(title):
32
+ print("=" * 60)
33
+ print(title)
34
+ print("=" * 60)
35
+
36
+
37
+ # ---------------------------------------------------------------------------
38
+ # 1. import / version
39
+ # ---------------------------------------------------------------------------
40
+ def import_sklearn():
41
+ import sklearn
42
+ print(" sklearn", sklearn.__version__)
43
+ assert hasattr(sklearn, "__version__")
44
+ assert hasattr(sklearn, "show_versions")
45
+
46
+ def check_c_extension():
47
+ import sklearn
48
+ # Check that at least one Cython extension loads (tree, metrics, etc.)
49
+ from sklearn.tree import _tree
50
+ assert hasattr(_tree, "Tree")
51
+
52
+
53
+ # ---------------------------------------------------------------------------
54
+ # 2. datasets
55
+ # ---------------------------------------------------------------------------
56
+ def load_iris():
57
+ from sklearn.datasets import load_iris
58
+ X, y = load_iris(return_X_y=True)
59
+ assert X.shape == (150, 4)
60
+ assert y.shape == (150,)
61
+
62
+ def load_digits():
63
+ from sklearn.datasets import load_digits
64
+ X, y = load_digits(return_X_y=True)
65
+ assert X.shape[0] == 1797
66
+
67
+
68
+ # ---------------------------------------------------------------------------
69
+ # 3. preprocessing
70
+ # ---------------------------------------------------------------------------
71
+ def scaler_standard():
72
+ from sklearn.preprocessing import StandardScaler
73
+ import numpy as np
74
+ X = np.array([[1, 2], [3, 4], [5, 6]], dtype=np.float64)
75
+ scaler = StandardScaler()
76
+ Xt = scaler.fit_transform(X)
77
+ assert Xt.shape == X.shape
78
+ assert abs(Xt.mean()) < 1e-6
79
+
80
+ def scaler_minmax():
81
+ from sklearn.preprocessing import MinMaxScaler
82
+ import numpy as np
83
+ X = np.array([[1, 2], [3, 4]], dtype=np.float64)
84
+ scaler = MinMaxScaler()
85
+ Xt = scaler.fit_transform(X)
86
+ assert Xt.min() >= 0 and Xt.max() <= 1
87
+
88
+
89
+ # ---------------------------------------------------------------------------
90
+ # 4. decomposition
91
+ # ---------------------------------------------------------------------------
92
+ def pca_test():
93
+ from sklearn.decomposition import PCA
94
+ import numpy as np
95
+ rng = np.random.default_rng(42)
96
+ X = rng.random((50, 10))
97
+ pca = PCA(n_components=2)
98
+ Xt = pca.fit_transform(X)
99
+ assert Xt.shape == (50, 2)
100
+
101
+
102
+ # ---------------------------------------------------------------------------
103
+ # 5. cluster
104
+ # ---------------------------------------------------------------------------
105
+ def kmeans_test():
106
+ from sklearn.cluster import KMeans
107
+ import numpy as np
108
+ rng = np.random.default_rng(42)
109
+ X = rng.random((30, 2))
110
+ km = KMeans(n_clusters=3, n_init=10, random_state=42)
111
+ labels = km.fit_predict(X)
112
+ assert labels.shape == (30,)
113
+ assert len(set(labels)) == 3
114
+
115
+
116
+ # ---------------------------------------------------------------------------
117
+ # 6. linear model
118
+ # ---------------------------------------------------------------------------
119
+ def logistic_regression():
120
+ from sklearn.linear_model import LogisticRegression
121
+ from sklearn.datasets import load_iris
122
+ X, y = load_iris(return_X_y=True)
123
+ # binary iris (first 100 samples, 2 classes)
124
+ Xb, yb = X[:100], y[:100]
125
+ clf = LogisticRegression(max_iter=200)
126
+ clf.fit(Xb, yb)
127
+ pred = clf.predict(Xb[:5])
128
+ assert pred.shape == (5,)
129
+
130
+ def linear_regression():
131
+ from sklearn.linear_model import LinearRegression
132
+ import numpy as np
133
+ X = np.array([[1], [2], [3], [4]], dtype=np.float64)
134
+ y = np.array([2, 4, 6, 8], dtype=np.float64)
135
+ reg = LinearRegression()
136
+ reg.fit(X, y)
137
+ pred = reg.predict([[5]])
138
+ assert abs(pred[0] - 10) < 1e-3
139
+
140
+
141
+ # ---------------------------------------------------------------------------
142
+ # 7. ensemble
143
+ # ---------------------------------------------------------------------------
144
+ def random_forest():
145
+ from sklearn.ensemble import RandomForestClassifier
146
+ from sklearn.datasets import load_iris
147
+ X, y = load_iris(return_X_y=True)
148
+ clf = RandomForestClassifier(n_estimators=10, random_state=42)
149
+ clf.fit(X[:100], y[:100])
150
+ pred = clf.predict(X[:5])
151
+ assert pred.shape == (5,)
152
+
153
+
154
+ # ---------------------------------------------------------------------------
155
+ # 8. metrics
156
+ # ---------------------------------------------------------------------------
157
+ def metrics_test():
158
+ from sklearn.metrics import accuracy_score, mean_squared_error
159
+ import numpy as np
160
+ y_true = np.array([0, 1, 1, 0])
161
+ y_pred = np.array([0, 1, 0, 0])
162
+ acc = accuracy_score(y_true, y_pred)
163
+ assert 0 <= acc <= 1
164
+ mse = mean_squared_error([1, 2, 3], [1, 2, 3])
165
+ assert mse == 0
166
+
167
+
168
+ # ---------------------------------------------------------------------------
169
+ # 9. model_selection
170
+ # ---------------------------------------------------------------------------
171
+ def train_test_split():
172
+ from sklearn.model_selection import train_test_split
173
+ import numpy as np
174
+ X = np.random.rand(20, 4)
175
+ y = np.random.randint(0, 2, 20)
176
+ Xtr, Xte, ytr, yte = train_test_split(X, y, test_size=0.25, random_state=42)
177
+ assert Xtr.shape[0] == 15 and Xte.shape[0] == 5
178
+
179
+ def cross_val():
180
+ from sklearn.model_selection import cross_val_score
181
+ from sklearn.linear_model import LogisticRegression
182
+ from sklearn.datasets import load_iris
183
+ X, y = load_iris(return_X_y=True)
184
+ clf = LogisticRegression(max_iter=200)
185
+ scores = cross_val_score(clf, X[:100], y[:100], cv=3)
186
+ assert len(scores) == 3
187
+
188
+
189
+ # ---------------------------------------------------------------------------
190
+ def main():
191
+ quick = "--quick" in sys.argv
192
+ section("1. import / version")
193
+ test("import sklearn", import_sklearn)
194
+ test("C extension _tree", check_c_extension)
195
+
196
+ section("2. datasets")
197
+ test("load_iris", load_iris)
198
+ test("load_digits", load_digits)
199
+
200
+ section("3. preprocessing")
201
+ test("StandardScaler", scaler_standard)
202
+ test("MinMaxScaler", scaler_minmax)
203
+
204
+ section("4. decomposition")
205
+ test("PCA", pca_test)
206
+
207
+ section("5. cluster")
208
+ test("KMeans", kmeans_test)
209
+
210
+ section("6. linear_model")
211
+ test("LogisticRegression", logistic_regression)
212
+ test("LinearRegression", linear_regression)
213
+
214
+ section("7. ensemble")
215
+ test("RandomForest", random_forest)
216
+
217
+ section("8. metrics")
218
+ test("metrics", metrics_test)
219
+
220
+ section("9. model_selection")
221
+ test("train_test_split", train_test_split)
222
+ test("cross_val_score", cross_val)
223
+
224
+ print()
225
+ print("=" * 60)
226
+ print("SUMMARY")
227
+ print("=" * 60)
228
+ fails = 0
229
+ skips = 0
230
+ for name, status, why in RESULTS:
231
+ mark = " OK" if status == "PASS" else (" SKIP" if status == "SKIP" else "FAIL")
232
+ print("%s %s" % (mark, name))
233
+ if why:
234
+ print(" -> %s" % why)
235
+ if status == "FAIL":
236
+ fails += 1
237
+ elif status == "SKIP":
238
+ skips += 1
239
+ print()
240
+ passed = len(RESULTS) - fails - skips
241
+ print("passed=%d skipped=%d failed=%d" % (passed, skips, fails))
242
+ if fails:
243
+ print("RESULT: FAILED")
244
+ elif skips and not quick:
245
+ print("RESULT: PASSED (with informational skips)")
246
+ else:
247
+ print("RESULT: PASSED")
248
+ sys.exit(1 if fails else 0)
249
+
250
+
251
+ if __name__ == "__main__":
252
+ main()