PythonSTB commited on
Commit
bc58406
·
verified ·
1 Parent(s): 8c6cee7

Upload lxml/Test_LXML.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. lxml/Test_LXML.py +290 -0
lxml/Test_LXML.py ADDED
@@ -0,0 +1,290 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ On-device verification for the cross-compiled lxml wheel (norelro).
3
+
4
+ Run after installing:
5
+ pip install lxml-6.1.1-cp312-cp312-android_24_x86_64.whl
6
+
7
+ Usage:
8
+ python Test_lxml.py
9
+
10
+ Exit code 0 = everything required PASSed.
11
+
12
+ Generated by RIMI
13
+ """
14
+ import sys
15
+
16
+ RESULTS = []
17
+
18
+
19
+ def test(name, fn):
20
+ try:
21
+ fn()
22
+ RESULTS.append((name, "PASS", None))
23
+ except Exception as exc:
24
+ RESULTS.append((name, "FAIL", "%s: %s" % (type(exc).__name__, exc)))
25
+ print(" ! %s -> %s: %s" % (name, type(exc).__name__, exc))
26
+
27
+
28
+ def section(title):
29
+ print("=" * 60)
30
+ print(title)
31
+ print("=" * 60)
32
+
33
+
34
+ # ---------------------------------------------------------------------------
35
+ # 1. import / version
36
+ # ---------------------------------------------------------------------------
37
+ def import_lxml():
38
+ import lxml.etree as etree
39
+ print(" lxml.etree version", etree.LXML_VERSION)
40
+ print(" libxml2", etree.LIBXML_VERSION)
41
+ print(" libxslt", etree.LIBXSLT_VERSION)
42
+
43
+
44
+ def import_objectify():
45
+ from lxml import objectify
46
+ print(" lxml.objectify OK")
47
+
48
+
49
+ def import_html():
50
+ from lxml import html
51
+ print(" lxml.html OK")
52
+
53
+
54
+ def import_html_clean():
55
+ from lxml_html_clean import clean
56
+ print(" lxml_html_clean OK")
57
+
58
+
59
+ # ---------------------------------------------------------------------------
60
+ # 2. etree basics
61
+ # ---------------------------------------------------------------------------
62
+ def parse_string():
63
+ from lxml import etree
64
+ xml = b"<root><item id='1'>hello</item><item id='2'>world</item></root>"
65
+ root = etree.fromstring(xml)
66
+ assert root.tag == "root"
67
+ items = root.findall("item")
68
+ assert len(items) == 2
69
+ assert items[0].get("id") == "1"
70
+ assert items[0].text == "hello"
71
+
72
+
73
+ def parse_file():
74
+ from lxml import etree
75
+ xml = b"<data><row><a>1</a><b>2</b></row></data>"
76
+ root = etree.fromstring(xml)
77
+ row = root.find("row")
78
+ assert int(row.find("a").text) == 1
79
+ assert int(row.find("b").text) == 2
80
+
81
+
82
+ def tostring():
83
+ from lxml import etree
84
+ root = etree.Element("parent")
85
+ child = etree.SubElement(root, "child")
86
+ child.text = "text"
87
+ s = etree.tostring(root, encoding="unicode")
88
+ assert "<child>text</child>" in s
89
+
90
+
91
+ def xpath():
92
+ from lxml import etree
93
+ xml = b"<root><a x='1'/><a x='2'/><b/></root>"
94
+ root = etree.fromstring(xml)
95
+ a_list = root.xpath("//a")
96
+ assert len(a_list) == 2
97
+ vals = root.xpath("//a/@x")
98
+ assert vals == ["1", "2"]
99
+ first = root.xpath("//a[@x='2']")
100
+ assert len(first) == 1
101
+
102
+
103
+ def namespaces():
104
+ from lxml import etree
105
+ ns = {"s": "http://example.com/ns"}
106
+ xml = b'<root xmlns:s="http://example.com/ns"><s:item>ok</s:item></root>'
107
+ root = etree.fromstring(xml)
108
+ items = root.xpath("//s:item", namespaces=ns)
109
+ assert len(items) == 1
110
+ assert items[0].text == "ok"
111
+
112
+
113
+ # ---------------------------------------------------------------------------
114
+ # 3. build / modify trees
115
+ # ---------------------------------------------------------------------------
116
+ def build_tree():
117
+ from lxml import etree
118
+ root = etree.Element("root")
119
+ for i in range(5):
120
+ child = etree.SubElement(root, "item")
121
+ child.set("index", str(i))
122
+ child.text = "val_%d" % i
123
+ assert len(root) == 5
124
+ assert root[2].get("index") == "2"
125
+ assert root[4].text == "val_4"
126
+
127
+
128
+ def modify_tree():
129
+ from lxml import etree
130
+ root = etree.Element("root")
131
+ a = etree.SubElement(root, "a")
132
+ b = etree.SubElement(root, "b")
133
+ root.remove(b)
134
+ assert len(root) == 1
135
+ a.text = "modified"
136
+ assert root[0].text == "modified"
137
+
138
+
139
+ # ---------------------------------------------------------------------------
140
+ # 4. HTML parsing
141
+ # ---------------------------------------------------------------------------
142
+ def html_parse():
143
+ from lxml import html
144
+ doc = html.fromstring("<html><body><p>Hello</p><p>World</p></body></html>")
145
+ ps = doc.xpath("//p")
146
+ assert len(ps) == 2
147
+ assert ps[0].text == "Hello"
148
+
149
+
150
+ def html_tostring():
151
+ from lxml import html
152
+ doc = html.fromstring("<html><body><div id='main'>content</div></body></html>")
153
+ s = html.tostring(doc, encoding="unicode")
154
+ assert "content" in s
155
+ assert 'id="main"' in s
156
+
157
+
158
+ # ---------------------------------------------------------------------------
159
+ # 5. html_clean (bundled dep)
160
+ # ---------------------------------------------------------------------------
161
+ def html_clean():
162
+ from lxml_html_clean import Cleaner
163
+ from lxml import html
164
+ dirty = '<html><body><p>safe</p></body></html>'
165
+ doc = html.fromstring(dirty)
166
+ c = Cleaner()
167
+ c.clean_html(doc)
168
+ s = html.tostring(doc, encoding="unicode")
169
+ assert "safe" in s
170
+
171
+
172
+ # ---------------------------------------------------------------------------
173
+ # 6. XSLT
174
+ # ---------------------------------------------------------------------------
175
+ def xslt_transform():
176
+ from lxml import etree
177
+ xml = b"<root><item>1</item><item>2</item></root>"
178
+ xslt_str = b"""<xsl:stylesheet version="1.0"
179
+ xmlns:xsl="http://www.w3.org/1999/XSL/Transform">
180
+ <xsl:template match="/">
181
+ <out><xsl:for-each select="root/item">
182
+ <val><xsl:value-of select="."/></val>
183
+ </xsl:for-each></out>
184
+ </xsl:template>
185
+ </xsl:stylesheet>"""
186
+ xml_doc = etree.fromstring(xml)
187
+ xslt_doc = etree.fromstring(xslt_str)
188
+ transform = etree.XSLT(xslt_doc)
189
+ result = transform(xml_doc)
190
+ s = etree.tostring(result, encoding="unicode")
191
+ assert "<val>1</val>" in s
192
+ assert "<val>2</val>" in s
193
+
194
+
195
+ # ---------------------------------------------------------------------------
196
+ # 7. XML schema validation
197
+ # ---------------------------------------------------------------------------
198
+ def schema_validate():
199
+ from lxml import etree
200
+ schema_xml = b"""<xs:schema xmlns:xs="http://www.w3.org/2001/XMLSchema">
201
+ <xs:element name="root">
202
+ <xs:complexType>
203
+ <xs:sequence>
204
+ <xs:element name="name" type="xs:string"/>
205
+ <xs:element name="age" type="xs:decimal"/>
206
+ </xs:sequence>
207
+ </xs:complexType>
208
+ </xs:element>
209
+ </xs:schema>"""
210
+ schema_doc = etree.fromstring(schema_xml)
211
+ schema = etree.XMLSchema(schema_doc)
212
+
213
+ valid = etree.fromstring(b"<root><name>Alice</name><age>30</age></root>")
214
+ assert schema.validate(valid)
215
+
216
+ invalid = etree.fromstring(b"<root><name>Alice</name></root>")
217
+ assert not schema.validate(invalid)
218
+
219
+
220
+ # ---------------------------------------------------------------------------
221
+ # 8. c14n (canonicalization)
222
+ # ---------------------------------------------------------------------------
223
+ def c14n():
224
+ from lxml import etree
225
+ xml = b'<root attr="1" ><child>text</child></root>'
226
+ root = etree.fromstring(xml)
227
+ c = etree.tostring(root, method="c14n")
228
+ assert b'attr="1"' in c
229
+ assert b"<child>" in c
230
+
231
+
232
+ # ---------------------------------------------------------------------------
233
+ def main():
234
+ section("1. import / version")
235
+ test("import lxml.etree", import_lxml)
236
+ test("import lxml.objectify", import_objectify)
237
+ test("import lxml.html", import_html)
238
+ test("import lxml_html_clean", import_html_clean)
239
+
240
+ section("2. etree basics")
241
+ test("parse fromstring", parse_string)
242
+ test("parse file-like", parse_file)
243
+ test("tostring", tostring)
244
+ test("xpath", xpath)
245
+ test("namespaces", namespaces)
246
+
247
+ section("3. build / modify trees")
248
+ test("build tree", build_tree)
249
+ test("modify tree", modify_tree)
250
+
251
+ section("4. HTML parsing")
252
+ test("html parse", html_parse)
253
+ test("html tostring", html_tostring)
254
+
255
+ section("5. html_clean (bundled dep)")
256
+ test("clean HTML", html_clean)
257
+
258
+ section("6. XSLT")
259
+ test("xslt transform", xslt_transform)
260
+
261
+ section("7. XML schema validation")
262
+ test("schema validate", schema_validate)
263
+
264
+ section("8. c14n")
265
+ test("canonicalization", c14n)
266
+
267
+ print()
268
+ print("=" * 60)
269
+ print("SUMMARY")
270
+ print("=" * 60)
271
+ fails = 0
272
+ for name, status, why in RESULTS:
273
+ mark = " OK" if status == "PASS" else "FAIL"
274
+ print("%s %s" % (mark, name))
275
+ if why:
276
+ print(" -> %s" % why)
277
+ if status == "FAIL":
278
+ fails += 1
279
+ print()
280
+ passed = len(RESULTS) - fails
281
+ print("passed=%d failed=%d" % (passed, fails))
282
+ if fails:
283
+ print("RESULT: FAILED")
284
+ else:
285
+ print("RESULT: PASSED")
286
+ sys.exit(1 if fails else 0)
287
+
288
+
289
+ if __name__ == "__main__":
290
+ main()