PythonSTB commited on
Commit
8c6cee7
·
verified ·
1 Parent(s): 2118f3d

Upload lxml/LXML_USER_GUIDE.txt with huggingface_hub

Browse files
Files changed (1) hide show
  1. lxml/LXML_USER_GUIDE.txt +331 -0
lxml/LXML_USER_GUIDE.txt ADDED
@@ -0,0 +1,331 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # lxml User Guide — PythonSTB Android App
2
+ Generated by RIMI
3
+
4
+ ## What is lxml?
5
+
6
+ lxml is the most feature-rich and performance-optimized library for processing XML
7
+ and HTML in Python. It wraps the C libraries libxml2 and libxslt, providing a
8
+ Pythonic API while maintaining near-native speed.
9
+
10
+ ---
11
+
12
+ ## Why We Need It
13
+
14
+ PythonSTB uses lxml for:
15
+
16
+ 1. **M3U playlist parsing** — Parse HLS/DASH playlists (XML-based)
17
+ 2. **EPG (Electronic Program Guide)** — Parse XMLTV format EPG data
18
+ 3. **HTML scraping** — Extract data from web pages (stream URLs, titles)
19
+ 4. **SOAP/XML-RPC communication** — Some IPTV APIs use XML responses
20
+ 5. **XSLT transforms** — Convert between XML formats
21
+
22
+ ---
23
+
24
+ ## Installed Packages
25
+
26
+ | Package | Version | Description |
27
+ |---------|---------|-------------|
28
+ | lxml | 6.1.1 | Core XML/HTML processing |
29
+ | lxml_html_clean | 0.4.5 | HTML sanitizer (bundled) |
30
+
31
+ ---
32
+
33
+ ## Basic Usage
34
+
35
+ ### Parsing XML
36
+
37
+ ```python
38
+ from lxml import etree
39
+
40
+ # Parse from string
41
+ xml_string = '''<?xml version="1.0" encoding="UTF-8"?>
42
+ <channel>
43
+ <name>Channel 1</name>
44
+ <url>http://example.com/stream.m3u8</url>
45
+ <logo>http://example.com/logo.png</logo>
46
+ </channel>'''
47
+
48
+ root = etree.fromstring(xml_string.encode())
49
+ print(root.find('name').text) # Channel 1
50
+ print(root.find('url').text) # http://example.com/stream.m3u8
51
+
52
+ # Parse from file
53
+ tree = etree.parse('playlist.xml')
54
+ root = tree.getroot()
55
+ ```
56
+
57
+ ### Parsing HTML
58
+
59
+ ```python
60
+ from lxml import html
61
+
62
+ # Parse HTML string
63
+ page = html.fromstring('''
64
+ <html>
65
+ <body>
66
+ <div class="stream">
67
+ <h2>Channel Name</h2>
68
+ <a href="http://example.com/stream.m3u8">Watch</a>
69
+ </div>
70
+ </body>
71
+ </html>''')
72
+
73
+ # XPath to find elements
74
+ channels = page.xpath('//div[@class="stream"]')
75
+ for ch in channels:
76
+ name = ch.xpath('.//h2/text()')[0]
77
+ url = ch.xpath('.//a/@href')[0]
78
+ print(f"{name}: {url}")
79
+ ```
80
+
81
+ ### M3U Playlist Parsing
82
+
83
+ ```python
84
+ from lxml import etree
85
+
86
+ def parse_m3u(content):
87
+ """Parse M3U/M3U8 playlist into list of channels."""
88
+ channels = []
89
+ lines = content.strip().split('\n')
90
+
91
+ i = 0
92
+ while i < len(lines):
93
+ line = lines[i].strip()
94
+
95
+ if line.startswith('#EXTINF:'):
96
+ # Parse info line
97
+ info = line[8:] # Remove #EXTINF:
98
+ attrs = {}
99
+
100
+ # Extract duration
101
+ if ',' in info:
102
+ duration, name = info.rsplit(',', 1)
103
+ attrs['duration'] = duration
104
+ attrs['name'] = name.strip()
105
+
106
+ # Next line should be the URL
107
+ if i + 1 < len(lines):
108
+ url = lines[i + 1].strip()
109
+ if not url.startswith('#'):
110
+ attrs['url'] = url
111
+ channels.append(attrs)
112
+ i += 2
113
+ continue
114
+
115
+ i += 1
116
+
117
+ return channels
118
+
119
+ # Usage
120
+ with open('playlist.m3u', 'r', encoding='utf-8') as f:
121
+ content = f.read()
122
+
123
+ channels = parse_m3u(content)
124
+ for ch in channels:
125
+ print(f"{ch.get('name', 'Unknown')}: {ch.get('url', 'N/A')}")
126
+ ```
127
+
128
+ ### EPG (XMLTV) Parsing
129
+
130
+ ```python
131
+ from lxml import etree
132
+ from datetime import datetime
133
+
134
+ def parse_epg(xml_content):
135
+ """Parse XMLTV format EPG data."""
136
+ root = etree.fromstring(xml_content.encode())
137
+
138
+ channels = {}
139
+ for channel in root.findall('.//channel'):
140
+ ch_id = channel.get('id')
141
+ display_name = channel.find('display-name').text
142
+ icon = channel.find('icon')
143
+ icon_url = icon.get('src') if icon is not None else None
144
+
145
+ channels[ch_id] = {
146
+ 'name': display_name,
147
+ 'icon': icon_url
148
+ }
149
+
150
+ programmes = []
151
+ for prog in root.findall('.//programme'):
152
+ programmes.append({
153
+ 'channel': prog.get('channel'),
154
+ 'start': prog.get('start'),
155
+ 'stop': prog.get('stop'),
156
+ 'title': prog.find('title').text if prog.find('title') is not None else '',
157
+ 'desc': prog.find('desc').text if prog.find('desc') is not None else ''
158
+ })
159
+
160
+ return channels, programmes
161
+ ```
162
+
163
+ ### XPath Examples
164
+
165
+ ```python
166
+ from lxml import etree
167
+
168
+ tree = etree.parse('data.xml')
169
+ root = tree.getroot()
170
+
171
+ # Find all elements with attribute
172
+ elements = root.xpath('//item[@type="channel"]')
173
+
174
+ # Find with text content
175
+ items = root.xpath('//item[name="Channel 1"]')
176
+
177
+ # Find with contains
178
+ links = root.xpath('//a[contains(@href, "m3u8")]')
179
+
180
+ # Find with multiple conditions
181
+ results = root.xpath('//channel[@lang="en" and @country="US"]')
182
+
183
+ # Find parent element
184
+ child = root.find('.//child')
185
+ parent = child.getparent()
186
+
187
+ # Find siblings
188
+ next_sibling = child.getnext()
189
+ prev_sibling = child.getprevious()
190
+ ```
191
+
192
+ ---
193
+
194
+ ## Features Used in App
195
+
196
+ | Feature | Module | Usage |
197
+ |---------|--------|-------|
198
+ | XML parsing | etree | EPG, API responses |
199
+ | HTML parsing | html | Web scraping |
200
+ | XPath | etree | Data extraction |
201
+ | ElementTree API | etree | Tree manipulation |
202
+ | HTML cleaning | lxml_html_clean | Sanitize scraped HTML |
203
+ | XSLT | etree | Format conversion |
204
+ | C14N | etree | Canonical XML output |
205
+
206
+ ---
207
+
208
+ ## Troubleshooting
209
+
210
+ ### Import Error
211
+
212
+ ```python
213
+ # If lxml fails to import:
214
+ try:
215
+ from lxml import etree
216
+ print("lxml is installed correctly")
217
+ except ImportError as e:
218
+ print(f"lxml import error: {e}")
219
+ # Check if .so files exist and have correct permissions
220
+ ```
221
+
222
+ ### Encoding Issues
223
+
224
+ ```python
225
+ # Always specify encoding when parsing
226
+ tree = etree.parse('file.xml') # Auto-detect
227
+ tree = etree.parse(open('file.xml', 'rb')) # Binary mode
228
+
229
+ # Force encoding
230
+ content = open('file.xml', 'r', encoding='latin-1').read()
231
+ root = etree.fromstring(content.encode('utf-8'))
232
+ ```
233
+
234
+ ### Memory Errors
235
+
236
+ ```python
237
+ # Use iterparse for large files
238
+ import xml.etree.ElementTree as ET # fallback for huge files
239
+ # Or use lxml's iterparse
240
+ context = etree.iterparse('large.xml', events=('end',), tag='item')
241
+ for event, elem in context:
242
+ process(elem)
243
+ elem.clear() # Free memory
244
+ ```
245
+
246
+ ---
247
+
248
+ ## Related Packages
249
+
250
+ | Package | Purpose |
251
+ |---------|---------|
252
+ | lxml_html_clean | HTML sanitization |
253
+ | cssselect | CSS selector support |
254
+ | cssutils | CSS parsing |
255
+ | beautifulsoup4 | Alternative HTML parser |
256
+
257
+ ---
258
+
259
+ ## Android-Specific Notes
260
+
261
+ - lxml is compiled with `-Wl,-z,norelro` for Android compatibility
262
+ - Uses 16KB page alignment for Android 15+ support
263
+ - Links statically against libxml2 and libxslt (no external .so files)
264
+ - All `.so` files include `librimi.so` dependency
265
+
266
+ Generated by RIMI
267
+
268
+ ---
269
+
270
+ ## App Integration
271
+
272
+ ```python
273
+ # Example: Full M3U parser with EPG support
274
+ from lxml import etree
275
+ from lxml.html import fromstring
276
+
277
+ class IPTVParser:
278
+ def __init__(self):
279
+ self.channels = []
280
+
281
+ def parse_m3u(self, url):
282
+ """Fetch and parse M3U playlist."""
283
+ import urllib.request
284
+ with urllib.request.urlopen(url) as response:
285
+ content = response.read().decode('utf-8')
286
+ return self._parse_m3u_content(content)
287
+
288
+ def _parse_m3u_content(self, content):
289
+ """Parse M3U content string."""
290
+ channels = []
291
+ lines = content.strip().split('\n')
292
+
293
+ for i, line in enumerate(lines):
294
+ if line.startswith('#EXTINF:'):
295
+ # Parse attributes
296
+ attrs = {}
297
+ if ',' in line:
298
+ attrs['name'] = line.split(',')[-1].strip()
299
+
300
+ # Get URL (next non-comment line)
301
+ url = lines[i + 1].strip() if i + 1 < len(lines) else ''
302
+ if url and not url.startswith('#'):
303
+ attrs['url'] = url
304
+ channels.append(attrs)
305
+
306
+ return channels
307
+
308
+ def parse_epg(self, xml_content):
309
+ """Parse XMLTV EPG data."""
310
+ root = etree.fromstring(xml_content.encode())
311
+ return {
312
+ 'channels': self._parse_epg_channels(root),
313
+ 'programmes': self._parse_epg_programmes(root)
314
+ }
315
+
316
+ def _parse_epg_channels(self, root):
317
+ channels = {}
318
+ for ch in root.findall('.//channel'):
319
+ channels[ch.get('id')] = ch.find('display-name').text
320
+ return channels
321
+
322
+ def _parse_epg_programmes(self, root):
323
+ programmes = []
324
+ for prog in root.findall('.//programme'):
325
+ programmes.append({
326
+ 'channel': prog.get('channel'),
327
+ 'title': prog.find('title').text,
328
+ 'start': prog.get('start')
329
+ })
330
+ return programmes
331
+ ```