Coverage for gws-app/gws/lib/xmlx/parser.py: 94%
133 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-05 13:35 +0200
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-05 13:35 +0200
1"""Expat-based XML parser."""
3from typing import Optional
5import re
6import pyexpat
8import gws
10from . import error, element, namespace
13def from_path(path: str, opts: Optional[gws.XmlOptions] = None) -> gws.XmlElement:
14 """Parse an XML file.
16 Args:
17 path: Path to the file.
18 opts: Parsing options (``removeNamespaces``, ``compactWhitespace``).
20 Returns:
21 The root element.
23 Raises:
24 ParseError: If the document is malformed, contains entity declarations or cannot be decoded.
25 """
27 with open(path, 'rb') as fp:
28 inp = fp.read()
29 return _parse(inp, opts)
32def from_string(inp: str | bytes, opts: Optional[gws.XmlOptions] = None) -> gws.XmlElement:
33 """Parse an XML document from a string or bytes.
35 Bytes are decoded as UTF-8, then with the declared encoding, then as Latin-1.
37 Args:
38 inp: The document as a string or bytes.
39 opts: Parsing options (``removeNamespaces``, ``compactWhitespace``).
41 Returns:
42 The root element.
44 Raises:
45 ParseError: If the document is malformed, contains entity declarations or cannot be decoded.
46 """
48 return _parse(inp, opts)
51##
54def _parse(inp, opts: Optional[gws.XmlOptions] = None) -> gws.XmlElement:
55 """Decode the input and run expat on it."""
57 inp = _decode_input(inp)
58 target = _ParserTarget(opts or gws.XmlOptions())
60 parser = pyexpat.ParserCreate()
61 parser.buffer_text = True
62 parser.StartElementHandler = target.start
63 parser.EndElementHandler = target.end
64 parser.CharacterDataHandler = target.data
65 parser.EntityDeclHandler = target.entity_decl
67 try:
68 parser.Parse(inp, True)
69 except pyexpat.ExpatError as exc:
70 raise error.ParseError(exc.args[0]) from exc
72 if target.root is None:
73 raise error.ParseError('no root element')
75 return target.root
78class _ParserTarget:
79 """Expat event handler that builds the element tree and tracks namespace scopes."""
81 def __init__(self, opts: gws.XmlOptions):
82 """Create the handler.
84 Args:
85 opts: Parsing options.
86 """
88 self.stack = []
89 self.scopes = []
90 self.root = None
91 self.opts = opts
93 def make(self, tag: str, attrib: dict) -> element.XmlElement:
94 """Create an element from a raw tag and raw attributes.
96 Without ``removeNamespaces``, ``xmlns`` declarations are stored in ``XmlElement.namespaces``
97 and pushed as a new scope, and prefixed names are resolved to Clark names.
99 Args:
100 tag: Tag name as in the document.
101 attrib: Attributes as in the document.
103 Returns:
104 The new element.
105 """
107 if self.opts.removeNamespaces:
108 el = element.XmlElement(namespace.plain_name(tag))
109 for key, val in attrib.items():
110 _, prefix, pname = namespace.parse_name(key)
111 if key == namespace.XMLNS or prefix == namespace.XMLNS:
112 continue
113 el.attrib[pname] = val
114 return el
116 namespaces = []
117 atts = {}
118 to_resolve = []
120 for key, val in attrib.items():
121 _, prefix, pname = namespace.parse_name(key)
122 if key == namespace.XMLNS:
123 namespaces.append(namespace.new('', val))
124 elif prefix == namespace.XMLNS:
125 namespaces.append(namespace.new(pname, val))
126 elif prefix == '':
127 atts[pname] = val
128 else:
129 to_resolve.append((prefix, pname, val))
131 if namespaces:
132 self.scopes.append(namespaces)
134 for prefix, pname, val in to_resolve:
135 atts[namespace.full_name(pname, self.uri_for(prefix))] = val
137 _, prefix, pname = namespace.parse_name(tag)
138 el = element.XmlElement(namespace.full_name(pname, self.uri_for(prefix)), atts)
139 el.namespaces = namespaces
141 return el
143 def uri_for(self, prefix: str) -> str:
144 """Resolve a prefix against the current namespace scopes.
146 Args:
147 prefix: Namespace prefix, empty for the default namespace.
149 Returns:
150 The namespace URI, an empty string for an undeclared default namespace,
151 or ``adhoc:<prefix>`` for an undeclared prefix.
152 """
154 if prefix == namespace.XML:
155 return namespace.XML_URI
156 for scope in reversed(self.scopes):
157 for ns in scope:
158 if ns.prefix == prefix:
159 return ns.uri
160 return '' if prefix == '' else namespace.ADHOC + prefix
162 def start(self, tag: str, attrib: dict):
163 """Handle a start tag.
165 Args:
166 tag: Tag name as in the document.
167 attrib: Attributes as in the document.
168 """
170 el = self.make(tag, attrib)
171 if self.stack:
172 self.stack[-1].append(el)
173 else:
174 self.root = el
175 self.stack.append(el)
177 def end(self, tag):
178 """Handle an end tag.
180 Args:
181 tag: Tag name as in the document.
182 """
184 el = self.stack.pop()
185 if el.namespaces and not self.opts.removeNamespaces:
186 self.scopes.pop()
188 def data(self, text):
189 """Handle character data, appending it to the text or the tail of the last child.
191 Args:
192 text: Character data.
193 """
195 if not self.stack:
196 return
197 if self.opts.compactWhitespace:
198 text = ' '.join(text.strip().split())
199 if not text:
200 return
201 top = self.stack[-1]
202 if len(top) > 0:
203 top[-1].tail += text
204 else:
205 top.text += text
207 def entity_decl(self, *args):
208 """Handle an entity declaration.
210 Raises:
211 ParseError: Always, entity declarations are not allowed.
212 """
214 raise error.ParseError('entity declarations are not allowed')
217def _decode_input(inp) -> str:
218 """Decode the input to a string without the XML declaration."""
220 # A document can be declared ISO-8859-1, but actually be UTF-8 and vice versa.
221 # Therefore, don't let expat do the decoding, always give it a `str`
222 # and remove the xml declaration with the (possibly incorrect) encoding.
224 if isinstance(inp, bytes):
225 return _decode_bytes_input(inp)
226 if isinstance(inp, str):
227 return _decode_str_input(inp)
228 raise error.ParseError(f'invalid input type {type(inp)}')
231def _decode_bytes_input(inp: bytes) -> str:
232 """Decode bytes as UTF-8, then with the declared encoding, then as Latin-1."""
234 inp = inp.removeprefix(_BOM).strip()
236 declared = ''
238 if inp.startswith(b'<?xml'):
239 try:
240 end = inp.index(b'?>')
241 except ValueError:
242 raise error.ParseError('invalid XML declaration')
243 head = inp[:end].decode('ascii', errors='replace').lower()
244 m = re.search(r'encoding\s*=\s*(\S+)', head)
245 if m:
246 declared = m.group(1).strip('\'"')
247 inp = inp[end + 2 :]
249 # UTF-8 is strict and fails on non-UTF-8 input, Latin-1 never fails.
251 encodings = ['utf-8']
252 if declared and declared not in encodings:
253 encodings.append(declared)
254 if 'iso-8859-1' not in encodings:
255 encodings.append('iso-8859-1')
257 for enc in encodings:
258 try:
259 return inp.decode(encoding=enc, errors='strict')
260 except (LookupError, UnicodeDecodeError):
261 pass
263 raise error.ParseError(f'invalid document encoding, tried {",".join(encodings)}')
266def _decode_str_input(inp: str) -> str:
267 """Strip a BOM and the XML declaration from a string."""
269 inp = inp.lstrip('\ufeff').strip()
270 if inp.startswith('<?xml'):
271 try:
272 end = inp.index('?>')
273 except ValueError:
274 raise error.ParseError('invalid XML declaration')
275 return inp[end + 2 :]
277 return inp
280_BOM = b'\xef\xbb\xbf'