Coverage for gws-app/gws/lib/xmlx/parser.py: 94%

133 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-05 13:35 +0200

1"""Expat-based XML parser.""" 

2 

3from typing import Optional 

4 

5import re 

6import pyexpat 

7 

8import gws 

9 

10from . import error, element, namespace 

11 

12 

13def from_path(path: str, opts: Optional[gws.XmlOptions] = None) -> gws.XmlElement: 

14 """Parse an XML file. 

15 

16 Args: 

17 path: Path to the file. 

18 opts: Parsing options (``removeNamespaces``, ``compactWhitespace``). 

19 

20 Returns: 

21 The root element. 

22 

23 Raises: 

24 ParseError: If the document is malformed, contains entity declarations or cannot be decoded. 

25 """ 

26 

27 with open(path, 'rb') as fp: 

28 inp = fp.read() 

29 return _parse(inp, opts) 

30 

31 

32def from_string(inp: str | bytes, opts: Optional[gws.XmlOptions] = None) -> gws.XmlElement: 

33 """Parse an XML document from a string or bytes. 

34 

35 Bytes are decoded as UTF-8, then with the declared encoding, then as Latin-1. 

36 

37 Args: 

38 inp: The document as a string or bytes. 

39 opts: Parsing options (``removeNamespaces``, ``compactWhitespace``). 

40 

41 Returns: 

42 The root element. 

43 

44 Raises: 

45 ParseError: If the document is malformed, contains entity declarations or cannot be decoded. 

46 """ 

47 

48 return _parse(inp, opts) 

49 

50 

51## 

52 

53 

54def _parse(inp, opts: Optional[gws.XmlOptions] = None) -> gws.XmlElement: 

55 """Decode the input and run expat on it.""" 

56 

57 inp = _decode_input(inp) 

58 target = _ParserTarget(opts or gws.XmlOptions()) 

59 

60 parser = pyexpat.ParserCreate() 

61 parser.buffer_text = True 

62 parser.StartElementHandler = target.start 

63 parser.EndElementHandler = target.end 

64 parser.CharacterDataHandler = target.data 

65 parser.EntityDeclHandler = target.entity_decl 

66 

67 try: 

68 parser.Parse(inp, True) 

69 except pyexpat.ExpatError as exc: 

70 raise error.ParseError(exc.args[0]) from exc 

71 

72 if target.root is None: 

73 raise error.ParseError('no root element') 

74 

75 return target.root 

76 

77 

78class _ParserTarget: 

79 """Expat event handler that builds the element tree and tracks namespace scopes.""" 

80 

81 def __init__(self, opts: gws.XmlOptions): 

82 """Create the handler. 

83 

84 Args: 

85 opts: Parsing options. 

86 """ 

87 

88 self.stack = [] 

89 self.scopes = [] 

90 self.root = None 

91 self.opts = opts 

92 

93 def make(self, tag: str, attrib: dict) -> element.XmlElement: 

94 """Create an element from a raw tag and raw attributes. 

95 

96 Without ``removeNamespaces``, ``xmlns`` declarations are stored in ``XmlElement.namespaces`` 

97 and pushed as a new scope, and prefixed names are resolved to Clark names. 

98 

99 Args: 

100 tag: Tag name as in the document. 

101 attrib: Attributes as in the document. 

102 

103 Returns: 

104 The new element. 

105 """ 

106 

107 if self.opts.removeNamespaces: 

108 el = element.XmlElement(namespace.plain_name(tag)) 

109 for key, val in attrib.items(): 

110 _, prefix, pname = namespace.parse_name(key) 

111 if key == namespace.XMLNS or prefix == namespace.XMLNS: 

112 continue 

113 el.attrib[pname] = val 

114 return el 

115 

116 namespaces = [] 

117 atts = {} 

118 to_resolve = [] 

119 

120 for key, val in attrib.items(): 

121 _, prefix, pname = namespace.parse_name(key) 

122 if key == namespace.XMLNS: 

123 namespaces.append(namespace.new('', val)) 

124 elif prefix == namespace.XMLNS: 

125 namespaces.append(namespace.new(pname, val)) 

126 elif prefix == '': 

127 atts[pname] = val 

128 else: 

129 to_resolve.append((prefix, pname, val)) 

130 

131 if namespaces: 

132 self.scopes.append(namespaces) 

133 

134 for prefix, pname, val in to_resolve: 

135 atts[namespace.full_name(pname, self.uri_for(prefix))] = val 

136 

137 _, prefix, pname = namespace.parse_name(tag) 

138 el = element.XmlElement(namespace.full_name(pname, self.uri_for(prefix)), atts) 

139 el.namespaces = namespaces 

140 

141 return el 

142 

143 def uri_for(self, prefix: str) -> str: 

144 """Resolve a prefix against the current namespace scopes. 

145 

146 Args: 

147 prefix: Namespace prefix, empty for the default namespace. 

148 

149 Returns: 

150 The namespace URI, an empty string for an undeclared default namespace, 

151 or ``adhoc:<prefix>`` for an undeclared prefix. 

152 """ 

153 

154 if prefix == namespace.XML: 

155 return namespace.XML_URI 

156 for scope in reversed(self.scopes): 

157 for ns in scope: 

158 if ns.prefix == prefix: 

159 return ns.uri 

160 return '' if prefix == '' else namespace.ADHOC + prefix 

161 

162 def start(self, tag: str, attrib: dict): 

163 """Handle a start tag. 

164 

165 Args: 

166 tag: Tag name as in the document. 

167 attrib: Attributes as in the document. 

168 """ 

169 

170 el = self.make(tag, attrib) 

171 if self.stack: 

172 self.stack[-1].append(el) 

173 else: 

174 self.root = el 

175 self.stack.append(el) 

176 

177 def end(self, tag): 

178 """Handle an end tag. 

179 

180 Args: 

181 tag: Tag name as in the document. 

182 """ 

183 

184 el = self.stack.pop() 

185 if el.namespaces and not self.opts.removeNamespaces: 

186 self.scopes.pop() 

187 

188 def data(self, text): 

189 """Handle character data, appending it to the text or the tail of the last child. 

190 

191 Args: 

192 text: Character data. 

193 """ 

194 

195 if not self.stack: 

196 return 

197 if self.opts.compactWhitespace: 

198 text = ' '.join(text.strip().split()) 

199 if not text: 

200 return 

201 top = self.stack[-1] 

202 if len(top) > 0: 

203 top[-1].tail += text 

204 else: 

205 top.text += text 

206 

207 def entity_decl(self, *args): 

208 """Handle an entity declaration. 

209 

210 Raises: 

211 ParseError: Always, entity declarations are not allowed. 

212 """ 

213 

214 raise error.ParseError('entity declarations are not allowed') 

215 

216 

217def _decode_input(inp) -> str: 

218 """Decode the input to a string without the XML declaration.""" 

219 

220 # A document can be declared ISO-8859-1, but actually be UTF-8 and vice versa. 

221 # Therefore, don't let expat do the decoding, always give it a `str` 

222 # and remove the xml declaration with the (possibly incorrect) encoding. 

223 

224 if isinstance(inp, bytes): 

225 return _decode_bytes_input(inp) 

226 if isinstance(inp, str): 

227 return _decode_str_input(inp) 

228 raise error.ParseError(f'invalid input type {type(inp)}') 

229 

230 

231def _decode_bytes_input(inp: bytes) -> str: 

232 """Decode bytes as UTF-8, then with the declared encoding, then as Latin-1.""" 

233 

234 inp = inp.removeprefix(_BOM).strip() 

235 

236 declared = '' 

237 

238 if inp.startswith(b'<?xml'): 

239 try: 

240 end = inp.index(b'?>') 

241 except ValueError: 

242 raise error.ParseError('invalid XML declaration') 

243 head = inp[:end].decode('ascii', errors='replace').lower() 

244 m = re.search(r'encoding\s*=\s*(\S+)', head) 

245 if m: 

246 declared = m.group(1).strip('\'"') 

247 inp = inp[end + 2 :] 

248 

249 # UTF-8 is strict and fails on non-UTF-8 input, Latin-1 never fails. 

250 

251 encodings = ['utf-8'] 

252 if declared and declared not in encodings: 

253 encodings.append(declared) 

254 if 'iso-8859-1' not in encodings: 

255 encodings.append('iso-8859-1') 

256 

257 for enc in encodings: 

258 try: 

259 return inp.decode(encoding=enc, errors='strict') 

260 except (LookupError, UnicodeDecodeError): 

261 pass 

262 

263 raise error.ParseError(f'invalid document encoding, tried {",".join(encodings)}') 

264 

265 

266def _decode_str_input(inp: str) -> str: 

267 """Strip a BOM and the XML declaration from a string.""" 

268 

269 inp = inp.lstrip('\ufeff').strip() 

270 if inp.startswith('<?xml'): 

271 try: 

272 end = inp.index('?>') 

273 except ValueError: 

274 raise error.ParseError('invalid XML declaration') 

275 return inp[end + 2 :] 

276 

277 return inp 

278 

279 

280_BOM = b'\xef\xbb\xbf'