Commit | Line | Data |
---|---|---|
1cd7adc6 | 1 | ;;; xml.el --- XML parser |
47db06aa | 2 | |
1300e272 | 3 | ;; Copyright (C) 2000, 2001 Free Software Foundation, Inc. |
47db06aa GM |
4 | |
5 | ;; Author: Emmanuel Briot <briot@gnat.com> | |
6 | ;; Maintainer: Emmanuel Briot <briot@gnat.com> | |
7 | ;; Keywords: xml | |
8 | ||
9 | ;; This file is part of GNU Emacs. | |
10 | ||
11 | ;; GNU Emacs is free software; you can redistribute it and/or modify | |
12 | ;; it under the terms of the GNU General Public License as published by | |
13 | ;; the Free Software Foundation; either version 2, or (at your option) | |
14 | ;; any later version. | |
15 | ||
16 | ;; GNU Emacs is distributed in the hope that it will be useful, | |
17 | ;; but WITHOUT ANY WARRANTY; without even the implied warranty of | |
18 | ;; MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the | |
19 | ;; GNU General Public License for more details. | |
20 | ||
21 | ;; You should have received a copy of the GNU General Public License | |
22 | ;; along with GNU Emacs; see the file COPYING. If not, write to the | |
23 | ;; Free Software Foundation, Inc., 59 Temple Place - Suite 330, | |
24 | ;; Boston, MA 02111-1307, USA. | |
25 | ||
26 | ;;; Commentary: | |
27 | ||
28 | ;; This file contains a full XML parser. It parses a file, and returns a list | |
29 | ;; that can be used internally by any other lisp file. | |
30 | ;; See some example in todo.el | |
31 | ||
32 | ;;; FILE FORMAT | |
33 | ||
34 | ;; It does not parse the DTD, if present in the XML file, but knows how to | |
35 | ;; ignore it. The XML file is assumed to be well-formed. In case of error, the | |
36 | ;; parsing stops and the XML file is shown where the parsing stopped. | |
37 | ;; | |
38 | ;; It also knows how to ignore comments, as well as the special ?xml? tag | |
39 | ;; in the XML file. | |
40 | ;; | |
41 | ;; The XML file should have the following format: | |
653558a1 GM |
42 | ;; <node1 attr1="name1" attr2="name2" ...>value |
43 | ;; <node2 attr3="name3" attr4="name4">value2</node2> | |
44 | ;; <node3 attr5="name5" attr6="name6">value3</node3> | |
47db06aa GM |
45 | ;; </node1> |
46 | ;; Of course, the name of the nodes and attributes can be anything. There can | |
47 | ;; be any number of attributes (or none), as well as any number of children | |
48 | ;; below the nodes. | |
49 | ;; | |
50 | ;; There can be only top level node, but with any number of children below. | |
51 | ||
52 | ;;; LIST FORMAT | |
53 | ||
54 | ;; The functions `xml-parse-file' and `xml-parse-tag' return a list with | |
55 | ;; the following format: | |
56 | ;; | |
57 | ;; xml-list ::= (node node ...) | |
58 | ;; node ::= (tag_name attribute-list . child_node_list) | |
59 | ;; child_node_list ::= child_node child_node ... | |
60 | ;; child_node ::= node | string | |
61 | ;; tag_name ::= string | |
62 | ;; attribute_list ::= (("attribute" . "value") ("attribute" . "value") ...) | |
63 | ;; | nil | |
64 | ;; string ::= "..." | |
65 | ;; | |
47db06aa GM |
66 | ;; Some macros are provided to ease the parsing of this list |
67 | ||
68 | ;;; Code: | |
69 | ||
70 | ;;******************************************************************* | |
71 | ;;** | |
72 | ;;** Macros to parse the list | |
73 | ;;** | |
74 | ;;******************************************************************* | |
75 | ||
971489ea | 76 | (defsubst xml-node-name (node) |
47db06aa GM |
77 | "Return the tag associated with NODE. |
78 | The tag is a lower-case symbol." | |
971489ea | 79 | (car node)) |
47db06aa | 80 | |
971489ea | 81 | (defsubst xml-node-attributes (node) |
47db06aa GM |
82 | "Return the list of attributes of NODE. |
83 | The list can be nil." | |
971489ea | 84 | (nth 1 node)) |
47db06aa | 85 | |
971489ea | 86 | (defsubst xml-node-children (node) |
47db06aa GM |
87 | "Return the list of children of NODE. |
88 | This is a list of nodes, and it can be nil." | |
971489ea | 89 | (cddr node)) |
47db06aa GM |
90 | |
91 | (defun xml-get-children (node child-name) | |
92 | "Return the children of NODE whose tag is CHILD-NAME. | |
93 | CHILD-NAME should be a lower case symbol." | |
971489ea SM |
94 | (let ((match ())) |
95 | (dolist (child (xml-node-children node)) | |
96 | (if child | |
97 | (if (equal (xml-node-name child) child-name) | |
98 | (push child match)))) | |
99 | (nreverse match))) | |
47db06aa GM |
100 | |
101 | (defun xml-get-attribute (node attribute) | |
102 | "Get from NODE the value of ATTRIBUTE. | |
103 | An empty string is returned if the attribute was not found." | |
104 | (if (xml-node-attributes node) | |
105 | (let ((value (assoc attribute (xml-node-attributes node)))) | |
106 | (if value | |
107 | (cdr value) | |
108 | "")) | |
109 | "")) | |
110 | ||
111 | ;;******************************************************************* | |
112 | ;;** | |
113 | ;;** Creating the list | |
114 | ;;** | |
115 | ;;******************************************************************* | |
116 | ||
117 | (defun xml-parse-file (file &optional parse-dtd) | |
118 | "Parse the well-formed XML FILE. | |
653558a1 | 119 | If FILE is already edited, this will keep the buffer alive. |
47db06aa GM |
120 | Returns the top node with all its children. |
121 | If PARSE-DTD is non-nil, the DTD is parsed rather than skipped." | |
653558a1 GM |
122 | (let ((keep)) |
123 | (if (get-file-buffer file) | |
124 | (progn | |
125 | (set-buffer (get-file-buffer file)) | |
126 | (setq keep (point))) | |
127 | (find-file file)) | |
524425ae | 128 | |
653558a1 GM |
129 | (let ((xml (xml-parse-region (point-min) |
130 | (point-max) | |
131 | (current-buffer) | |
132 | parse-dtd))) | |
133 | (if keep | |
134 | (goto-char keep) | |
135 | (kill-buffer (current-buffer))) | |
136 | xml))) | |
47db06aa GM |
137 | |
138 | (defun xml-parse-region (beg end &optional buffer parse-dtd) | |
139 | "Parse the region from BEG to END in BUFFER. | |
140 | If BUFFER is nil, it defaults to the current buffer. | |
141 | Returns the XML list for the region, or raises an error if the region | |
142 | is not a well-formed XML file. | |
143 | If PARSE-DTD is non-nil, the DTD is parsed rather than skipped, | |
144 | and returned as the first element of the list" | |
145 | (let (xml result dtd) | |
146 | (save-excursion | |
147 | (if buffer | |
148 | (set-buffer buffer)) | |
149 | (goto-char beg) | |
150 | (while (< (point) end) | |
151 | (if (search-forward "<" end t) | |
152 | (progn | |
153 | (forward-char -1) | |
154 | (if (null xml) | |
155 | (progn | |
971489ea | 156 | (setq result (xml-parse-tag end parse-dtd)) |
47db06aa | 157 | (cond |
971489ea | 158 | ((null result)) |
47db06aa | 159 | ((listp (car result)) |
971489ea | 160 | (setq dtd (car result)) |
47db06aa GM |
161 | (add-to-list 'xml (cdr result))) |
162 | (t | |
163 | (add-to-list 'xml result)))) | |
164 | ||
165 | ;; translation of rule [1] of XML specifications | |
1cd7adc6 | 166 | (error "XML files can have only one toplevel tag"))) |
47db06aa GM |
167 | (goto-char end))) |
168 | (if parse-dtd | |
169 | (cons dtd (reverse xml)) | |
170 | (reverse xml))))) | |
171 | ||
172 | ||
173 | (defun xml-parse-tag (end &optional parse-dtd) | |
174 | "Parse the tag that is just in front of point. | |
175 | The end tag must be found before the position END in the current buffer. | |
176 | If PARSE-DTD is non-nil, the DTD of the document, if any, is parsed and | |
177 | returned as the first element in the list. | |
178 | Returns one of: | |
179 | - a list : the matching node | |
180 | - nil : the point is not looking at a tag. | |
181 | - a cons cell: the first element is the DTD, the second is the node" | |
182 | (cond | |
183 | ;; Processing instructions (like the <?xml version="1.0"?> tag at the | |
184 | ;; beginning of a document) | |
185 | ((looking-at "<\\?") | |
186 | (search-forward "?>" end) | |
a158ff81 | 187 | (goto-char (- (re-search-forward "[^ \t\n\r]") 1)) |
47db06aa GM |
188 | (xml-parse-tag end)) |
189 | ;; Character data (CDATA) sections, in which no tag should be interpreted | |
190 | ((looking-at "<!\\[CDATA\\[") | |
191 | (let ((pos (match-end 0))) | |
192 | (unless (search-forward "]]>" end t) | |
193 | (error "CDATA section does not end anywhere in the document")) | |
194 | (buffer-substring-no-properties pos (match-beginning 0)))) | |
195 | ;; DTD for the document | |
196 | ((looking-at "<!DOCTYPE") | |
197 | (let (dtd) | |
198 | (if parse-dtd | |
971489ea | 199 | (setq dtd (xml-parse-dtd end)) |
47db06aa | 200 | (xml-skip-dtd end)) |
a158ff81 | 201 | (goto-char (- (re-search-forward "[^ \t\n\r]") 1)) |
47db06aa GM |
202 | (if dtd |
203 | (cons dtd (xml-parse-tag end)) | |
204 | (xml-parse-tag end)))) | |
205 | ;; skip comments | |
206 | ((looking-at "<!--") | |
207 | (search-forward "-->" end) | |
971489ea | 208 | nil) |
47db06aa GM |
209 | ;; end tag |
210 | ((looking-at "</") | |
211 | '()) | |
212 | ;; opening tag | |
a158ff81 | 213 | ((looking-at "<\\([^/> \t\n\r]+\\)") |
971489ea SM |
214 | (goto-char (match-end 1)) |
215 | (let* ((case-fold-search nil) ;; XML is case-sensitive. | |
216 | (node-name (match-string 1)) | |
217 | ;; Parse the attribute list. | |
218 | (children (list (xml-parse-attlist end) (intern node-name))) | |
47db06aa | 219 | pos) |
47db06aa GM |
220 | |
221 | ;; is this an empty element ? | |
a158ff81 | 222 | (if (looking-at "/[ \t\n\r]*>") |
47db06aa GM |
223 | (progn |
224 | (forward-char 2) | |
971489ea | 225 | (nreverse (cons '("") children))) |
47db06aa GM |
226 | |
227 | ;; is this a valid start tag ? | |
e54030af | 228 | (if (eq (char-after) ?>) |
47db06aa GM |
229 | (progn |
230 | (forward-char 1) | |
971489ea SM |
231 | ;; Now check that we have the right end-tag. Note that this |
232 | ;; one might contain spaces after the tag name | |
a158ff81 | 233 | (while (not (looking-at (concat "</" node-name "[ \t\n\r]*>"))) |
47db06aa GM |
234 | (cond |
235 | ((looking-at "</") | |
236 | (error (concat | |
237 | "XML: invalid syntax -- invalid end tag (expecting " | |
238 | node-name | |
653558a1 | 239 | ") at pos " (number-to-string (point))))) |
47db06aa | 240 | ((= (char-after) ?<) |
971489ea SM |
241 | (let ((tag (xml-parse-tag end))) |
242 | (when tag | |
243 | (push tag children)))) | |
47db06aa | 244 | (t |
971489ea | 245 | (setq pos (point)) |
47db06aa GM |
246 | (search-forward "<" end) |
247 | (forward-char -1) | |
248 | (let ((string (buffer-substring-no-properties pos (point))) | |
249 | (pos 0)) | |
524425ae | 250 | |
a158ff81 JB |
251 | ;; Clean up the string. As per XML |
252 | ;; specifications, the XML processor should | |
253 | ;; always pass the whole string to the | |
254 | ;; application. But \r's should be replaced: | |
255 | ;; http://www.w3.org/TR/2000/REC-xml-20001006#sec-line-ends | |
256 | (while (string-match "\r\n?" string pos) | |
257 | (setq string (replace-match "\n" t t string)) | |
258 | (setq pos (1+ (match-beginning 0)))) | |
971489ea SM |
259 | |
260 | (setq string (xml-substitute-special string)) | |
261 | (setq children | |
262 | (if (stringp (car children)) | |
263 | ;; The two strings were separated by a comment. | |
264 | (cons (concat (car children) string) | |
265 | (cdr children)) | |
266 | (cons string children))))))) | |
47db06aa | 267 | (goto-char (match-end 0)) |
47db06aa | 268 | (if (> (point) end) |
1cd7adc6 | 269 | (error "XML: End tag for %s not found before end of region" |
47db06aa | 270 | node-name)) |
971489ea | 271 | (nreverse children)) |
47db06aa GM |
272 | |
273 | ;; This was an invalid start tag | |
274 | (error "XML: Invalid attribute list") | |
275 | )))) | |
1300e272 | 276 | (t ;; This is not a tag. |
1cd7adc6 | 277 | (error "XML: Invalid character")) |
47db06aa GM |
278 | )) |
279 | ||
280 | (defun xml-parse-attlist (end) | |
281 | "Return the attribute-list that point is looking at. | |
282 | The search for attributes end at the position END in the current buffer. | |
283 | Leaves the point on the first non-blank character after the tag." | |
971489ea | 284 | (let ((attlist ()) |
a158ff81 JB |
285 | start-pos name) |
286 | (goto-char (- (re-search-forward "[^ \t\n\r]") 1)) | |
287 | (while (looking-at "\\([a-zA-Z_:][-a-zA-Z0-9._:]*\\)[ \t\n\r]*=[ \t\n\r]*") | |
971489ea | 288 | (setq name (intern (match-string 1))) |
47db06aa GM |
289 | (goto-char (match-end 0)) |
290 | ||
a158ff81 JB |
291 | ;; See also: http://www.w3.org/TR/2000/REC-xml-20001006#AVNormalize |
292 | ||
47db06aa GM |
293 | ;; Do we have a string between quotes (or double-quotes), |
294 | ;; or a simple word ? | |
a158ff81 JB |
295 | (if (looking-at "\"\\([^\"]*\\)\"") |
296 | (setq start-pos (match-beginning 0)) | |
297 | (if (looking-at "'\\([^']*\\)") | |
298 | (setq start-pos (match-beginning 0)) | |
1cd7adc6 | 299 | (error "XML: Attribute values must be given between quotes"))) |
47db06aa GM |
300 | |
301 | ;; Each attribute must be unique within a given element | |
302 | (if (assoc name attlist) | |
1cd7adc6 | 303 | (error "XML: each attribute must be unique within an element")) |
524425ae | 304 | |
a158ff81 JB |
305 | ;; Multiple whitespace characters should be replaced with a single one |
306 | ;; in the attributes | |
307 | (let ((string (match-string-no-properties 1)) | |
308 | (pos 0)) | |
309 | (while (string-match "[ \t\n\r]+" string pos) | |
310 | (setq string (replace-match " " t nil string)) | |
311 | (setq pos (1+ (match-beginning 0)))) | |
312 | (push (cons name (xml-substitute-special string)) attlist)) | |
313 | ||
314 | (goto-char start-pos) | |
315 | (if (looking-at "\"\\([^\"]*\\)\"") | |
316 | (goto-char (match-end 0)) | |
317 | (if (looking-at "'\\([^']*\\)") | |
318 | (goto-char (match-end 0)))) | |
319 | ||
320 | (goto-char (- (re-search-forward "[^ \t\n\r]") 1)) | |
47db06aa | 321 | (if (> (point) end) |
a158ff81 | 322 | (error "XML: end of attribute list not found before end of region"))) |
971489ea | 323 | (nreverse attlist))) |
47db06aa GM |
324 | |
325 | ;;******************************************************************* | |
326 | ;;** | |
327 | ;;** The DTD (document type declaration) | |
328 | ;;** The following functions know how to skip or parse the DTD of | |
329 | ;;** a document | |
330 | ;;** | |
331 | ;;******************************************************************* | |
332 | ||
333 | (defun xml-skip-dtd (end) | |
334 | "Skip the DTD that point is looking at. | |
335 | The DTD must end before the position END in the current buffer. | |
336 | The point must be just before the starting tag of the DTD. | |
337 | This follows the rule [28] in the XML specifications." | |
338 | (forward-char (length "<!DOCTYPE")) | |
a158ff81 | 339 | (if (looking-at "[ \t\n\r]*>") |
47db06aa GM |
340 | (error "XML: invalid DTD (excepting name of the document)")) |
341 | (condition-case nil | |
342 | (progn | |
a158ff81 JB |
343 | (forward-word 1) |
344 | (goto-char (- (re-search-forward "[ \t\n\r]") 1)) | |
345 | (goto-char (- (re-search-forward "[^ \t\n\r]") 1)) | |
47db06aa | 346 | (if (looking-at "\\[") |
a158ff81 | 347 | (re-search-forward "\\][ \t\n\r]*>" end) |
47db06aa GM |
348 | (search-forward ">" end))) |
349 | (error (error "XML: No end to the DTD")))) | |
350 | ||
351 | (defun xml-parse-dtd (end) | |
352 | "Parse the DTD that point is looking at. | |
353 | The DTD must end before the position END in the current buffer." | |
971489ea | 354 | (forward-char (length "<!DOCTYPE")) |
a158ff81 | 355 | (goto-char (- (re-search-forward "[^ \t\n\r]") 1)) |
971489ea SM |
356 | (if (looking-at ">") |
357 | (error "XML: invalid DTD (excepting name of the document)")) | |
524425ae | 358 | |
971489ea SM |
359 | ;; Get the name of the document |
360 | (looking-at "\\sw+") | |
361 | (let ((dtd (list (match-string-no-properties 0) 'dtd)) | |
362 | type element end-pos) | |
47db06aa GM |
363 | (goto-char (match-end 0)) |
364 | ||
a158ff81 | 365 | (goto-char (- (re-search-forward "[^ \t\n\r]") 1)) |
47db06aa | 366 | |
a158ff81 | 367 | ;; External DTDs => don't know how to handle them yet |
47db06aa | 368 | (if (looking-at "SYSTEM") |
1cd7adc6 | 369 | (error "XML: Don't know how to handle external DTDs")) |
524425ae | 370 | |
47db06aa | 371 | (if (not (= (char-after) ?\[)) |
1cd7adc6 | 372 | (error "XML: Unknown declaration in the DTD")) |
47db06aa | 373 | |
a158ff81 | 374 | ;; Parse the rest of the DTD |
47db06aa | 375 | (forward-char 1) |
a158ff81 | 376 | (while (and (not (looking-at "[ \t\n\r]*\\]")) |
47db06aa GM |
377 | (<= (point) end)) |
378 | (cond | |
379 | ||
380 | ;; Translation of rule [45] of XML specifications | |
381 | ((looking-at | |
a158ff81 | 382 | "[ \t\n\r]*<!ELEMENT[ \t\n\r]+\\([a-zA-Z0-9.%;]+\\)[ \t\n\r]+\\([^>]+\\)>") |
47db06aa | 383 | |
6eb6c4c1 | 384 | (setq element (intern (match-string-no-properties 1)) |
47db06aa | 385 | type (match-string-no-properties 2)) |
971489ea | 386 | (setq end-pos (match-end 0)) |
524425ae | 387 | |
47db06aa GM |
388 | ;; Translation of rule [46] of XML specifications |
389 | (cond | |
a158ff81 | 390 | ((string-match "^EMPTY[ \t\n\r]*$" type) ;; empty declaration |
971489ea | 391 | (setq type 'empty)) |
a158ff81 | 392 | ((string-match "^ANY[ \t\n\r]*$" type) ;; any type of contents |
971489ea | 393 | (setq type 'any)) |
a158ff81 | 394 | ((string-match "^(\\(.*\\))[ \t\n\r]*$" type) ;; children ([47]) |
971489ea | 395 | (setq type (xml-parse-elem-type (match-string-no-properties 1 type)))) |
a158ff81 | 396 | ((string-match "^%[^;]+;[ \t\n\r]*$" type) ;; substitution |
47db06aa GM |
397 | nil) |
398 | (t | |
399 | (error "XML: Invalid element type in the DTD"))) | |
400 | ||
401 | ;; rule [45]: the element declaration must be unique | |
402 | (if (assoc element dtd) | |
1cd7adc6 | 403 | (error "XML: elements declaration must be unique in a DTD (<%s>)" |
47db06aa | 404 | (symbol-name element))) |
524425ae | 405 | |
47db06aa | 406 | ;; Store the element in the DTD |
971489ea SM |
407 | (push (list element type) dtd) |
408 | (goto-char end-pos)) | |
47db06aa GM |
409 | |
410 | ||
411 | (t | |
412 | (error "XML: Invalid DTD item")) | |
413 | ) | |
414 | ) | |
415 | ||
416 | ;; Skip the end of the DTD | |
417 | (search-forward ">" end) | |
971489ea | 418 | (nreverse dtd))) |
47db06aa GM |
419 | |
420 | ||
421 | (defun xml-parse-elem-type (string) | |
422 | "Convert a STRING for an element type into an elisp structure." | |
423 | ||
424 | (let (elem modifier) | |
425 | (if (string-match "(\\([^)]+\\))\\([+*?]?\\)" string) | |
426 | (progn | |
427 | (setq elem (match-string 1 string) | |
428 | modifier (match-string 2 string)) | |
429 | (if (string-match "|" elem) | |
971489ea | 430 | (setq elem (cons 'choice |
47db06aa GM |
431 | (mapcar 'xml-parse-elem-type |
432 | (split-string elem "|")))) | |
433 | (if (string-match "," elem) | |
971489ea | 434 | (setq elem (cons 'seq |
47db06aa GM |
435 | (mapcar 'xml-parse-elem-type |
436 | (split-string elem ",")))) | |
437 | ))) | |
a158ff81 JB |
438 | (if (string-match "[ \t\n\r]*\\([^+*?]+\\)\\([+*?]?\\)" string) |
439 | (setq elem (match-string 1 string) | |
47db06aa GM |
440 | modifier (match-string 2 string)))) |
441 | ||
971489ea SM |
442 | (if (and (stringp elem) (string= elem "#PCDATA")) |
443 | (setq elem 'pcdata)) | |
524425ae | 444 | |
971489ea SM |
445 | (cond |
446 | ((string= modifier "+") | |
447 | (list '+ elem)) | |
448 | ((string= modifier "*") | |
449 | (list '* elem)) | |
450 | ((string= modifier "?") | |
0fa6f70c | 451 | (list '\? elem)) |
971489ea SM |
452 | (t |
453 | elem)))) | |
47db06aa | 454 | |
a158ff81 JB |
455 | ;;******************************************************************* |
456 | ;;** | |
457 | ;;** Converting code points to strings | |
458 | ;;** | |
459 | ;;******************************************************************* | |
460 | ||
461 | (defun xml-ucs-to-string (codepoint) | |
462 | "Return a string representation of CODEPOINT. If it can't be | |
463 | converted, return '?'." | |
464 | (cond ((boundp 'decode-char) | |
465 | (char-to-string (decode-char 'ucs codepoint))) | |
466 | ((and (< codepoint 128) | |
467 | (> codepoint 31)) | |
468 | (char-to-string codepoint)) | |
469 | (t "?"))) ; FIXME: There's gotta be a better way to | |
470 | ; designate an unknown character. | |
47db06aa GM |
471 | |
472 | ;;******************************************************************* | |
473 | ;;** | |
474 | ;;** Substituting special XML sequences | |
475 | ;;** | |
476 | ;;******************************************************************* | |
477 | ||
478 | (defun xml-substitute-special (string) | |
479 | "Return STRING, after subsituting special XML sequences." | |
47db06aa | 480 | (while (string-match "<" string) |
971489ea | 481 | (setq string (replace-match "<" t nil string))) |
47db06aa | 482 | (while (string-match ">" string) |
971489ea | 483 | (setq string (replace-match ">" t nil string))) |
47db06aa | 484 | (while (string-match "'" string) |
971489ea | 485 | (setq string (replace-match "'" t nil string))) |
47db06aa | 486 | (while (string-match """ string) |
971489ea | 487 | (setq string (replace-match "\"" t nil string))) |
a158ff81 JB |
488 | (while (string-match "&#\\([0-9]+\\);" string) |
489 | (setq string (replace-match (xml-ucs-to-string | |
490 | (string-to-number | |
491 | (match-string-no-properties 1 string))) | |
492 | t nil string))) | |
493 | (while (string-match "&#x\\([0-9a-fA-F]+\\);" string) | |
494 | (setq string (replace-match (xml-ucs-to-string | |
495 | (string-to-number | |
496 | (match-string-no-properties 1 string) | |
497 | 16)) | |
498 | t nil string))) | |
499 | ||
ceabd272 | 500 | ;; This goes last so it doesn't confuse the matches above. |
524425ae TTN |
501 | (while (string-match "&" string) |
502 | (setq string (replace-match "&" t nil string))) | |
47db06aa GM |
503 | string) |
504 | ||
505 | ;;******************************************************************* | |
506 | ;;** | |
507 | ;;** Printing a tree. | |
508 | ;;** This function is intended mainly for debugging purposes. | |
509 | ;;** | |
510 | ;;******************************************************************* | |
511 | ||
512 | (defun xml-debug-print (xml) | |
971489ea SM |
513 | (dolist (node xml) |
514 | (xml-debug-print-internal node ""))) | |
47db06aa | 515 | |
971489ea | 516 | (defun xml-debug-print-internal (xml indent-string) |
47db06aa GM |
517 | "Outputs the XML tree in the current buffer. |
518 | The first line indented with INDENT-STRING." | |
519 | (let ((tree xml) | |
520 | attlist) | |
47db06aa | 521 | (insert indent-string "<" (symbol-name (xml-node-name tree))) |
524425ae | 522 | |
47db06aa | 523 | ;; output the attribute list |
971489ea | 524 | (setq attlist (xml-node-attributes tree)) |
47db06aa GM |
525 | (while attlist |
526 | (insert " ") | |
527 | (insert (symbol-name (caar attlist)) "=\"" (cdar attlist) "\"") | |
971489ea | 528 | (setq attlist (cdr attlist))) |
524425ae | 529 | |
47db06aa | 530 | (insert ">") |
524425ae | 531 | |
971489ea | 532 | (setq tree (xml-node-children tree)) |
47db06aa GM |
533 | |
534 | ;; output the children | |
971489ea | 535 | (dolist (node tree) |
47db06aa | 536 | (cond |
971489ea | 537 | ((listp node) |
47db06aa | 538 | (insert "\n") |
971489ea SM |
539 | (xml-debug-print-internal node (concat indent-string " "))) |
540 | ((stringp node) (insert node)) | |
47db06aa | 541 | (t |
971489ea | 542 | (error "Invalid XML tree")))) |
47db06aa GM |
543 | |
544 | (insert "\n" indent-string | |
971489ea | 545 | "</" (symbol-name (xml-node-name xml)) ">"))) |
47db06aa GM |
546 | |
547 | (provide 'xml) | |
548 | ||
549 | ;;; xml.el ends here |