| 1 | ;;; xml.el --- XML parser |
| 2 | |
| 3 | ;; Copyright (C) 2000, 2001, 2002, 2003, 2004, |
| 4 | ;; 2005, 2006, 2007, 2008 Free Software Foundation, Inc. |
| 5 | |
| 6 | ;; Author: Emmanuel Briot <briot@gnat.com> |
| 7 | ;; Maintainer: Mark A. Hershberger <mah@everybody.org> |
| 8 | ;; Keywords: xml, data |
| 9 | |
| 10 | ;; This file is part of GNU Emacs. |
| 11 | |
| 12 | ;; GNU Emacs is free software: you can redistribute it and/or modify |
| 13 | ;; it under the terms of the GNU General Public License as published by |
| 14 | ;; the Free Software Foundation, either version 3 of the License, or |
| 15 | ;; (at your option) any later version. |
| 16 | |
| 17 | ;; GNU Emacs is distributed in the hope that it will be useful, |
| 18 | ;; but WITHOUT ANY WARRANTY; without even the implied warranty of |
| 19 | ;; MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the |
| 20 | ;; GNU General Public License for more details. |
| 21 | |
| 22 | ;; You should have received a copy of the GNU General Public License |
| 23 | ;; along with GNU Emacs. If not, see <http://www.gnu.org/licenses/>. |
| 24 | |
| 25 | ;;; Commentary: |
| 26 | |
| 27 | ;; This file contains a somewhat incomplete non-validating XML parser. It |
| 28 | ;; parses a file, and returns a list that can be used internally by |
| 29 | ;; any other Lisp libraries. |
| 30 | |
| 31 | ;;; FILE FORMAT |
| 32 | |
| 33 | ;; The document type declaration may either be ignored or (optionally) |
| 34 | ;; parsed, but currently the parsing will only accept element |
| 35 | ;; declarations. The XML file is assumed to be well-formed. In case |
| 36 | ;; of error, the parsing stops and the XML file is shown where the |
| 37 | ;; parsing stopped. |
| 38 | ;; |
| 39 | ;; It also knows how to ignore comments and processing instructions. |
| 40 | ;; |
| 41 | ;; The XML file should have the following format: |
| 42 | ;; <node1 attr1="name1" attr2="name2" ...>value |
| 43 | ;; <node2 attr3="name3" attr4="name4">value2</node2> |
| 44 | ;; <node3 attr5="name5" attr6="name6">value3</node3> |
| 45 | ;; </node1> |
| 46 | ;; Of course, the name of the nodes and attributes can be anything. There can |
| 47 | ;; be any number of attributes (or none), as well as any number of children |
| 48 | ;; below the nodes. |
| 49 | ;; |
| 50 | ;; There can be only top level node, but with any number of children below. |
| 51 | |
| 52 | ;;; LIST FORMAT |
| 53 | |
| 54 | ;; The functions `xml-parse-file', `xml-parse-region' and |
| 55 | ;; `xml-parse-tag' return a list with the following format: |
| 56 | ;; |
| 57 | ;; xml-list ::= (node node ...) |
| 58 | ;; node ::= (qname attribute-list . child_node_list) |
| 59 | ;; child_node_list ::= child_node child_node ... |
| 60 | ;; child_node ::= node | string |
| 61 | ;; qname ::= (:namespace-uri . "name") | "name" |
| 62 | ;; attribute_list ::= ((qname . "value") (qname . "value") ...) |
| 63 | ;; | nil |
| 64 | ;; string ::= "..." |
| 65 | ;; |
| 66 | ;; Some macros are provided to ease the parsing of this list. |
| 67 | ;; Whitespace is preserved. Fixme: There should be a tree-walker that |
| 68 | ;; can remove it. |
| 69 | |
| 70 | ;; TODO: |
| 71 | ;; * xml:base, xml:space support |
| 72 | ;; * more complete DOCTYPE parsing |
| 73 | ;; * pi support |
| 74 | |
| 75 | ;;; Code: |
| 76 | |
| 77 | ;; Note that buffer-substring and match-string were formerly used in |
| 78 | ;; several places, because the -no-properties variants remove |
| 79 | ;; composition info. However, after some discussion on emacs-devel, |
| 80 | ;; the consensus was that the speed of the -no-properties variants was |
| 81 | ;; a worthwhile tradeoff especially since we're usually parsing files |
| 82 | ;; instead of hand-crafted XML. |
| 83 | |
| 84 | ;;******************************************************************* |
| 85 | ;;** |
| 86 | ;;** Macros to parse the list |
| 87 | ;;** |
| 88 | ;;******************************************************************* |
| 89 | |
| 90 | (defconst xml-undefined-entity "?" |
| 91 | "What to substitute for undefined entities") |
| 92 | |
| 93 | (defvar xml-entity-alist |
| 94 | '(("lt" . "<") |
| 95 | ("gt" . ">") |
| 96 | ("apos" . "'") |
| 97 | ("quot" . "\"") |
| 98 | ("amp" . "&")) |
| 99 | "The defined entities. Entities are added to this when the DTD is parsed.") |
| 100 | |
| 101 | (defvar xml-sub-parser nil |
| 102 | "Dynamically set this to a non-nil value if you want to parse an XML fragment.") |
| 103 | |
| 104 | (defvar xml-validating-parser nil |
| 105 | "Set to non-nil to get validity checking.") |
| 106 | |
| 107 | (defsubst xml-node-name (node) |
| 108 | "Return the tag associated with NODE. |
| 109 | Without namespace-aware parsing, the tag is a symbol. |
| 110 | |
| 111 | With namespace-aware parsing, the tag is a cons of a string |
| 112 | representing the uri of the namespace with the local name of the |
| 113 | tag. For example, |
| 114 | |
| 115 | <foo> |
| 116 | |
| 117 | would be represented by |
| 118 | |
| 119 | '(\"\" . \"foo\")." |
| 120 | |
| 121 | (car node)) |
| 122 | |
| 123 | (defsubst xml-node-attributes (node) |
| 124 | "Return the list of attributes of NODE. |
| 125 | The list can be nil." |
| 126 | (nth 1 node)) |
| 127 | |
| 128 | (defsubst xml-node-children (node) |
| 129 | "Return the list of children of NODE. |
| 130 | This is a list of nodes, and it can be nil." |
| 131 | (cddr node)) |
| 132 | |
| 133 | (defun xml-get-children (node child-name) |
| 134 | "Return the children of NODE whose tag is CHILD-NAME. |
| 135 | CHILD-NAME should match the value returned by `xml-node-name'." |
| 136 | (let ((match ())) |
| 137 | (dolist (child (xml-node-children node)) |
| 138 | (if (and (listp child) |
| 139 | (equal (xml-node-name child) child-name)) |
| 140 | (push child match))) |
| 141 | (nreverse match))) |
| 142 | |
| 143 | (defun xml-get-attribute-or-nil (node attribute) |
| 144 | "Get from NODE the value of ATTRIBUTE. |
| 145 | Return nil if the attribute was not found. |
| 146 | |
| 147 | See also `xml-get-attribute'." |
| 148 | (cdr (assoc attribute (xml-node-attributes node)))) |
| 149 | |
| 150 | (defsubst xml-get-attribute (node attribute) |
| 151 | "Get from NODE the value of ATTRIBUTE. |
| 152 | An empty string is returned if the attribute was not found. |
| 153 | |
| 154 | See also `xml-get-attribute-or-nil'." |
| 155 | (or (xml-get-attribute-or-nil node attribute) "")) |
| 156 | |
| 157 | ;;******************************************************************* |
| 158 | ;;** |
| 159 | ;;** Creating the list |
| 160 | ;;** |
| 161 | ;;******************************************************************* |
| 162 | |
| 163 | ;;;###autoload |
| 164 | (defun xml-parse-file (file &optional parse-dtd parse-ns) |
| 165 | "Parse the well-formed XML file FILE. |
| 166 | If FILE is already visited, use its buffer and don't kill it. |
| 167 | Returns the top node with all its children. |
| 168 | If PARSE-DTD is non-nil, the DTD is parsed rather than skipped. |
| 169 | If PARSE-NS is non-nil, then QNAMES are expanded." |
| 170 | (if (get-file-buffer file) |
| 171 | (with-current-buffer (get-file-buffer file) |
| 172 | (save-excursion |
| 173 | (xml-parse-region (point-min) |
| 174 | (point-max) |
| 175 | (current-buffer) |
| 176 | parse-dtd parse-ns))) |
| 177 | (with-temp-buffer |
| 178 | (insert-file-contents file) |
| 179 | (xml-parse-region (point-min) |
| 180 | (point-max) |
| 181 | (current-buffer) |
| 182 | parse-dtd parse-ns)))) |
| 183 | |
| 184 | |
| 185 | (defvar xml-name-re) |
| 186 | (defvar xml-entity-value-re) |
| 187 | (defvar xml-att-def-re) |
| 188 | (let* ((start-chars (concat "[:alpha:]:_")) |
| 189 | (name-chars (concat "-[:digit:]." start-chars)) |
| 190 | ;;[3] S ::= (#x20 | #x9 | #xD | #xA)+ |
| 191 | (whitespace "[ \t\n\r]")) |
| 192 | ;;[4] NameStartChar ::= ":" | [A-Z] | "_" | [a-z] | [#xC0-#xD6] |
| 193 | ;; | [#xD8-#xF6] | [#xF8-#x2FF] | [#x370-#x37D] | [#x37F-#x1FFF] |
| 194 | ;; | [#x200C-#x200D] | [#x2070-#x218F] | [#x2C00-#x2FEF] | [#x3001-#xD7FF] |
| 195 | ;; | [#xF900-#xFDCF] | [#xFDF0-#xFFFD] | [#x10000-#xEFFFF] |
| 196 | (defvar xml-name-start-char-re (concat "[" start-chars "]")) |
| 197 | ;;[4a] NameChar ::= NameStartChar | "-" | "." | [0-9] | #xB7 | [#x0300-#x036F] | [#x203F-#x2040] |
| 198 | (defvar xml-name-char-re (concat "[" name-chars "]")) |
| 199 | ;;[5] Name ::= NameStartChar (NameChar)* |
| 200 | (defvar xml-name-re (concat xml-name-start-char-re xml-name-char-re "*")) |
| 201 | ;;[6] Names ::= Name (#x20 Name)* |
| 202 | (defvar xml-names-re (concat xml-name-re "\\(?: " xml-name-re "\\)*")) |
| 203 | ;;[7] Nmtoken ::= (NameChar)+ |
| 204 | (defvar xml-nmtoken-re (concat xml-name-char-re "+")) |
| 205 | ;;[8] Nmtokens ::= Nmtoken (#x20 Nmtoken)* |
| 206 | (defvar xml-nmtokens-re (concat xml-nmtoken-re "\\(?: " xml-name-re "\\)*")) |
| 207 | ;;[66] CharRef ::= '&#' [0-9]+ ';' | '&#x' [0-9a-fA-F]+ ';' |
| 208 | (defvar xml-char-ref-re "\\(?:&#[0-9]+;\\|&#x[0-9a-fA-F]+;\\)") |
| 209 | ;;[68] EntityRef ::= '&' Name ';' |
| 210 | (defvar xml-entity-ref (concat "&" xml-name-re ";")) |
| 211 | ;;[69] PEReference ::= '%' Name ';' |
| 212 | (defvar xml-pe-reference-re (concat "%" xml-name-re ";")) |
| 213 | ;;[67] Reference ::= EntityRef | CharRef |
| 214 | (defvar xml-reference-re (concat "\\(?:" xml-entity-ref "\\|" xml-char-ref-re "\\)")) |
| 215 | ;;[10] AttValue ::= '"' ([^<&"] | Reference)* '"' | "'" ([^<&'] | Reference)* "'" |
| 216 | (defvar xml-att-value-re (concat "\\(?:\"\\(?:[^&\"]\\|" xml-reference-re "\\)*\"\\|" |
| 217 | "'\\(?:[^&']\\|" xml-reference-re "\\)*'\\)")) |
| 218 | ;;[56] TokenizedType ::= 'ID' [VC: ID] [VC: One ID per Element Type] [VC: ID Attribute Default] |
| 219 | ;; | 'IDREF' [VC: IDREF] |
| 220 | ;; | 'IDREFS' [VC: IDREF] |
| 221 | ;; | 'ENTITY' [VC: Entity Name] |
| 222 | ;; | 'ENTITIES' [VC: Entity Name] |
| 223 | ;; | 'NMTOKEN' [VC: Name Token] |
| 224 | ;; | 'NMTOKENS' [VC: Name Token] |
| 225 | (defvar xml-tokenized-type-re "\\(?:ID\\|IDREF\\|IDREFS\\|ENTITY\\|ENTITIES\\|NMTOKEN\\|NMTOKENS\\)") |
| 226 | ;;[58] NotationType ::= 'NOTATION' S '(' S? Name (S? '|' S? Name)* S? ')' |
| 227 | (defvar xml-notation-type-re (concat "\\(?:NOTATION" whitespace "(" whitespace "*" xml-name-re |
| 228 | "\\(?:" whitespace "*|" whitespace "*" xml-name-re "\\)*" whitespace "*)\\)")) |
| 229 | ;;[59] Enumeration ::= '(' S? Nmtoken (S? '|' S? Nmtoken)* S? ')' [VC: Enumeration] [VC: No Duplicate Tokens] |
| 230 | (defvar xml-enumeration-re (concat "\\(?:(" whitespace "*" xml-nmtoken-re |
| 231 | "\\(?:" whitespace "*|" whitespace "*" xml-nmtoken-re "\\)*" |
| 232 | whitespace ")\\)")) |
| 233 | ;;[57] EnumeratedType ::= NotationType | Enumeration |
| 234 | (defvar xml-enumerated-type-re (concat "\\(?:" xml-notation-type-re "\\|" xml-enumeration-re "\\)")) |
| 235 | ;;[54] AttType ::= StringType | TokenizedType | EnumeratedType |
| 236 | ;;[55] StringType ::= 'CDATA' |
| 237 | (defvar xml-att-type-re (concat "\\(?:CDATA\\|" xml-tokenized-type-re "\\|" xml-notation-type-re"\\|" xml-enumerated-type-re "\\)")) |
| 238 | ;;[60] DefaultDecl ::= '#REQUIRED' | '#IMPLIED' | (('#FIXED' S)? AttValue) |
| 239 | (defvar xml-default-decl-re (concat "\\(?:#REQUIRED\\|#IMPLIED\\|\\(?:#FIXED" whitespace "\\)*" xml-att-value-re "\\)")) |
| 240 | ;;[53] AttDef ::= S Name S AttType S DefaultDecl |
| 241 | (defvar xml-att-def-re (concat "\\(?:" whitespace "*" xml-name-re |
| 242 | whitespace "*" xml-att-type-re |
| 243 | whitespace "*" xml-default-decl-re "\\)")) |
| 244 | ;;[9] EntityValue ::= '"' ([^%&"] | PEReference | Reference)* '"' |
| 245 | ;; | "'" ([^%&'] | PEReference | Reference)* "'" |
| 246 | (defvar xml-entity-value-re (concat "\\(?:\"\\(?:[^%&\"]\\|" xml-pe-reference-re |
| 247 | "\\|" xml-reference-re "\\)*\"\\|'\\(?:[^%&']\\|" |
| 248 | xml-pe-reference-re "\\|" xml-reference-re "\\)*'\\)"))) |
| 249 | ;;[75] ExternalID ::= 'SYSTEM' S SystemLiteral |
| 250 | ;; | 'PUBLIC' S PubidLiteral S SystemLiteral |
| 251 | ;;[76] NDataDecl ::= S 'NDATA' S |
| 252 | ;;[73] EntityDef ::= EntityValue| (ExternalID NDataDecl?) |
| 253 | ;;[71] GEDecl ::= '<!ENTITY' S Name S EntityDef S? '>' |
| 254 | ;;[74] PEDef ::= EntityValue | ExternalID |
| 255 | ;;[72] PEDecl ::= '<!ENTITY' S '%' S Name S PEDef S? '>' |
| 256 | ;;[70] EntityDecl ::= GEDecl | PEDecl |
| 257 | |
| 258 | ;; Note that this is setup so that we can do whitespace-skipping with |
| 259 | ;; `(skip-syntax-forward " ")', inter alia. Previously this was slow |
| 260 | ;; compared with `re-search-forward', but that has been fixed. Also |
| 261 | ;; note that the standard syntax table contains other characters with |
| 262 | ;; whitespace syntax, like NBSP, but they are invalid in contexts in |
| 263 | ;; which we might skip whitespace -- specifically, they're not |
| 264 | ;; NameChars [XML 4]. |
| 265 | |
| 266 | (defvar xml-syntax-table |
| 267 | (let ((table (make-syntax-table))) |
| 268 | ;; Get space syntax correct per XML [3]. |
| 269 | (dotimes (c 31) |
| 270 | (modify-syntax-entry c "." table)) ; all are space in standard table |
| 271 | (dolist (c '(?\t ?\n ?\r)) ; these should be space |
| 272 | (modify-syntax-entry c " " table)) |
| 273 | ;; For skipping attributes. |
| 274 | (modify-syntax-entry ?\" "\"" table) |
| 275 | (modify-syntax-entry ?' "\"" table) |
| 276 | ;; Non-alnum name chars should be symbol constituents (`-' and `_' |
| 277 | ;; are OK by default). |
| 278 | (modify-syntax-entry ?. "_" table) |
| 279 | (modify-syntax-entry ?: "_" table) |
| 280 | ;; XML [89] |
| 281 | (unless (featurep 'xemacs) |
| 282 | (dolist (c '(#x00B7 #x02D0 #x02D1 #x0387 #x0640 #x0E46 #x0EC6 #x3005 |
| 283 | #x3031 #x3032 #x3033 #x3034 #x3035 #x309D #x309E #x30FC |
| 284 | #x30FD #x30FE)) |
| 285 | (modify-syntax-entry (decode-char 'ucs c) "w" table))) |
| 286 | ;; Fixme: rest of [4] |
| 287 | table) |
| 288 | "Syntax table used by `xml-parse-region'.") |
| 289 | |
| 290 | ;; XML [5] |
| 291 | ;; Note that [:alpha:] matches all multibyte chars with word syntax. |
| 292 | (eval-and-compile |
| 293 | (defconst xml-name-regexp "[[:alpha:]_:][[:alnum:]._:-]*")) |
| 294 | |
| 295 | ;; Fixme: This needs re-writing to deal with the XML grammar properly, i.e. |
| 296 | ;; document ::= prolog element Misc* |
| 297 | ;; prolog ::= XMLDecl? Misc* (doctypedecl Misc*)? |
| 298 | |
| 299 | ;;;###autoload |
| 300 | (defun xml-parse-region (beg end &optional buffer parse-dtd parse-ns) |
| 301 | "Parse the region from BEG to END in BUFFER. |
| 302 | If BUFFER is nil, it defaults to the current buffer. |
| 303 | Returns the XML list for the region, or raises an error if the region |
| 304 | is not well-formed XML. |
| 305 | If PARSE-DTD is non-nil, the DTD is parsed rather than skipped, |
| 306 | and returned as the first element of the list. |
| 307 | If PARSE-NS is non-nil, then QNAMES are expanded." |
| 308 | ;; Use fixed syntax table to ensure regexp char classes and syntax |
| 309 | ;; specs DTRT. |
| 310 | (with-syntax-table (standard-syntax-table) |
| 311 | (let ((case-fold-search nil) ; XML is case-sensitive. |
| 312 | xml result dtd) |
| 313 | (save-excursion |
| 314 | (if buffer |
| 315 | (set-buffer buffer)) |
| 316 | (save-restriction |
| 317 | (narrow-to-region beg end) |
| 318 | (goto-char (point-min)) |
| 319 | (while (not (eobp)) |
| 320 | (if (search-forward "<" nil t) |
| 321 | (progn |
| 322 | (forward-char -1) |
| 323 | (setq result (xml-parse-tag parse-dtd parse-ns)) |
| 324 | (if (and xml result (not xml-sub-parser)) |
| 325 | ;; translation of rule [1] of XML specifications |
| 326 | (error "XML: (Not Well-Formed) Only one root tag allowed") |
| 327 | (cond |
| 328 | ((null result)) |
| 329 | ((and (listp (car result)) |
| 330 | parse-dtd) |
| 331 | (setq dtd (car result)) |
| 332 | (if (cdr result) ; possible leading comment |
| 333 | (add-to-list 'xml (cdr result)))) |
| 334 | (t |
| 335 | (add-to-list 'xml result))))) |
| 336 | (goto-char (point-max)))) |
| 337 | (if parse-dtd |
| 338 | (cons dtd (nreverse xml)) |
| 339 | (nreverse xml))))))) |
| 340 | |
| 341 | (defun xml-maybe-do-ns (name default xml-ns) |
| 342 | "Perform any namespace expansion. |
| 343 | NAME is the name to perform the expansion on. |
| 344 | DEFAULT is the default namespace. XML-NS is a cons of namespace |
| 345 | names to uris. When namespace-aware parsing is off, then XML-NS |
| 346 | is nil. |
| 347 | |
| 348 | During namespace-aware parsing, any name without a namespace is |
| 349 | put into the namespace identified by DEFAULT. nil is used to |
| 350 | specify that the name shouldn't be given a namespace." |
| 351 | (if (consp xml-ns) |
| 352 | (let* ((nsp (string-match ":" name)) |
| 353 | (lname (if nsp (substring name (match-end 0)) name)) |
| 354 | (prefix (if nsp (substring name 0 (match-beginning 0)) default)) |
| 355 | (special (and (string-equal lname "xmlns") (not prefix))) |
| 356 | ;; Setting default to nil will insure that there is not |
| 357 | ;; matching cons in xml-ns. In which case we |
| 358 | (ns (or (cdr (assoc (if special "xmlns" prefix) |
| 359 | xml-ns)) |
| 360 | ""))) |
| 361 | (cons ns (if special "" lname))) |
| 362 | (intern name))) |
| 363 | |
| 364 | (defun xml-parse-fragment (&optional parse-dtd parse-ns) |
| 365 | "Parse xml-like fragments." |
| 366 | (let ((xml-sub-parser t) |
| 367 | children) |
| 368 | (while (not (eobp)) |
| 369 | (let ((bit (xml-parse-tag |
| 370 | parse-dtd parse-ns))) |
| 371 | (if children |
| 372 | (setq children (append (list bit) children)) |
| 373 | (if (stringp bit) |
| 374 | (setq children (list bit)) |
| 375 | (setq children bit))))) |
| 376 | (reverse children))) |
| 377 | |
| 378 | (defun xml-parse-tag (&optional parse-dtd parse-ns) |
| 379 | "Parse the tag at point. |
| 380 | If PARSE-DTD is non-nil, the DTD of the document, if any, is parsed and |
| 381 | returned as the first element in the list. |
| 382 | If PARSE-NS is non-nil, then QNAMES are expanded. |
| 383 | Returns one of: |
| 384 | - a list : the matching node |
| 385 | - nil : the point is not looking at a tag. |
| 386 | - a pair : the first element is the DTD, the second is the node." |
| 387 | (let ((xml-validating-parser (or parse-dtd xml-validating-parser)) |
| 388 | (xml-ns (if (consp parse-ns) |
| 389 | parse-ns |
| 390 | (if parse-ns |
| 391 | (list |
| 392 | ;; Default for empty prefix is no namespace |
| 393 | (cons "" "") |
| 394 | ;; "xml" namespace |
| 395 | (cons "xml" "http://www.w3.org/XML/1998/namespace") |
| 396 | ;; We need to seed the xmlns namespace |
| 397 | (cons "xmlns" "http://www.w3.org/2000/xmlns/")))))) |
| 398 | (cond |
| 399 | ;; Processing instructions (like the <?xml version="1.0"?> tag at the |
| 400 | ;; beginning of a document). |
| 401 | ((looking-at "<\\?") |
| 402 | (search-forward "?>") |
| 403 | (skip-syntax-forward " ") |
| 404 | (xml-parse-tag parse-dtd xml-ns)) |
| 405 | ;; Character data (CDATA) sections, in which no tag should be interpreted |
| 406 | ((looking-at "<!\\[CDATA\\[") |
| 407 | (let ((pos (match-end 0))) |
| 408 | (unless (search-forward "]]>" nil t) |
| 409 | (error "XML: (Not Well Formed) CDATA section does not end anywhere in the document")) |
| 410 | (concat |
| 411 | (buffer-substring-no-properties pos (match-beginning 0)) |
| 412 | (xml-parse-string)))) |
| 413 | ;; DTD for the document |
| 414 | ((looking-at "<!DOCTYPE") |
| 415 | (let ((dtd (xml-parse-dtd parse-ns))) |
| 416 | (skip-syntax-forward " ") |
| 417 | (if xml-validating-parser |
| 418 | (cons dtd (xml-parse-tag nil xml-ns)) |
| 419 | (xml-parse-tag nil xml-ns)))) |
| 420 | ;; skip comments |
| 421 | ((looking-at "<!--") |
| 422 | (search-forward "-->") |
| 423 | nil) |
| 424 | ;; end tag |
| 425 | ((looking-at "</") |
| 426 | '()) |
| 427 | ;; opening tag |
| 428 | ((looking-at "<\\([^/>[:space:]]+\\)") |
| 429 | (goto-char (match-end 1)) |
| 430 | |
| 431 | ;; Parse this node |
| 432 | (let* ((node-name (match-string-no-properties 1)) |
| 433 | ;; Parse the attribute list. |
| 434 | (attrs (xml-parse-attlist xml-ns)) |
| 435 | children pos) |
| 436 | |
| 437 | ;; add the xmlns:* attrs to our cache |
| 438 | (when (consp xml-ns) |
| 439 | (dolist (attr attrs) |
| 440 | (when (and (consp (car attr)) |
| 441 | (equal "http://www.w3.org/2000/xmlns/" |
| 442 | (caar attr))) |
| 443 | (push (cons (cdar attr) (cdr attr)) |
| 444 | xml-ns)))) |
| 445 | |
| 446 | (setq children (list attrs (xml-maybe-do-ns node-name "" xml-ns))) |
| 447 | |
| 448 | ;; is this an empty element ? |
| 449 | (if (looking-at "/>") |
| 450 | (progn |
| 451 | (forward-char 2) |
| 452 | (nreverse children)) |
| 453 | |
| 454 | ;; is this a valid start tag ? |
| 455 | (if (eq (char-after) ?>) |
| 456 | (progn |
| 457 | (forward-char 1) |
| 458 | ;; Now check that we have the right end-tag. Note that this |
| 459 | ;; one might contain spaces after the tag name |
| 460 | (let ((end (concat "</" node-name "\\s-*>"))) |
| 461 | (while (not (looking-at end)) |
| 462 | (cond |
| 463 | ((looking-at "</") |
| 464 | (error "XML: (Not Well-Formed) Invalid end tag (expecting %s) at pos %d" |
| 465 | node-name (point))) |
| 466 | ((= (char-after) ?<) |
| 467 | (let ((tag (xml-parse-tag nil xml-ns))) |
| 468 | (when tag |
| 469 | (push tag children)))) |
| 470 | (t |
| 471 | (let ((expansion (xml-parse-string))) |
| 472 | (setq children |
| 473 | (if (stringp expansion) |
| 474 | (if (stringp (car children)) |
| 475 | ;; The two strings were separated by a comment. |
| 476 | (setq children (append (list (concat (car children) expansion)) |
| 477 | (cdr children))) |
| 478 | (setq children (append (list expansion) children))) |
| 479 | (setq children (append expansion children)))))))) |
| 480 | |
| 481 | (goto-char (match-end 0)) |
| 482 | (nreverse children))) |
| 483 | ;; This was an invalid start tag (Expected ">", but didn't see it.) |
| 484 | (error "XML: (Well-Formed) Couldn't parse tag: %s" |
| 485 | (buffer-substring-no-properties (- (point) 10) (+ (point) 1))))))) |
| 486 | (t ;; (Not one of PI, CDATA, Comment, End tag, or Start tag) |
| 487 | (unless xml-sub-parser ; Usually, we error out. |
| 488 | (error "XML: (Well-Formed) Invalid character")) |
| 489 | |
| 490 | ;; However, if we're parsing incrementally, then we need to deal |
| 491 | ;; with stray CDATA. |
| 492 | (xml-parse-string))))) |
| 493 | |
| 494 | (defun xml-parse-string () |
| 495 | "Parse the next whatever. Could be a string, or an element." |
| 496 | (let* ((pos (point)) |
| 497 | (string (progn (if (search-forward "<" nil t) |
| 498 | (forward-char -1) |
| 499 | (goto-char (point-max))) |
| 500 | (buffer-substring-no-properties pos (point))))) |
| 501 | ;; Clean up the string. As per XML specifications, the XML |
| 502 | ;; processor should always pass the whole string to the |
| 503 | ;; application. But \r's should be replaced: |
| 504 | ;; http://www.w3.org/TR/2000/REC-xml-20001006#sec-line-ends |
| 505 | (setq pos 0) |
| 506 | (while (string-match "\r\n?" string pos) |
| 507 | (setq string (replace-match "\n" t t string)) |
| 508 | (setq pos (1+ (match-beginning 0)))) |
| 509 | |
| 510 | (xml-substitute-special string))) |
| 511 | |
| 512 | (defun xml-parse-attlist (&optional xml-ns) |
| 513 | "Return the attribute-list after point. |
| 514 | Leave point at the first non-blank character after the tag." |
| 515 | (let ((attlist ()) |
| 516 | end-pos name) |
| 517 | (skip-syntax-forward " ") |
| 518 | (while (looking-at (eval-when-compile |
| 519 | (concat "\\(" xml-name-regexp "\\)\\s-*=\\s-*"))) |
| 520 | (setq end-pos (match-end 0)) |
| 521 | (setq name (xml-maybe-do-ns (match-string-no-properties 1) nil xml-ns)) |
| 522 | (goto-char end-pos) |
| 523 | |
| 524 | ;; See also: http://www.w3.org/TR/2000/REC-xml-20001006#AVNormalize |
| 525 | |
| 526 | ;; Do we have a string between quotes (or double-quotes), |
| 527 | ;; or a simple word ? |
| 528 | (if (looking-at "\"\\([^\"]*\\)\"") |
| 529 | (setq end-pos (match-end 0)) |
| 530 | (if (looking-at "'\\([^']*\\)'") |
| 531 | (setq end-pos (match-end 0)) |
| 532 | (error "XML: (Not Well-Formed) Attribute values must be given between quotes"))) |
| 533 | |
| 534 | ;; Each attribute must be unique within a given element |
| 535 | (if (assoc name attlist) |
| 536 | (error "XML: (Not Well-Formed) Each attribute must be unique within an element")) |
| 537 | |
| 538 | ;; Multiple whitespace characters should be replaced with a single one |
| 539 | ;; in the attributes |
| 540 | (let ((string (match-string-no-properties 1)) |
| 541 | (pos 0)) |
| 542 | (replace-regexp-in-string "\\s-\\{2,\\}" " " string) |
| 543 | (let ((expansion (xml-substitute-special string))) |
| 544 | (unless (stringp expansion) |
| 545 | ; We say this is the constraint. It is acctually that |
| 546 | ; external entities nor "<" can be in an attribute value. |
| 547 | (error "XML: (Not Well-Formed) Entities in attributes cannot expand into elements")) |
| 548 | (push (cons name expansion) attlist))) |
| 549 | |
| 550 | (goto-char end-pos) |
| 551 | (skip-syntax-forward " ")) |
| 552 | (nreverse attlist))) |
| 553 | |
| 554 | ;;******************************************************************* |
| 555 | ;;** |
| 556 | ;;** The DTD (document type declaration) |
| 557 | ;;** The following functions know how to skip or parse the DTD of |
| 558 | ;;** a document |
| 559 | ;;** |
| 560 | ;;******************************************************************* |
| 561 | |
| 562 | ;; Fixme: This fails at least if the DTD contains conditional sections. |
| 563 | |
| 564 | (defun xml-skip-dtd () |
| 565 | "Skip the DTD at point. |
| 566 | This follows the rule [28] in the XML specifications." |
| 567 | (let ((xml-validating-parser nil)) |
| 568 | (xml-parse-dtd))) |
| 569 | |
| 570 | (defun xml-parse-dtd (&optional parse-ns) |
| 571 | "Parse the DTD at point." |
| 572 | (forward-char (eval-when-compile (length "<!DOCTYPE"))) |
| 573 | (skip-syntax-forward " ") |
| 574 | (if (and (looking-at ">") |
| 575 | xml-validating-parser) |
| 576 | (error "XML: (Validity) Invalid DTD (expecting name of the document)")) |
| 577 | |
| 578 | ;; Get the name of the document |
| 579 | (looking-at xml-name-regexp) |
| 580 | (let ((dtd (list (match-string-no-properties 0) 'dtd)) |
| 581 | type element end-pos) |
| 582 | (goto-char (match-end 0)) |
| 583 | |
| 584 | (skip-syntax-forward " ") |
| 585 | ;; XML [75] |
| 586 | (cond ((looking-at "PUBLIC\\s-+") |
| 587 | (goto-char (match-end 0)) |
| 588 | (unless (or (re-search-forward |
| 589 | "\\=\"\\([[:space:][:alnum:]-'()+,./:=?;!*#@$_%]*\\)\"" |
| 590 | nil t) |
| 591 | (re-search-forward |
| 592 | "\\='\\([[:space:][:alnum:]-()+,./:=?;!*#@$_%]*\\)'" |
| 593 | nil t)) |
| 594 | (error "XML: Missing Public ID")) |
| 595 | (let ((pubid (match-string-no-properties 1))) |
| 596 | (skip-syntax-forward " ") |
| 597 | (unless (or (re-search-forward "\\='\\([^']*\\)'" nil t) |
| 598 | (re-search-forward "\\=\"\\([^\"]*\\)\"" nil t)) |
| 599 | (error "XML: Missing System ID")) |
| 600 | (push (list pubid (match-string-no-properties 1) 'public) dtd))) |
| 601 | ((looking-at "SYSTEM\\s-+") |
| 602 | (goto-char (match-end 0)) |
| 603 | (unless (or (re-search-forward "\\='\\([^']*\\)'" nil t) |
| 604 | (re-search-forward "\\=\"\\([^\"]*\\)\"" nil t)) |
| 605 | (error "XML: Missing System ID")) |
| 606 | (push (list (match-string-no-properties 1) 'system) dtd))) |
| 607 | (skip-syntax-forward " ") |
| 608 | (if (eq ?> (char-after)) |
| 609 | (forward-char) |
| 610 | (if (not (eq (char-after) ?\[)) |
| 611 | (error "XML: Bad DTD") |
| 612 | (forward-char) |
| 613 | ;; Parse the rest of the DTD |
| 614 | ;; Fixme: Deal with NOTATION, PIs. |
| 615 | (while (not (looking-at "\\s-*\\]")) |
| 616 | (skip-syntax-forward " ") |
| 617 | (cond |
| 618 | |
| 619 | ;; Translation of rule [45] of XML specifications |
| 620 | ((looking-at |
| 621 | "<!ELEMENT\\s-+\\([[:alnum:].%;]+\\)\\s-+\\([^>]+\\)>") |
| 622 | |
| 623 | (setq element (match-string-no-properties 1) |
| 624 | type (match-string-no-properties 2)) |
| 625 | (setq end-pos (match-end 0)) |
| 626 | |
| 627 | ;; Translation of rule [46] of XML specifications |
| 628 | (cond |
| 629 | ((string-match "^EMPTY[ \t\n\r]*$" type) ;; empty declaration |
| 630 | (setq type 'empty)) |
| 631 | ((string-match "^ANY[ \t\n\r]*$" type) ;; any type of contents |
| 632 | (setq type 'any)) |
| 633 | ((string-match "^(\\(.*\\))[ \t\n\r]*$" type) ;; children ([47]) |
| 634 | (setq type (xml-parse-elem-type (match-string-no-properties 1 type)))) |
| 635 | ((string-match "^%[^;]+;[ \t\n\r]*$" type) ;; substitution |
| 636 | nil) |
| 637 | (t |
| 638 | (if xml-validating-parser |
| 639 | (error "XML: (Validity) Invalid element type in the DTD")))) |
| 640 | |
| 641 | ;; rule [45]: the element declaration must be unique |
| 642 | (if (and (assoc element dtd) |
| 643 | xml-validating-parser) |
| 644 | (error "XML: (Validity) Element declarations must be unique in a DTD (<%s>)" |
| 645 | element)) |
| 646 | |
| 647 | ;; Store the element in the DTD |
| 648 | (push (list element type) dtd) |
| 649 | (goto-char end-pos)) |
| 650 | |
| 651 | ;; Translation of rule [52] of XML specifications |
| 652 | ((looking-at (concat "<!ATTLIST[ \t\n\r]*\\(" xml-name-re |
| 653 | "\\)[ \t\n\r]*\\(" xml-att-def-re |
| 654 | "\\)*[ \t\n\r]*>")) |
| 655 | |
| 656 | ;; We don't do anything with ATTLIST currently |
| 657 | (goto-char (match-end 0))) |
| 658 | |
| 659 | ((looking-at "<!--") |
| 660 | (search-forward "-->")) |
| 661 | ((looking-at (concat "<!ENTITY[ \t\n\r]*\\(" xml-name-re |
| 662 | "\\)[ \t\n\r]*\\(" xml-entity-value-re |
| 663 | "\\)[ \t\n\r]*>")) |
| 664 | (let ((name (match-string-no-properties 1)) |
| 665 | (value (substring (match-string-no-properties 2) 1 |
| 666 | (- (length (match-string-no-properties 2)) 1)))) |
| 667 | (goto-char (match-end 0)) |
| 668 | (setq xml-entity-alist |
| 669 | (append xml-entity-alist |
| 670 | (list (cons name |
| 671 | (with-temp-buffer |
| 672 | (insert value) |
| 673 | (goto-char (point-min)) |
| 674 | (xml-parse-fragment |
| 675 | xml-validating-parser |
| 676 | parse-ns)))))))) |
| 677 | ((or (looking-at (concat "<!ENTITY[ \t\n\r]+\\(" xml-name-re |
| 678 | "\\)[ \t\n\r]+SYSTEM[ \t\n\r]+" |
| 679 | "\\(\"[^\"]*\"\\|'[^']*'\\)[ \t\n\r]*>")) |
| 680 | (looking-at (concat "<!ENTITY[ \t\n\r]+\\(" xml-name-re |
| 681 | "\\)[ \t\n\r]+PUBLIC[ \t\n\r]+" |
| 682 | "\"[- \r\na-zA-Z0-9'()+,./:=?;!*#@$_%]*\"" |
| 683 | "\\|'[- \r\na-zA-Z0-9()+,./:=?;!*#@$_%]*'" |
| 684 | "[ \t\n\r]+\\(\"[^\"]*\"\\|'[^']*'\\)" |
| 685 | "[ \t\n\r]*>"))) |
| 686 | (let ((name (match-string-no-properties 1)) |
| 687 | (file (substring (match-string-no-properties 2) 1 |
| 688 | (- (length (match-string-no-properties 2)) 1)))) |
| 689 | (goto-char (match-end 0)) |
| 690 | (setq xml-entity-alist |
| 691 | (append xml-entity-alist |
| 692 | (list (cons name (with-temp-buffer |
| 693 | (insert-file-contents file) |
| 694 | (goto-char (point-min)) |
| 695 | (xml-parse-fragment |
| 696 | xml-validating-parser |
| 697 | parse-ns)))))))) |
| 698 | ;; skip parameter entity declarations |
| 699 | ((or (looking-at (concat "<!ENTITY[ \t\n\r]+%[ \t\n\r]+\\(" xml-name-re |
| 700 | "\\)[ \t\n\r]+SYSTEM[ \t\n\r]+" |
| 701 | "\\(\"[^\"]*\"\\|'[^']*'\\)[ \t\n\r]*>")) |
| 702 | (looking-at (concat "<!ENTITY[ \t\n\r]+" |
| 703 | "%[ \t\n\r]+" |
| 704 | "\\(" xml-name-re "\\)[ \t\n\r]+" |
| 705 | "PUBLIC[ \t\n\r]+" |
| 706 | "\\(\"[- \r\na-zA-Z0-9'()+,./:=?;!*#@$_%]*\"" |
| 707 | "\\|'[- \r\na-zA-Z0-9()+,./:=?;!*#@$_%]*'\\)[ \t\n\r]+" |
| 708 | "\\(\"[^\"]+\"\\|'[^']+'\\)" |
| 709 | "[ \t\n\r]*>"))) |
| 710 | (goto-char (match-end 0))) |
| 711 | ;; skip parameter entities |
| 712 | ((looking-at (concat "%" xml-name-re ";")) |
| 713 | (goto-char (match-end 0))) |
| 714 | (t |
| 715 | (when xml-validating-parser |
| 716 | (error "XML: (Validity) Invalid DTD item")))))) |
| 717 | (if (looking-at "\\s-*]>") |
| 718 | (goto-char (match-end 0)))) |
| 719 | (nreverse dtd))) |
| 720 | |
| 721 | (defun xml-parse-elem-type (string) |
| 722 | "Convert element type STRING into a Lisp structure." |
| 723 | |
| 724 | (let (elem modifier) |
| 725 | (if (string-match "(\\([^)]+\\))\\([+*?]?\\)" string) |
| 726 | (progn |
| 727 | (setq elem (match-string-no-properties 1 string) |
| 728 | modifier (match-string-no-properties 2 string)) |
| 729 | (if (string-match "|" elem) |
| 730 | (setq elem (cons 'choice |
| 731 | (mapcar 'xml-parse-elem-type |
| 732 | (split-string elem "|")))) |
| 733 | (if (string-match "," elem) |
| 734 | (setq elem (cons 'seq |
| 735 | (mapcar 'xml-parse-elem-type |
| 736 | (split-string elem ","))))))) |
| 737 | (if (string-match "[ \t\n\r]*\\([^+*?]+\\)\\([+*?]?\\)" string) |
| 738 | (setq elem (match-string-no-properties 1 string) |
| 739 | modifier (match-string-no-properties 2 string)))) |
| 740 | |
| 741 | (if (and (stringp elem) (string= elem "#PCDATA")) |
| 742 | (setq elem 'pcdata)) |
| 743 | |
| 744 | (cond |
| 745 | ((string= modifier "+") |
| 746 | (list '+ elem)) |
| 747 | ((string= modifier "*") |
| 748 | (list '* elem)) |
| 749 | ((string= modifier "?") |
| 750 | (list '\? elem)) |
| 751 | (t |
| 752 | elem)))) |
| 753 | |
| 754 | ;;******************************************************************* |
| 755 | ;;** |
| 756 | ;;** Substituting special XML sequences |
| 757 | ;;** |
| 758 | ;;******************************************************************* |
| 759 | |
| 760 | (defun xml-substitute-special (string) |
| 761 | "Return STRING, after subsituting entity references." |
| 762 | ;; This originally made repeated passes through the string from the |
| 763 | ;; beginning, which isn't correct, since then either "&amp;" or |
| 764 | ;; "&amp;" won't DTRT. |
| 765 | |
| 766 | (let ((point 0) |
| 767 | children end-point) |
| 768 | (while (string-match "&\\([^;]*\\);" string point) |
| 769 | (setq end-point (match-end 0)) |
| 770 | (let* ((this-part (match-string-no-properties 1 string)) |
| 771 | (prev-part (substring string point (match-beginning 0))) |
| 772 | (entity (assoc this-part xml-entity-alist)) |
| 773 | (expansion |
| 774 | (cond ((string-match "#\\([0-9]+\\)" this-part) |
| 775 | (let ((c (decode-char |
| 776 | 'ucs |
| 777 | (string-to-number (match-string-no-properties 1 this-part))))) |
| 778 | (if c (string c)))) |
| 779 | ((string-match "#x\\([[:xdigit:]]+\\)" this-part) |
| 780 | (let ((c (decode-char |
| 781 | 'ucs |
| 782 | (string-to-number (match-string-no-properties 1 this-part) 16)))) |
| 783 | (if c (string c)))) |
| 784 | (entity |
| 785 | (cdr entity)) |
| 786 | ((eq (length this-part) 0) |
| 787 | (error "XML: (Not Well-Formed) No entity given")) |
| 788 | (t |
| 789 | (if xml-validating-parser |
| 790 | (error "XML: (Validity) Undefined entity `%s'" |
| 791 | this-part) |
| 792 | xml-undefined-entity))))) |
| 793 | |
| 794 | (cond ((null children) |
| 795 | ;; FIXME: If we have an entity that expands into XML, this won't work. |
| 796 | (setq children |
| 797 | (concat prev-part expansion))) |
| 798 | ((stringp children) |
| 799 | (if (stringp expansion) |
| 800 | (setq children (concat children prev-part expansion)) |
| 801 | (setq children (list expansion (concat prev-part children))))) |
| 802 | ((and (stringp expansion) |
| 803 | (stringp (car children))) |
| 804 | (setcar children (concat prev-part expansion (car children)))) |
| 805 | ((stringp expansion) |
| 806 | (setq children (append (concat prev-part expansion) |
| 807 | children))) |
| 808 | ((stringp (car children)) |
| 809 | (setcar children (concat (car children) prev-part)) |
| 810 | (setq children (append expansion children))) |
| 811 | (t |
| 812 | (setq children (list expansion |
| 813 | prev-part |
| 814 | children)))) |
| 815 | (setq point end-point))) |
| 816 | (cond ((stringp children) |
| 817 | (concat children (substring string point))) |
| 818 | ((stringp (car (last children))) |
| 819 | (concat (car (last children)) (substring string point))) |
| 820 | ((null children) |
| 821 | string) |
| 822 | (t |
| 823 | (concat (mapconcat 'identity |
| 824 | (nreverse children) |
| 825 | "") |
| 826 | (substring string point)))))) |
| 827 | |
| 828 | ;;******************************************************************* |
| 829 | ;;** |
| 830 | ;;** Printing a tree. |
| 831 | ;;** This function is intended mainly for debugging purposes. |
| 832 | ;;** |
| 833 | ;;******************************************************************* |
| 834 | |
| 835 | (defun xml-debug-print (xml &optional indent-string) |
| 836 | "Outputs the XML in the current buffer. |
| 837 | XML can be a tree or a list of nodes. |
| 838 | The first line is indented with the optional INDENT-STRING." |
| 839 | (setq indent-string (or indent-string "")) |
| 840 | (dolist (node xml) |
| 841 | (xml-debug-print-internal node indent-string))) |
| 842 | |
| 843 | (defalias 'xml-print 'xml-debug-print) |
| 844 | |
| 845 | (defun xml-escape-string (string) |
| 846 | "Return the string with entity substitutions made from |
| 847 | xml-entity-alist." |
| 848 | (mapconcat (lambda (byte) |
| 849 | (let ((char (char-to-string byte))) |
| 850 | (if (rassoc char xml-entity-alist) |
| 851 | (concat "&" (car (rassoc char xml-entity-alist)) ";") |
| 852 | char))) |
| 853 | ;; This differs from the non-unicode branch. Just |
| 854 | ;; grabbing the string works here. |
| 855 | string "")) |
| 856 | |
| 857 | (defun xml-debug-print-internal (xml indent-string) |
| 858 | "Outputs the XML tree in the current buffer. |
| 859 | The first line is indented with INDENT-STRING." |
| 860 | (let ((tree xml) |
| 861 | attlist) |
| 862 | (insert indent-string ?< (symbol-name (xml-node-name tree))) |
| 863 | |
| 864 | ;; output the attribute list |
| 865 | (setq attlist (xml-node-attributes tree)) |
| 866 | (while attlist |
| 867 | (insert ?\ (symbol-name (caar attlist)) "=\"" |
| 868 | (xml-escape-string (cdar attlist)) ?\") |
| 869 | (setq attlist (cdr attlist))) |
| 870 | |
| 871 | (setq tree (xml-node-children tree)) |
| 872 | |
| 873 | (if (null tree) |
| 874 | (insert ?/ ?>) |
| 875 | (insert ?>) |
| 876 | |
| 877 | ;; output the children |
| 878 | (dolist (node tree) |
| 879 | (cond |
| 880 | ((listp node) |
| 881 | (insert ?\n) |
| 882 | (xml-debug-print-internal node (concat indent-string " "))) |
| 883 | ((stringp node) |
| 884 | (insert (xml-escape-string node))) |
| 885 | (t |
| 886 | (error "Invalid XML tree")))) |
| 887 | |
| 888 | (when (not (and (null (cdr tree)) |
| 889 | (stringp (car tree)))) |
| 890 | (insert ?\n indent-string)) |
| 891 | (insert ?< ?/ (symbol-name (xml-node-name xml)) ?>)))) |
| 892 | |
| 893 | (provide 'xml) |
| 894 | |
| 895 | ;; arch-tag: 5864b283-5a68-4b59-a20d-36a72b353b9b |
| 896 | ;;; xml.el ends here |