annotate lisp/xml.el @ 49403:7d900f9e80ee

*** empty log message ***
author Juanma Barranquero <lekktu@gmail.com>
date Thu, 23 Jan 2003 09:12:03 +0000
parents 4d76986458e8
children 6269b5c10aec
Ignore whitespace changes - Everywhere: Within whitespace: At end of lines:
rev   line source
38409
153f1b1f2efd Emacs lisp coding convention fixes.
Pavel Janík <Pavel@Janik.cz>
parents: 37958
diff changeset
1 ;;; xml.el --- XML parser
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
2
37958
d1fdbba91c71 (xml-parse-tag): The document may contain invalid characters.
Gerd Moellmann <gerd@gnu.org>
parents: 34825
diff changeset
3 ;; Copyright (C) 2000, 2001 Free Software Foundation, Inc.
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
4
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
5 ;; Author: Emmanuel Briot <briot@gnat.com>
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
6 ;; Maintainer: Emmanuel Briot <briot@gnat.com>
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
7 ;; Keywords: xml
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
8
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
9 ;; This file is part of GNU Emacs.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
10
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
11 ;; GNU Emacs is free software; you can redistribute it and/or modify
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
12 ;; it under the terms of the GNU General Public License as published by
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
13 ;; the Free Software Foundation; either version 2, or (at your option)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
14 ;; any later version.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
15
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
16 ;; GNU Emacs is distributed in the hope that it will be useful,
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
17 ;; but WITHOUT ANY WARRANTY; without even the implied warranty of
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
18 ;; MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
19 ;; GNU General Public License for more details.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
20
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
21 ;; You should have received a copy of the GNU General Public License
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
22 ;; along with GNU Emacs; see the file COPYING. If not, write to the
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
23 ;; Free Software Foundation, Inc., 59 Temple Place - Suite 330,
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
24 ;; Boston, MA 02111-1307, USA.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
25
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
26 ;;; Commentary:
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
27
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
28 ;; This file contains a full XML parser. It parses a file, and returns a list
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
29 ;; that can be used internally by any other lisp file.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
30 ;; See some example in todo.el
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
31
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
32 ;;; FILE FORMAT
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
33
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
34 ;; It does not parse the DTD, if present in the XML file, but knows how to
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
35 ;; ignore it. The XML file is assumed to be well-formed. In case of error, the
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
36 ;; parsing stops and the XML file is shown where the parsing stopped.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
37 ;;
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
38 ;; It also knows how to ignore comments, as well as the special ?xml? tag
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
39 ;; in the XML file.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
40 ;;
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
41 ;; The XML file should have the following format:
34825
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
42 ;; <node1 attr1="name1" attr2="name2" ...>value
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
43 ;; <node2 attr3="name3" attr4="name4">value2</node2>
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
44 ;; <node3 attr5="name5" attr6="name6">value3</node3>
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
45 ;; </node1>
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
46 ;; Of course, the name of the nodes and attributes can be anything. There can
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
47 ;; be any number of attributes (or none), as well as any number of children
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
48 ;; below the nodes.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
49 ;;
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
50 ;; There can be only top level node, but with any number of children below.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
51
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
52 ;;; LIST FORMAT
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
53
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
54 ;; The functions `xml-parse-file' and `xml-parse-tag' return a list with
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
55 ;; the following format:
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
56 ;;
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
57 ;; xml-list ::= (node node ...)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
58 ;; node ::= (tag_name attribute-list . child_node_list)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
59 ;; child_node_list ::= child_node child_node ...
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
60 ;; child_node ::= node | string
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
61 ;; tag_name ::= string
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
62 ;; attribute_list ::= (("attribute" . "value") ("attribute" . "value") ...)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
63 ;; | nil
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
64 ;; string ::= "..."
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
65 ;;
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
66 ;; Some macros are provided to ease the parsing of this list
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
67
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
68 ;;; Code:
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
69
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
70 ;;*******************************************************************
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
71 ;;**
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
72 ;;** Macros to parse the list
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
73 ;;**
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
74 ;;*******************************************************************
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
75
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
76 (defsubst xml-node-name (node)
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
77 "Return the tag associated with NODE.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
78 The tag is a lower-case symbol."
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
79 (car node))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
80
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
81 (defsubst xml-node-attributes (node)
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
82 "Return the list of attributes of NODE.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
83 The list can be nil."
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
84 (nth 1 node))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
85
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
86 (defsubst xml-node-children (node)
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
87 "Return the list of children of NODE.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
88 This is a list of nodes, and it can be nil."
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
89 (cddr node))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
90
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
91 (defun xml-get-children (node child-name)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
92 "Return the children of NODE whose tag is CHILD-NAME.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
93 CHILD-NAME should be a lower case symbol."
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
94 (let ((match ()))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
95 (dolist (child (xml-node-children node))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
96 (if child
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
97 (if (equal (xml-node-name child) child-name)
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
98 (push child match))))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
99 (nreverse match)))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
100
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
101 (defun xml-get-attribute (node attribute)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
102 "Get from NODE the value of ATTRIBUTE.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
103 An empty string is returned if the attribute was not found."
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
104 (if (xml-node-attributes node)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
105 (let ((value (assoc attribute (xml-node-attributes node))))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
106 (if value
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
107 (cdr value)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
108 ""))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
109 ""))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
110
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
111 ;;*******************************************************************
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
112 ;;**
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
113 ;;** Creating the list
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
114 ;;**
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
115 ;;*******************************************************************
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
116
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
117 (defun xml-parse-file (file &optional parse-dtd)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
118 "Parse the well-formed XML FILE.
34825
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
119 If FILE is already edited, this will keep the buffer alive.
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
120 Returns the top node with all its children.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
121 If PARSE-DTD is non-nil, the DTD is parsed rather than skipped."
34825
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
122 (let ((keep))
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
123 (if (get-file-buffer file)
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
124 (progn
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
125 (set-buffer (get-file-buffer file))
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
126 (setq keep (point)))
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
127 (find-file file))
49036
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
128
34825
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
129 (let ((xml (xml-parse-region (point-min)
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
130 (point-max)
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
131 (current-buffer)
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
132 parse-dtd)))
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
133 (if keep
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
134 (goto-char keep)
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
135 (kill-buffer (current-buffer)))
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
136 xml)))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
137
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
138 (defun xml-parse-region (beg end &optional buffer parse-dtd)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
139 "Parse the region from BEG to END in BUFFER.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
140 If BUFFER is nil, it defaults to the current buffer.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
141 Returns the XML list for the region, or raises an error if the region
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
142 is not a well-formed XML file.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
143 If PARSE-DTD is non-nil, the DTD is parsed rather than skipped,
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
144 and returned as the first element of the list"
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
145 (let (xml result dtd)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
146 (save-excursion
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
147 (if buffer
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
148 (set-buffer buffer))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
149 (goto-char beg)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
150 (while (< (point) end)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
151 (if (search-forward "<" end t)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
152 (progn
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
153 (forward-char -1)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
154 (if (null xml)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
155 (progn
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
156 (setq result (xml-parse-tag end parse-dtd))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
157 (cond
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
158 ((null result))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
159 ((listp (car result))
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
160 (setq dtd (car result))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
161 (add-to-list 'xml (cdr result)))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
162 (t
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
163 (add-to-list 'xml result))))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
164
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
165 ;; translation of rule [1] of XML specifications
38409
153f1b1f2efd Emacs lisp coding convention fixes.
Pavel Janík <Pavel@Janik.cz>
parents: 37958
diff changeset
166 (error "XML files can have only one toplevel tag")))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
167 (goto-char end)))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
168 (if parse-dtd
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
169 (cons dtd (reverse xml))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
170 (reverse xml)))))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
171
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
172
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
173 (defun xml-parse-tag (end &optional parse-dtd)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
174 "Parse the tag that is just in front of point.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
175 The end tag must be found before the position END in the current buffer.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
176 If PARSE-DTD is non-nil, the DTD of the document, if any, is parsed and
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
177 returned as the first element in the list.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
178 Returns one of:
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
179 - a list : the matching node
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
180 - nil : the point is not looking at a tag.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
181 - a cons cell: the first element is the DTD, the second is the node"
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
182 (cond
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
183 ;; Processing instructions (like the <?xml version="1.0"?> tag at the
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
184 ;; beginning of a document)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
185 ((looking-at "<\\?")
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
186 (search-forward "?>" end)
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
187 (goto-char (- (re-search-forward "[^[:space:]]") 1))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
188 (xml-parse-tag end))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
189 ;; Character data (CDATA) sections, in which no tag should be interpreted
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
190 ((looking-at "<!\\[CDATA\\[")
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
191 (let ((pos (match-end 0)))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
192 (unless (search-forward "]]>" end t)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
193 (error "CDATA section does not end anywhere in the document"))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
194 (buffer-substring-no-properties pos (match-beginning 0))))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
195 ;; DTD for the document
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
196 ((looking-at "<!DOCTYPE")
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
197 (let (dtd)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
198 (if parse-dtd
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
199 (setq dtd (xml-parse-dtd end))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
200 (xml-skip-dtd end))
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
201 (goto-char (- (re-search-forward "[^[:space:]]") 1))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
202 (if dtd
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
203 (cons dtd (xml-parse-tag end))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
204 (xml-parse-tag end))))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
205 ;; skip comments
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
206 ((looking-at "<!--")
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
207 (search-forward "-->" end)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
208 nil)
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
209 ;; end tag
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
210 ((looking-at "</")
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
211 '())
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
212 ;; opening tag
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
213 ((looking-at "<\\([^/>[:space:]]+\\)")
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
214 (goto-char (match-end 1))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
215 (let* ((case-fold-search nil) ;; XML is case-sensitive.
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
216 (node-name (match-string 1))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
217 ;; Parse the attribute list.
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
218 (children (list (xml-parse-attlist end) (intern node-name)))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
219 pos)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
220
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
221 ;; is this an empty element ?
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
222 (if (looking-at "/[[:space:]]*>")
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
223 (progn
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
224 (forward-char 2)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
225 (nreverse (cons '("") children)))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
226
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
227 ;; is this a valid start tag ?
40030
7507bd185307 (xml-parse-tag): Use eq on char-after's return value.
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 39407
diff changeset
228 (if (eq (char-after) ?>)
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
229 (progn
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
230 (forward-char 1)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
231 ;; Now check that we have the right end-tag. Note that this
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
232 ;; one might contain spaces after the tag name
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
233 (while (not (looking-at (concat "</" node-name "[[:space:]]*>")))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
234 (cond
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
235 ((looking-at "</")
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
236 (error (concat
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
237 "XML: invalid syntax -- invalid end tag (expecting "
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
238 node-name
34825
2cad4cde52bd (top level comment): Updated to reflect the fact that
Gerd Moellmann <gerd@gnu.org>
parents: 33977
diff changeset
239 ") at pos " (number-to-string (point)))))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
240 ((= (char-after) ?<)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
241 (let ((tag (xml-parse-tag end)))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
242 (when tag
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
243 (push tag children))))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
244 (t
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
245 (setq pos (point))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
246 (search-forward "<" end)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
247 (forward-char -1)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
248 (let ((string (buffer-substring-no-properties pos (point)))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
249 (pos 0))
49036
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
250
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
251 ;; Clean up the string (no newline characters)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
252 ;; Not done, since as per XML specifications, the XML processor
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
253 ;; should always pass the whole string to the application.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
254 ;; (while (string-match "\\s +" string pos)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
255 ;; (setq string (replace-match " " t t string))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
256 ;; (setq pos (1+ (match-beginning 0))))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
257
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
258 (setq string (xml-substitute-special string))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
259 (setq children
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
260 (if (stringp (car children))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
261 ;; The two strings were separated by a comment.
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
262 (cons (concat (car children) string)
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
263 (cdr children))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
264 (cons string children)))))))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
265 (goto-char (match-end 0))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
266 (if (> (point) end)
38409
153f1b1f2efd Emacs lisp coding convention fixes.
Pavel Janík <Pavel@Janik.cz>
parents: 37958
diff changeset
267 (error "XML: End tag for %s not found before end of region"
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
268 node-name))
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
269 (nreverse children))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
270
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
271 ;; This was an invalid start tag
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
272 (error "XML: Invalid attribute list")
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
273 ))))
37958
d1fdbba91c71 (xml-parse-tag): The document may contain invalid characters.
Gerd Moellmann <gerd@gnu.org>
parents: 34825
diff changeset
274 (t ;; This is not a tag.
38409
153f1b1f2efd Emacs lisp coding convention fixes.
Pavel Janík <Pavel@Janik.cz>
parents: 37958
diff changeset
275 (error "XML: Invalid character"))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
276 ))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
277
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
278 (defun xml-parse-attlist (end)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
279 "Return the attribute-list that point is looking at.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
280 The search for attributes end at the position END in the current buffer.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
281 Leaves the point on the first non-blank character after the tag."
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
282 (let ((attlist ())
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
283 name)
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
284 (goto-char (- (re-search-forward "[^[:space:]]") 1))
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
285 (while (looking-at "\\([a-zA-Z_:][-a-zA-Z0-9._:]*\\)[[:space:]]*=[[:space:]]*")
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
286 (setq name (intern (match-string 1)))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
287 (goto-char (match-end 0))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
288
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
289 ;; Do we have a string between quotes (or double-quotes),
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
290 ;; or a simple word ?
43739
373960858ccc * xml.el (xml-parse-attlist): Accept empty strings.
ShengHuo ZHU <zsh@cs.rochester.edu>
parents: 42031
diff changeset
291 (unless (looking-at "\"\\([^\"]*\\)\"")
373960858ccc * xml.el (xml-parse-attlist): Accept empty strings.
ShengHuo ZHU <zsh@cs.rochester.edu>
parents: 42031
diff changeset
292 (unless (looking-at "'\\([^']*\\)'")
38409
153f1b1f2efd Emacs lisp coding convention fixes.
Pavel Janík <Pavel@Janik.cz>
parents: 37958
diff changeset
293 (error "XML: Attribute values must be given between quotes")))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
294
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
295 ;; Each attribute must be unique within a given element
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
296 (if (assoc name attlist)
38409
153f1b1f2efd Emacs lisp coding convention fixes.
Pavel Janík <Pavel@Janik.cz>
parents: 37958
diff changeset
297 (error "XML: each attribute must be unique within an element"))
49036
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
298
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
299 (push (cons name (match-string-no-properties 1)) attlist)
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
300 (goto-char (match-end 0))
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
301 (goto-char (- (re-search-forward "[^[:space:]]") 1))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
302 (if (> (point) end)
38409
153f1b1f2efd Emacs lisp coding convention fixes.
Pavel Janík <Pavel@Janik.cz>
parents: 37958
diff changeset
303 (error "XML: end of attribute list not found before end of region"))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
304 )
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
305 (nreverse attlist)))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
306
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
307 ;;*******************************************************************
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
308 ;;**
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
309 ;;** The DTD (document type declaration)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
310 ;;** The following functions know how to skip or parse the DTD of
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
311 ;;** a document
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
312 ;;**
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
313 ;;*******************************************************************
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
314
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
315 (defun xml-skip-dtd (end)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
316 "Skip the DTD that point is looking at.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
317 The DTD must end before the position END in the current buffer.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
318 The point must be just before the starting tag of the DTD.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
319 This follows the rule [28] in the XML specifications."
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
320 (forward-char (length "<!DOCTYPE"))
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
321 (if (looking-at "[[:space:]]*>")
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
322 (error "XML: invalid DTD (excepting name of the document)"))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
323 (condition-case nil
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
324 (progn
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
325 (forward-word 1) ;; name of the document
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
326 (goto-char (- (re-search-forward "[^[:space:]]") 1))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
327 (if (looking-at "\\[")
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
328 (re-search-forward "\\][[:space:]]*>" end)
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
329 (search-forward ">" end)))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
330 (error (error "XML: No end to the DTD"))))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
331
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
332 (defun xml-parse-dtd (end)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
333 "Parse the DTD that point is looking at.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
334 The DTD must end before the position END in the current buffer."
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
335 (forward-char (length "<!DOCTYPE"))
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
336 (goto-char (- (re-search-forward "[^[:space:]]") 1))
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
337 (if (looking-at ">")
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
338 (error "XML: invalid DTD (excepting name of the document)"))
49036
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
339
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
340 ;; Get the name of the document
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
341 (looking-at "\\sw+")
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
342 (let ((dtd (list (match-string-no-properties 0) 'dtd))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
343 type element end-pos)
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
344 (goto-char (match-end 0))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
345
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
346 (goto-char (- (re-search-forward "[^[:space:]]") 1))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
347
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
348 ;; External DTDs => don't know how to handle them yet
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
349 (if (looking-at "SYSTEM")
38409
153f1b1f2efd Emacs lisp coding convention fixes.
Pavel Janík <Pavel@Janik.cz>
parents: 37958
diff changeset
350 (error "XML: Don't know how to handle external DTDs"))
49036
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
351
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
352 (if (not (= (char-after) ?\[))
38409
153f1b1f2efd Emacs lisp coding convention fixes.
Pavel Janík <Pavel@Janik.cz>
parents: 37958
diff changeset
353 (error "XML: Unknown declaration in the DTD"))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
354
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
355 ;; Parse the rest of the DTD
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
356 (forward-char 1)
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
357 (while (and (not (looking-at "[[:space:]]*\\]"))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
358 (<= (point) end))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
359 (cond
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
360
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
361 ;; Translation of rule [45] of XML specifications
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
362 ((looking-at
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
363 "[[:space:]]*<!ELEMENT[[:space:]]+\\([a-zA-Z0-9.%;]+\\)[[:space:]]+\\([^>]+\\)>")
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
364
30779
aa097d8d4f1a (xml-parse-tag, xml-parse-attlist): Do not downcase
Gerd Moellmann <gerd@gnu.org>
parents: 30329
diff changeset
365 (setq element (intern (match-string-no-properties 1))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
366 type (match-string-no-properties 2))
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
367 (setq end-pos (match-end 0))
49036
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
368
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
369 ;; Translation of rule [46] of XML specifications
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
370 (cond
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
371 ((string-match "^EMPTY[[:space:]]*$" type) ;; empty declaration
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
372 (setq type 'empty))
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
373 ((string-match "^ANY[[:space:]]*$" type) ;; any type of contents
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
374 (setq type 'any))
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
375 ((string-match "^(\\(.*\\))[[:space:]]*$" type) ;; children ([47])
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
376 (setq type (xml-parse-elem-type (match-string-no-properties 1 type))))
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
377 ((string-match "^%[^;]+;[[:space:]]*$" type) ;; substitution
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
378 nil)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
379 (t
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
380 (error "XML: Invalid element type in the DTD")))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
381
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
382 ;; rule [45]: the element declaration must be unique
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
383 (if (assoc element dtd)
38409
153f1b1f2efd Emacs lisp coding convention fixes.
Pavel Janík <Pavel@Janik.cz>
parents: 37958
diff changeset
384 (error "XML: elements declaration must be unique in a DTD (<%s>)"
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
385 (symbol-name element)))
49036
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
386
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
387 ;; Store the element in the DTD
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
388 (push (list element type) dtd)
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
389 (goto-char end-pos))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
390
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
391
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
392 (t
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
393 (error "XML: Invalid DTD item"))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
394 )
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
395 )
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
396
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
397 ;; Skip the end of the DTD
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
398 (search-forward ">" end)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
399 (nreverse dtd)))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
400
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
401
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
402 (defun xml-parse-elem-type (string)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
403 "Convert a STRING for an element type into an elisp structure."
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
404
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
405 (let (elem modifier)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
406 (if (string-match "(\\([^)]+\\))\\([+*?]?\\)" string)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
407 (progn
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
408 (setq elem (match-string 1 string)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
409 modifier (match-string 2 string))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
410 (if (string-match "|" elem)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
411 (setq elem (cons 'choice
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
412 (mapcar 'xml-parse-elem-type
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
413 (split-string elem "|"))))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
414 (if (string-match "," elem)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
415 (setq elem (cons 'seq
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
416 (mapcar 'xml-parse-elem-type
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
417 (split-string elem ","))))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
418 )))
49133
4d76986458e8 (xml-parse-tag, xml-parse-attlist, xml-skip-dtd, xml-parse-dtd,
Juanma Barranquero <lekktu@gmail.com>
parents: 49065
diff changeset
419 (if (string-match "[[:space:]]*\\([^+*?]+\\)\\([+*?]?\\)" string)
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
420 (setq elem (match-string 1 string)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
421 modifier (match-string 2 string))))
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
422
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
423 (if (and (stringp elem) (string= elem "#PCDATA"))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
424 (setq elem 'pcdata))
49036
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
425
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
426 (cond
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
427 ((string= modifier "+")
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
428 (list '+ elem))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
429 ((string= modifier "*")
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
430 (list '* elem))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
431 ((string= modifier "?")
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
432 (list '? elem))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
433 (t
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
434 elem))))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
435
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
436
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
437 ;;*******************************************************************
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
438 ;;**
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
439 ;;** Substituting special XML sequences
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
440 ;;**
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
441 ;;*******************************************************************
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
442
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
443 (defun xml-substitute-special (string)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
444 "Return STRING, after subsituting special XML sequences."
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
445 (while (string-match "&lt;" string)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
446 (setq string (replace-match "<" t nil string)))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
447 (while (string-match "&gt;" string)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
448 (setq string (replace-match ">" t nil string)))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
449 (while (string-match "&apos;" string)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
450 (setq string (replace-match "'" t nil string)))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
451 (while (string-match "&quot;" string)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
452 (setq string (replace-match "\"" t nil string)))
49065
265dc22fb2c0 Comment change.
Richard M. Stallman <rms@gnu.org>
parents: 49036
diff changeset
453 ;; This goes last so it doesn't confuse the matches above.
49036
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
454 (while (string-match "&amp;" string)
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
455 (setq string (replace-match "&" t nil string)))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
456 string)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
457
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
458 ;;*******************************************************************
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
459 ;;**
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
460 ;;** Printing a tree.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
461 ;;** This function is intended mainly for debugging purposes.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
462 ;;**
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
463 ;;*******************************************************************
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
464
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
465 (defun xml-debug-print (xml)
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
466 (dolist (node xml)
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
467 (xml-debug-print-internal node "")))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
468
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
469 (defun xml-debug-print-internal (xml indent-string)
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
470 "Outputs the XML tree in the current buffer.
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
471 The first line indented with INDENT-STRING."
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
472 (let ((tree xml)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
473 attlist)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
474 (insert indent-string "<" (symbol-name (xml-node-name tree)))
49036
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
475
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
476 ;; output the attribute list
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
477 (setq attlist (xml-node-attributes tree))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
478 (while attlist
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
479 (insert " ")
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
480 (insert (symbol-name (caar attlist)) "=\"" (cdar attlist) "\"")
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
481 (setq attlist (cdr attlist)))
49036
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
482
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
483 (insert ">")
49036
466922eb2b8d (xml-substitute-special): Move "&amp;" -> "&" last.
Thien-Thi Nguyen <ttn@gnuvola.org>
parents: 48869
diff changeset
484
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
485 (setq tree (xml-node-children tree))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
486
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
487 ;; output the children
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
488 (dolist (node tree)
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
489 (cond
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
490 ((listp node)
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
491 (insert "\n")
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
492 (xml-debug-print-internal node (concat indent-string " ")))
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
493 ((stringp node) (insert node))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
494 (t
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
495 (error "Invalid XML tree"))))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
496
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
497 (insert "\n" indent-string
42031
54db4085a7df Use setq rather than (set 'foo bar).
Stefan Monnier <monnier@iro.umontreal.ca>
parents: 40030
diff changeset
498 "</" (symbol-name (xml-node-name xml)) ">")))
30329
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
499
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
500 (provide 'xml)
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
501
1ea701655cf8 *** empty log message ***
Gerd Moellmann <gerd@gnu.org>
parents:
diff changeset
502 ;;; xml.el ends here