Mercurial > genshi > genshi-test
annotate genshi/output.py @ 658:a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
author | cmlenz |
---|---|
date | Thu, 22 Nov 2007 22:07:15 +0000 |
parents | 49848aaa7839 |
children | 641d76ae430c |
rev | line source |
---|---|
1 | 1 # -*- coding: utf-8 -*- |
2 # | |
408 | 3 # Copyright (C) 2006-2007 Edgewall Software |
1 | 4 # All rights reserved. |
5 # | |
6 # This software is licensed as described in the file COPYING, which | |
7 # you should have received as part of this distribution. The terms | |
230 | 8 # are also available at http://genshi.edgewall.org/wiki/License. |
1 | 9 # |
10 # This software consists of voluntary contributions made by many | |
11 # individuals. For the exact contribution history, see the revision | |
230 | 12 # history and logs, available at http://genshi.edgewall.org/log/. |
1 | 13 |
14 """This module provides different kinds of serialization methods for XML event | |
15 streams. | |
16 """ | |
17 | |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
18 from itertools import chain |
1 | 19 try: |
20 frozenset | |
21 except NameError: | |
22 from sets import ImmutableSet as frozenset | |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
23 import re |
1 | 24 |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
25 from genshi.core import escape, Attrs, Markup, Namespace, QName, StreamEventKind |
460
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
26 from genshi.core import START, END, TEXT, XML_DECL, DOCTYPE, START_NS, END_NS, \ |
402
cc7f5b3fbbed
Fix output of namespace declarations for namespace URLs appearing more than once in a stream. Thanks to Jeff Cutsinger for reporting the problem.
cmlenz
parents:
397
diff
changeset
|
27 START_CDATA, END_CDATA, PI, COMMENT, XML_NAMESPACE |
1 | 28 |
462 | 29 __all__ = ['encode', 'get_serializer', 'DocType', 'XMLSerializer', |
30 'XHTMLSerializer', 'HTMLSerializer', 'TextSerializer'] | |
425
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
31 __docformat__ = 'restructuredtext en' |
1 | 32 |
462 | 33 def encode(iterator, method='xml', encoding='utf-8'): |
34 """Encode serializer output into a string. | |
35 | |
36 :param iterator: the iterator returned from serializing a stream (basically | |
37 any iterator that yields unicode objects) | |
38 :param method: the serialization method; determines how characters not | |
39 representable in the specified encoding are treated | |
40 :param encoding: how the output string should be encoded; if set to `None`, | |
41 this method returns a `unicode` object | |
42 :return: a string or unicode object (depending on the `encoding` parameter) | |
43 :since: version 0.4.1 | |
44 """ | |
45 output = u''.join(list(iterator)) | |
46 if encoding is not None: | |
47 errors = 'replace' | |
48 if method != 'text' and not isinstance(method, TextSerializer): | |
49 errors = 'xmlcharrefreplace' | |
50 return output.encode(encoding, errors) | |
51 return output | |
52 | |
53 def get_serializer(method='xml', **kwargs): | |
54 """Return a serializer object for the given method. | |
55 | |
56 :param method: the serialization method; can be either "xml", "xhtml", | |
57 "html", "text", or a custom serializer class | |
58 | |
59 Any additional keyword arguments are passed to the serializer, and thus | |
60 depend on the `method` parameter value. | |
61 | |
62 :see: `XMLSerializer`, `XHTMLSerializer`, `HTMLSerializer`, `TextSerializer` | |
63 :since: version 0.4.1 | |
64 """ | |
65 if isinstance(method, basestring): | |
66 method = {'xml': XMLSerializer, | |
67 'xhtml': XHTMLSerializer, | |
68 'html': HTMLSerializer, | |
69 'text': TextSerializer}[method.lower()] | |
70 return method(**kwargs) | |
71 | |
1 | 72 |
85 | 73 class DocType(object): |
74 """Defines a number of commonly used DOCTYPE declarations as constants.""" | |
75 | |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
76 HTML_STRICT = ( |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
77 'html', '-//W3C//DTD HTML 4.01//EN', |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
78 'http://www.w3.org/TR/html4/strict.dtd' |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
79 ) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
80 HTML_TRANSITIONAL = ( |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
81 'html', '-//W3C//DTD HTML 4.01 Transitional//EN', |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
82 'http://www.w3.org/TR/html4/loose.dtd' |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
83 ) |
464
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
84 HTML_FRAMESET = ( |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
85 'html', '-//W3C//DTD HTML 4.01 Frameset//EN', |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
86 'http://www.w3.org/TR/html4/frameset.dtd' |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
87 ) |
85 | 88 HTML = HTML_STRICT |
89 | |
464
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
90 HTML5 = ('html', None, None) |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
91 |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
92 XHTML_STRICT = ( |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
93 'html', '-//W3C//DTD XHTML 1.0 Strict//EN', |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
94 'http://www.w3.org/TR/xhtml1/DTD/xhtml1-strict.dtd' |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
95 ) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
96 XHTML_TRANSITIONAL = ( |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
97 'html', '-//W3C//DTD XHTML 1.0 Transitional//EN', |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
98 'http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd' |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
99 ) |
464
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
100 XHTML_FRAMESET = ( |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
101 'html', '-//W3C//DTD XHTML 1.0 Frameset//EN', |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
102 'http://www.w3.org/TR/xhtml1/DTD/xhtml1-frameset.dtd' |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
103 ) |
85 | 104 XHTML = XHTML_STRICT |
105 | |
464
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
106 def get(cls, name): |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
107 """Return the ``(name, pubid, sysid)`` tuple of the ``DOCTYPE`` |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
108 declaration for the specified name. |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
109 |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
110 The following names are recognized in this version: |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
111 * "html" or "html-strict" for the HTML 4.01 strict DTD |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
112 * "html-transitional" for the HTML 4.01 transitional DTD |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
113 * "html-transitional" for the HTML 4.01 frameset DTD |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
114 * "html5" for the ``DOCTYPE`` proposed for HTML5 |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
115 * "xhtml" or "xhtml-strict" for the XHTML 1.0 strict DTD |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
116 * "xhtml-transitional" for the XHTML 1.0 transitional DTD |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
117 * "xhtml-frameset" for the XHTML 1.0 frameset DTD |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
118 |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
119 :param name: the name of the ``DOCTYPE`` |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
120 :return: the ``(name, pubid, sysid)`` tuple for the requested |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
121 ``DOCTYPE``, or ``None`` if the name is not recognized |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
122 :since: version 0.4.1 |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
123 """ |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
124 return { |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
125 'html': cls.HTML, 'html-strict': cls.HTML_STRICT, |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
126 'html-transitional': DocType.HTML_TRANSITIONAL, |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
127 'html-frameset': DocType.HTML_FRAMESET, |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
128 'html5': cls.HTML5, |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
129 'xhtml': cls.XHTML, 'xhtml-strict': cls.XHTML_STRICT, |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
130 'xhtml-transitional': cls.XHTML_TRANSITIONAL, |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
131 'xhtml-frameset': cls.XHTML_FRAMESET, |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
132 }.get(name.lower()) |
dafc6f4c20fb
Move the mapping of doctype names to tuples out of the plugin into the `DocType` class.
cmlenz
parents:
462
diff
changeset
|
133 get = classmethod(get) |
448 | 134 |
85 | 135 |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
136 class XMLSerializer(object): |
1 | 137 """Produces XML text from an event stream. |
138 | |
230 | 139 >>> from genshi.builder import tag |
20 | 140 >>> elem = tag.div(tag.a(href='foo'), tag.br, tag.hr(noshade=True)) |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
141 >>> print ''.join(XMLSerializer()(elem.generate())) |
1 | 142 <div><a href="foo"/><br/><hr noshade="True"/></div> |
143 """ | |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
144 |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
145 _PRESERVE_SPACE = frozenset() |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
146 |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
147 def __init__(self, doctype=None, strip_whitespace=True, |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
148 namespace_prefixes=None): |
85 | 149 """Initialize the XML serializer. |
150 | |
425
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
151 :param doctype: a ``(name, pubid, sysid)`` tuple that represents the |
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
152 DOCTYPE declaration that should be included at the top |
494
8a90e761d5ff
The `doctype` parameter for serializers can now be a string.
cmlenz
parents:
464
diff
changeset
|
153 of the generated output, or the name of a DOCTYPE as |
8a90e761d5ff
The `doctype` parameter for serializers can now be a string.
cmlenz
parents:
464
diff
changeset
|
154 defined in `DocType.get` |
425
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
155 :param strip_whitespace: whether extraneous whitespace should be |
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
156 stripped from the output |
494
8a90e761d5ff
The `doctype` parameter for serializers can now be a string.
cmlenz
parents:
464
diff
changeset
|
157 :note: Changed in 0.4.2: The `doctype` parameter can now be a string. |
85 | 158 """ |
159 self.preamble = [] | |
160 if doctype: | |
494
8a90e761d5ff
The `doctype` parameter for serializers can now be a string.
cmlenz
parents:
464
diff
changeset
|
161 if isinstance(doctype, basestring): |
8a90e761d5ff
The `doctype` parameter for serializers can now be a string.
cmlenz
parents:
464
diff
changeset
|
162 doctype = DocType.get(doctype) |
85 | 163 self.preamble.append((DOCTYPE, doctype, (None, -1, -1))) |
212
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
164 self.filters = [EmptyTagFilter()] |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
165 if strip_whitespace: |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
166 self.filters.append(WhitespaceFilter(self._PRESERVE_SPACE)) |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
167 self.filters.append(NamespaceFlattener(prefixes=namespace_prefixes)) |
1 | 168 |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
169 def __call__(self, stream): |
460
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
170 have_decl = have_doctype = False |
143
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
171 in_cdata = False |
1 | 172 |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
173 stream = chain(self.preamble, stream) |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
174 for filter_ in self.filters: |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
175 stream = filter_(stream) |
1 | 176 for kind, data, pos in stream: |
177 | |
212
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
178 if kind is START or kind is EMPTY: |
1 | 179 tag, attrib = data |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
180 buf = ['<', tag] |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
181 for attr, value in attrib: |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
182 buf += [' ', attr, '="', escape(value), '"'] |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
183 buf.append(kind is EMPTY and '/>' or '>') |
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
184 yield Markup(u''.join(buf)) |
1 | 185 |
69 | 186 elif kind is END: |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
187 yield Markup('</%s>' % data) |
1 | 188 |
69 | 189 elif kind is TEXT: |
143
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
190 if in_cdata: |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
191 yield data |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
192 else: |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
193 yield escape(data, quotes=False) |
1 | 194 |
89
d4c7617900e3
Support comments in templates that are not included in the output, in the same way Kid does: if the comment text starts with a `!` character, it is stripped from the output.
cmlenz
parents:
85
diff
changeset
|
195 elif kind is COMMENT: |
d4c7617900e3
Support comments in templates that are not included in the output, in the same way Kid does: if the comment text starts with a `!` character, it is stripped from the output.
cmlenz
parents:
85
diff
changeset
|
196 yield Markup('<!--%s-->' % data) |
d4c7617900e3
Support comments in templates that are not included in the output, in the same way Kid does: if the comment text starts with a `!` character, it is stripped from the output.
cmlenz
parents:
85
diff
changeset
|
197 |
460
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
198 elif kind is XML_DECL and not have_decl: |
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
199 version, encoding, standalone = data |
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
200 buf = ['<?xml version="%s"' % version] |
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
201 if encoding: |
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
202 buf.append(' encoding="%s"' % encoding) |
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
203 if standalone != -1: |
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
204 standalone = standalone and 'yes' or 'no' |
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
205 buf.append(' standalone="%s"' % standalone) |
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
206 buf.append('?>\n') |
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
207 yield Markup(u''.join(buf)) |
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
208 have_decl = True |
6b5544bb5a99
Apply patch by Alec Thomas for processing XML declarations (#111). Thanks!
cmlenz
parents:
448
diff
changeset
|
209 |
136 | 210 elif kind is DOCTYPE and not have_doctype: |
211 name, pubid, sysid = data | |
212 buf = ['<!DOCTYPE %s'] | |
213 if pubid: | |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
214 buf.append(' PUBLIC "%s"') |
136 | 215 elif sysid: |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
216 buf.append(' SYSTEM') |
136 | 217 if sysid: |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
218 buf.append(' "%s"') |
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
219 buf.append('>\n') |
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
220 yield Markup(u''.join(buf), *filter(None, data)) |
136 | 221 have_doctype = True |
109
2de3f9d84a1c
Reorder the conditional branches in the serializers so that the more common event kinds are on top.
cmlenz
parents:
105
diff
changeset
|
222 |
143
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
223 elif kind is START_CDATA: |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
224 yield Markup('<![CDATA[') |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
225 in_cdata = True |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
226 |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
227 elif kind is END_CDATA: |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
228 yield Markup(']]>') |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
229 in_cdata = False |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
230 |
105
334a338847af
Include processing instructions in serialized streams.
cmlenz
parents:
96
diff
changeset
|
231 elif kind is PI: |
334a338847af
Include processing instructions in serialized streams.
cmlenz
parents:
96
diff
changeset
|
232 yield Markup('<?%s %s?>' % data) |
334a338847af
Include processing instructions in serialized streams.
cmlenz
parents:
96
diff
changeset
|
233 |
1 | 234 |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
235 class XHTMLSerializer(XMLSerializer): |
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
236 """Produces XHTML text from an event stream. |
1 | 237 |
230 | 238 >>> from genshi.builder import tag |
20 | 239 >>> elem = tag.div(tag.a(href='foo'), tag.br, tag.hr(noshade=True)) |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
240 >>> print ''.join(XHTMLSerializer()(elem.generate())) |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
241 <div><a href="foo"></a><br /><hr noshade="noshade" /></div> |
1 | 242 """ |
243 | |
244 _EMPTY_ELEMS = frozenset(['area', 'base', 'basefont', 'br', 'col', 'frame', | |
245 'hr', 'img', 'input', 'isindex', 'link', 'meta', | |
246 'param']) | |
247 _BOOLEAN_ATTRS = frozenset(['selected', 'checked', 'compact', 'declare', | |
248 'defer', 'disabled', 'ismap', 'multiple', | |
249 'nohref', 'noresize', 'noshade', 'nowrap']) | |
346
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
250 _PRESERVE_SPACE = frozenset([ |
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
251 QName('pre'), QName('http://www.w3.org/1999/xhtml}pre'), |
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
252 QName('textarea'), QName('http://www.w3.org/1999/xhtml}textarea') |
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
253 ]) |
1 | 254 |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
255 def __init__(self, doctype=None, strip_whitespace=True, |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
256 namespace_prefixes=None): |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
257 super(XHTMLSerializer, self).__init__(doctype, False) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
258 self.filters = [EmptyTagFilter()] |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
259 if strip_whitespace: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
260 self.filters.append(WhitespaceFilter(self._PRESERVE_SPACE)) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
261 namespace_prefixes = namespace_prefixes or {} |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
262 namespace_prefixes['http://www.w3.org/1999/xhtml'] = '' |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
263 self.filters.append(NamespaceFlattener(prefixes=namespace_prefixes)) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
264 |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
265 def __call__(self, stream): |
136 | 266 boolean_attrs = self._BOOLEAN_ATTRS |
267 empty_elems = self._EMPTY_ELEMS | |
85 | 268 have_doctype = False |
143
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
269 in_cdata = False |
1 | 270 |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
271 stream = chain(self.preamble, stream) |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
272 for filter_ in self.filters: |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
273 stream = filter_(stream) |
1 | 274 for kind, data, pos in stream: |
275 | |
212
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
276 if kind is START or kind is EMPTY: |
1 | 277 tag, attrib = data |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
278 buf = ['<', tag] |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
279 for attr, value in attrib: |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
280 if attr in boolean_attrs: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
281 value = attr |
524
49848aaa7839
Add special handling for `xml:lang` to HTML/XHTML serialization.
cmlenz
parents:
494
diff
changeset
|
282 elif attr == u'xml:lang' and u'lang' not in attrib: |
49848aaa7839
Add special handling for `xml:lang` to HTML/XHTML serialization.
cmlenz
parents:
494
diff
changeset
|
283 buf += [' lang="', escape(value), '"'] |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
284 buf += [' ', attr, '="', escape(value), '"'] |
212
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
285 if kind is EMPTY: |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
286 if tag in empty_elems: |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
287 buf.append(' />') |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
288 else: |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
289 buf.append('></%s>' % tag) |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
290 else: |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
291 buf.append('>') |
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
292 yield Markup(u''.join(buf)) |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
293 |
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
294 elif kind is END: |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
295 yield Markup('</%s>' % data) |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
296 |
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
297 elif kind is TEXT: |
143
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
298 if in_cdata: |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
299 yield data |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
300 else: |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
301 yield escape(data, quotes=False) |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
302 |
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
303 elif kind is COMMENT: |
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
304 yield Markup('<!--%s-->' % data) |
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
305 |
136 | 306 elif kind is DOCTYPE and not have_doctype: |
307 name, pubid, sysid = data | |
308 buf = ['<!DOCTYPE %s'] | |
309 if pubid: | |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
310 buf.append(' PUBLIC "%s"') |
136 | 311 elif sysid: |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
312 buf.append(' SYSTEM') |
136 | 313 if sysid: |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
314 buf.append(' "%s"') |
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
315 buf.append('>\n') |
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
316 yield Markup(u''.join(buf), *filter(None, data)) |
136 | 317 have_doctype = True |
109
2de3f9d84a1c
Reorder the conditional branches in the serializers so that the more common event kinds are on top.
cmlenz
parents:
105
diff
changeset
|
318 |
143
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
319 elif kind is START_CDATA: |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
320 yield Markup('<![CDATA[') |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
321 in_cdata = True |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
322 |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
323 elif kind is END_CDATA: |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
324 yield Markup(']]>') |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
325 in_cdata = False |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
326 |
105
334a338847af
Include processing instructions in serialized streams.
cmlenz
parents:
96
diff
changeset
|
327 elif kind is PI: |
334a338847af
Include processing instructions in serialized streams.
cmlenz
parents:
96
diff
changeset
|
328 yield Markup('<?%s %s?>' % data) |
334a338847af
Include processing instructions in serialized streams.
cmlenz
parents:
96
diff
changeset
|
329 |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
330 |
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
331 class HTMLSerializer(XHTMLSerializer): |
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
332 """Produces HTML text from an event stream. |
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
333 |
230 | 334 >>> from genshi.builder import tag |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
335 >>> elem = tag.div(tag.a(href='foo'), tag.br, tag.hr(noshade=True)) |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
336 >>> print ''.join(HTMLSerializer()(elem.generate())) |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
337 <div><a href="foo"></a><br><hr noshade></div> |
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
338 """ |
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
339 |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
340 _NOESCAPE_ELEMS = frozenset([ |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
341 QName('script'), QName('http://www.w3.org/1999/xhtml}script'), |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
342 QName('style'), QName('http://www.w3.org/1999/xhtml}style') |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
343 ]) |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
344 |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
345 def __init__(self, doctype=None, strip_whitespace=True): |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
346 """Initialize the HTML serializer. |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
347 |
425
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
348 :param doctype: a ``(name, pubid, sysid)`` tuple that represents the |
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
349 DOCTYPE declaration that should be included at the top |
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
350 of the generated output |
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
351 :param strip_whitespace: whether extraneous whitespace should be |
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
352 stripped from the output |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
353 """ |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
354 super(HTMLSerializer, self).__init__(doctype, False) |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
355 self.filters = [EmptyTagFilter()] |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
356 if strip_whitespace: |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
357 self.filters.append(WhitespaceFilter(self._PRESERVE_SPACE, |
305 | 358 self._NOESCAPE_ELEMS)) |
524
49848aaa7839
Add special handling for `xml:lang` to HTML/XHTML serialization.
cmlenz
parents:
494
diff
changeset
|
359 self.filters.append(NamespaceFlattener(prefixes={ |
49848aaa7839
Add special handling for `xml:lang` to HTML/XHTML serialization.
cmlenz
parents:
494
diff
changeset
|
360 'http://www.w3.org/1999/xhtml': '' |
49848aaa7839
Add special handling for `xml:lang` to HTML/XHTML serialization.
cmlenz
parents:
494
diff
changeset
|
361 })) |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
362 |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
363 def __call__(self, stream): |
136 | 364 boolean_attrs = self._BOOLEAN_ATTRS |
365 empty_elems = self._EMPTY_ELEMS | |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
366 noescape_elems = self._NOESCAPE_ELEMS |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
367 have_doctype = False |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
368 noescape = False |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
369 |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
370 stream = chain(self.preamble, stream) |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
371 for filter_ in self.filters: |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
372 stream = filter_(stream) |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
373 for kind, data, pos in stream: |
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
374 |
212
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
375 if kind is START or kind is EMPTY: |
96
35d681a94763
Add an XHTML serialization method. Now really need to get rid of some code duplication in the `markup.output` module.
cmlenz
parents:
89
diff
changeset
|
376 tag, attrib = data |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
377 buf = ['<', tag] |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
378 for attr, value in attrib: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
379 if attr in boolean_attrs: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
380 if value: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
381 buf += [' ', attr] |
524
49848aaa7839
Add special handling for `xml:lang` to HTML/XHTML serialization.
cmlenz
parents:
494
diff
changeset
|
382 elif ':' in attr: |
49848aaa7839
Add special handling for `xml:lang` to HTML/XHTML serialization.
cmlenz
parents:
494
diff
changeset
|
383 if attr == 'xml:lang' and u'lang' not in attrib: |
49848aaa7839
Add special handling for `xml:lang` to HTML/XHTML serialization.
cmlenz
parents:
494
diff
changeset
|
384 buf += [' lang="', escape(value), '"'] |
49848aaa7839
Add special handling for `xml:lang` to HTML/XHTML serialization.
cmlenz
parents:
494
diff
changeset
|
385 elif attr != 'xmlns': |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
386 buf += [' ', attr, '="', escape(value), '"'] |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
387 buf.append('>') |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
388 if kind is EMPTY: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
389 if tag not in empty_elems: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
390 buf.append('</%s>' % tag) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
391 yield Markup(u''.join(buf)) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
392 if tag in noescape_elems: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
393 noescape = True |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
394 |
69 | 395 elif kind is END: |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
396 yield Markup('</%s>' % data) |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
397 noescape = False |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
398 |
69 | 399 elif kind is TEXT: |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
400 if noescape: |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
401 yield data |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
402 else: |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
403 yield escape(data, quotes=False) |
1 | 404 |
89
d4c7617900e3
Support comments in templates that are not included in the output, in the same way Kid does: if the comment text starts with a `!` character, it is stripped from the output.
cmlenz
parents:
85
diff
changeset
|
405 elif kind is COMMENT: |
d4c7617900e3
Support comments in templates that are not included in the output, in the same way Kid does: if the comment text starts with a `!` character, it is stripped from the output.
cmlenz
parents:
85
diff
changeset
|
406 yield Markup('<!--%s-->' % data) |
d4c7617900e3
Support comments in templates that are not included in the output, in the same way Kid does: if the comment text starts with a `!` character, it is stripped from the output.
cmlenz
parents:
85
diff
changeset
|
407 |
136 | 408 elif kind is DOCTYPE and not have_doctype: |
409 name, pubid, sysid = data | |
410 buf = ['<!DOCTYPE %s'] | |
411 if pubid: | |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
412 buf.append(' PUBLIC "%s"') |
136 | 413 elif sysid: |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
414 buf.append(' SYSTEM') |
136 | 415 if sysid: |
397
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
416 buf.append(' "%s"') |
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
417 buf.append('>\n') |
d6e9170c5ccc
* Moved some utility functions from `genshi.core` to `genshi.util` (backwards compatibility preserved via imports)
cmlenz
parents:
346
diff
changeset
|
418 yield Markup(u''.join(buf), *filter(None, data)) |
136 | 419 have_doctype = True |
109
2de3f9d84a1c
Reorder the conditional branches in the serializers so that the more common event kinds are on top.
cmlenz
parents:
105
diff
changeset
|
420 |
105
334a338847af
Include processing instructions in serialized streams.
cmlenz
parents:
96
diff
changeset
|
421 elif kind is PI: |
334a338847af
Include processing instructions in serialized streams.
cmlenz
parents:
96
diff
changeset
|
422 yield Markup('<?%s %s?>' % data) |
334a338847af
Include processing instructions in serialized streams.
cmlenz
parents:
96
diff
changeset
|
423 |
1 | 424 |
200
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
425 class TextSerializer(object): |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
426 """Produces plain text from an event stream. |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
427 |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
428 Only text events are included in the output. Unlike the other serializer, |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
429 special XML characters are not escaped: |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
430 |
230 | 431 >>> from genshi.builder import tag |
200
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
432 >>> elem = tag.div(tag.a('<Hello!>', href='foo'), tag.br) |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
433 >>> print elem |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
434 <div><a href="foo"><Hello!></a><br/></div> |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
435 >>> print ''.join(TextSerializer()(elem.generate())) |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
436 <Hello!> |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
437 |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
438 If text events contain literal markup (instances of the `Markup` class), |
658
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
439 that markup is by default passed through unchanged: |
200
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
440 |
658
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
441 >>> elem = tag.div(Markup('<a href="foo">Hello & Bye!</a><br/>')) |
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
442 >>> print elem.generate().render(TextSerializer) |
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
443 <a href="foo">Hello & Bye!</a><br/> |
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
444 |
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
445 You can use the `strip_markup` to change this behavior, so that tags and |
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
446 entities are stripped from the output (or in the case of entities, |
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
447 replaced with the equivalent character): |
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
448 |
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
449 >>> print elem.generate().render(TextSerializer, strip_markup=True) |
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
450 Hello & Bye! |
200
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
451 """ |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
452 |
658
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
453 def __init__(self, strip_markup=False): |
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
454 self.strip_markup = strip_markup |
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
455 |
200
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
456 def __call__(self, stream): |
658
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
457 strip_markup = self.strip_markup |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
458 for event in stream: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
459 if event[0] is TEXT: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
460 data = event[1] |
658
a445c9e5ee96
The `TextSerializer` class no longer strips all markup in text by default, so that it is still possible to use the Genshi `escape` function even with text templates. The old behavior is available via the `strip_markup` option of the serializer. Closes #146.
cmlenz
parents:
524
diff
changeset
|
461 if strip_markup and type(data) is Markup: |
200
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
462 data = data.striptags().stripentities() |
201
0f16c907077e
The `TextSerializer` should produce `unicode` objects, not `Markup` objects.
cmlenz
parents:
200
diff
changeset
|
463 yield unicode(data) |
200
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
464 |
50eab0469148
Add serialization to plain text, based on cboos' patch. Closes #41.
cmlenz
parents:
178
diff
changeset
|
465 |
212
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
466 class EmptyTagFilter(object): |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
467 """Combines `START` and `STOP` events into `EMPTY` events for elements that |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
468 have no contents. |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
469 """ |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
470 |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
471 EMPTY = StreamEventKind('EMPTY') |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
472 |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
473 def __call__(self, stream): |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
474 prev = (None, None, None) |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
475 for ev in stream: |
212
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
476 if prev[0] is START: |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
477 if ev[0] is END: |
212
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
478 prev = EMPTY, prev[1], prev[2] |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
479 yield prev |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
480 continue |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
481 else: |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
482 yield prev |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
483 if ev[0] is not START: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
484 yield ev |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
485 prev = ev |
212
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
486 |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
487 |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
488 EMPTY = EmptyTagFilter.EMPTY |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
489 |
e8c43127d9a9
Refactored the handling of empty tags in the serializer: use an `EmptyTagFilter` that combines adjacent start/end events, instead of the generic pushback-iterator.
cmlenz
parents:
201
diff
changeset
|
490 |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
491 class NamespaceFlattener(object): |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
492 r"""Output stream filter that removes namespace information from the stream, |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
493 instead adding namespace attributes and prefixes as needed. |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
494 |
425
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
495 :param prefixes: optional mapping of namespace URIs to prefixes |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
496 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
497 >>> from genshi.input import XML |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
498 >>> xml = XML('''<doc xmlns="NS1" xmlns:two="NS2"> |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
499 ... <two:item/> |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
500 ... </doc>''') |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
501 >>> for kind, data, pos in NamespaceFlattener()(xml): |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
502 ... print kind, repr(data) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
503 START (u'doc', Attrs([(u'xmlns', u'NS1'), (u'xmlns:two', u'NS2')])) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
504 TEXT u'\n ' |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
505 START (u'two:item', Attrs()) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
506 END u'two:item' |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
507 TEXT u'\n' |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
508 END u'doc' |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
509 """ |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
510 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
511 def __init__(self, prefixes=None): |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
512 self.prefixes = {XML_NAMESPACE.uri: 'xml'} |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
513 if prefixes is not None: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
514 self.prefixes.update(prefixes) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
515 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
516 def __call__(self, stream): |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
517 prefixes = dict([(v, [k]) for k, v in self.prefixes.items()]) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
518 namespaces = {XML_NAMESPACE.uri: ['xml']} |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
519 def _push_ns(prefix, uri): |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
520 namespaces.setdefault(uri, []).append(prefix) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
521 prefixes.setdefault(prefix, []).append(uri) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
522 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
523 ns_attrs = [] |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
524 _push_ns_attr = ns_attrs.append |
437 | 525 def _make_ns_attr(prefix, uri): |
526 return u'xmlns%s' % (prefix and ':%s' % prefix or ''), uri | |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
527 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
528 def _gen_prefix(): |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
529 val = 0 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
530 while 1: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
531 val += 1 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
532 yield 'ns%d' % val |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
533 _gen_prefix = _gen_prefix().next |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
534 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
535 for kind, data, pos in stream: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
536 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
537 if kind is START or kind is EMPTY: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
538 tag, attrs = data |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
539 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
540 tagname = tag.localname |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
541 tagns = tag.namespace |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
542 if tagns: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
543 if tagns in namespaces: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
544 prefix = namespaces[tagns][-1] |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
545 if prefix: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
546 tagname = u'%s:%s' % (prefix, tagname) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
547 else: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
548 _push_ns_attr((u'xmlns', tagns)) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
549 _push_ns('', tagns) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
550 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
551 new_attrs = [] |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
552 for attr, value in attrs: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
553 attrname = attr.localname |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
554 attrns = attr.namespace |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
555 if attrns: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
556 if attrns not in namespaces: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
557 prefix = _gen_prefix() |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
558 _push_ns(prefix, attrns) |
412
29cddd600245
Actually write xmlns declaratons for generated attribute namespace prefixes.
cmlenz
parents:
410
diff
changeset
|
559 _push_ns_attr(('xmlns:%s' % prefix, attrns)) |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
560 else: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
561 prefix = namespaces[attrns][-1] |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
562 if prefix: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
563 attrname = u'%s:%s' % (prefix, attrname) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
564 new_attrs.append((attrname, value)) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
565 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
566 yield kind, (tagname, Attrs(ns_attrs + new_attrs)), pos |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
567 del ns_attrs[:] |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
568 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
569 elif kind is END: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
570 tagname = data.localname |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
571 tagns = data.namespace |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
572 if tagns: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
573 prefix = namespaces[tagns][-1] |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
574 if prefix: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
575 tagname = u'%s:%s' % (prefix, tagname) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
576 yield kind, tagname, pos |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
577 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
578 elif kind is START_NS: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
579 prefix, uri = data |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
580 if uri not in namespaces: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
581 prefix = prefixes.get(uri, [prefix])[-1] |
437 | 582 _push_ns_attr(_make_ns_attr(prefix, uri)) |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
583 _push_ns(prefix, uri) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
584 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
585 elif kind is END_NS: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
586 if data in prefixes: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
587 uris = prefixes.get(data) |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
588 uri = uris.pop() |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
589 if not uris: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
590 del prefixes[data] |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
591 if uri not in uris or uri != uris[-1]: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
592 uri_prefixes = namespaces[uri] |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
593 uri_prefixes.pop() |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
594 if not uri_prefixes: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
595 del namespaces[uri] |
437 | 596 if ns_attrs: |
597 attr = _make_ns_attr(data, uri) | |
598 if attr in ns_attrs: | |
599 ns_attrs.remove(attr) | |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
600 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
601 else: |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
602 yield kind, data, pos |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
603 |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
604 |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
605 class WhitespaceFilter(object): |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
606 """A filter that removes extraneous ignorable white space from the |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
607 stream. |
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
608 """ |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
609 |
305 | 610 def __init__(self, preserve=None, noescape=None): |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
611 """Initialize the filter. |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
612 |
425
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
613 :param preserve: a set or sequence of tag names for which white-space |
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
614 should be preserved |
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
615 :param noescape: a set or sequence of tag names for which text content |
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
616 should not be escaped |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
617 |
346
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
618 The `noescape` set is expected to refer to elements that cannot contain |
425
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
619 further child elements (such as ``<style>`` or ``<script>`` in HTML |
5b248708bbed
Try to use proper reStructuredText for docstrings throughout.
cmlenz
parents:
412
diff
changeset
|
620 documents). |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
621 """ |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
622 if preserve is None: |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
623 preserve = [] |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
624 self.preserve = frozenset(preserve) |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
625 if noescape is None: |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
626 noescape = [] |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
627 self.noescape = frozenset(noescape) |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
628 |
219 | 629 def __call__(self, stream, ctxt=None, space=XML_NAMESPACE['space'], |
630 trim_trailing_space=re.compile('[ \t]+(?=\n)').sub, | |
631 collapse_lines=re.compile('\n{2,}').sub): | |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
632 mjoin = Markup('').join |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
633 preserve_elems = self.preserve |
346
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
634 preserve = 0 |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
635 noescape_elems = self.noescape |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
636 noescape = False |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
637 |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
638 textbuf = [] |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
639 push_text = textbuf.append |
136 | 640 pop_text = textbuf.pop |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
641 for kind, data, pos in chain(stream, [(None, None, None)]): |
410
3460b04daeac
Improve the handling of namespaces in serialization.
cmlenz
parents:
408
diff
changeset
|
642 |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
643 if kind is TEXT: |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
644 if noescape: |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
645 data = Markup(data) |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
646 push_text(data) |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
647 else: |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
648 if textbuf: |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
649 if len(textbuf) > 1: |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
650 text = mjoin(textbuf, escape_quotes=False) |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
651 del textbuf[:] |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
652 else: |
136 | 653 text = escape(pop_text(), quotes=False) |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
654 if not preserve: |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
655 text = collapse_lines('\n', trim_trailing_space('', text)) |
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
656 yield TEXT, Markup(text), pos |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
657 |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
658 if kind is START: |
346
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
659 tag, attrs = data |
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
660 if preserve or (tag in preserve_elems or |
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
661 attrs.get(space) == 'preserve'): |
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
662 preserve += 1 |
219 | 663 if not noescape and tag in noescape_elems: |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
664 noescape = True |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
665 |
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
666 elif kind is END: |
346
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
667 noescape = False |
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
668 if preserve: |
2304e080ec07
Whitespace was not getting preserved in HTML `<pre>` elements that contained other HTML elements.
cmlenz
parents:
345
diff
changeset
|
669 preserve -= 1 |
141
b3ceaa35fb6b
* No escaping of `<script>` or `<style>` tags in HTML output (see #24)
cmlenz
parents:
140
diff
changeset
|
670 |
305 | 671 elif kind is START_CDATA: |
143
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
672 noescape = True |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
673 |
305 | 674 elif kind is END_CDATA: |
143
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
675 noescape = False |
ef761afcedff
CDATA sections in XML input now appear as CDATA sections in the output. This should address the problem with escaping the contents of `<style>` and `<script>` elements, which would only get interpreted correctly if the output was served as `application/xhtml+xml`. Closes #24.
cmlenz
parents:
141
diff
changeset
|
676 |
136 | 677 if kind: |
123
93bbdcf9428b
Fix for #18: whitespace in space-sensitive elements such as `<pre>` and `<textarea>` is now preserved.
cmlenz
parents:
109
diff
changeset
|
678 yield kind, data, pos |