django/tests/utils_tests/test_html.py

# -*- coding: utf-8 -*-
from __future__ import unicode_literals

from datetime import datetime
import os
from unittest import TestCase

from django.utils import html, safestring
from django.utils._os import upath
from django.utils.encoding import force_text


class TestUtilsHtml(TestCase):

    def check_output(self, function, value, output=None):
        """
        Check that function(value) equals output.  If output is None,
        check that function(value) equals value.
        """
        if output is None:
            output = value
        self.assertEqual(function(value), output)

    def test_escape(self):
        f = html.escape
        items = (
            ('&', '&amp;'),
            ('<', '&lt;'),
            ('>', '&gt;'),
            ('"', '&quot;'),
            ("'", '&#39;'),
        )
        # Substitution patterns for testing the above items.
        patterns = ("%s", "asdf%sfdsa", "%s1", "1%sb")
        for value, output in items:
            for pattern in patterns:
                self.check_output(f, pattern % value, pattern % output)
            # Check repeated values.
            self.check_output(f, value * 2, output * 2)
        # Verify it doesn't double replace &.
        self.check_output(f, '<&', '&lt;&amp;')

    def test_format_html(self):
        self.assertEqual(
            html.format_html("{0} {1} {third} {fourth}",
                             "< Dangerous >",
                             html.mark_safe("<b>safe</b>"),
                             third="< dangerous again",
                             fourth=html.mark_safe("<i>safe again</i>")
                             ),
            "&lt; Dangerous &gt; <b>safe</b> &lt; dangerous again <i>safe again</i>"
        )

    def test_linebreaks(self):
        f = html.linebreaks
        items = (
            ("para1\n\npara2\r\rpara3", "<p>para1</p>\n\n<p>para2</p>\n\n<p>para3</p>"),
            ("para1\nsub1\rsub2\n\npara2", "<p>para1<br />sub1<br />sub2</p>\n\n<p>para2</p>"),
            ("para1\r\n\r\npara2\rsub1\r\rpara4", "<p>para1</p>\n\n<p>para2<br />sub1</p>\n\n<p>para4</p>"),
            ("para1\tmore\n\npara2", "<p>para1\tmore</p>\n\n<p>para2</p>"),
        )
        for value, output in items:
            self.check_output(f, value, output)

    def test_strip_tags(self):
        f = html.strip_tags
        items = (
            ('<p>See: &#39;&eacute; is an apostrophe followed by e acute</p>',
             'See: &#39;&eacute; is an apostrophe followed by e acute'),
            ('<adf>a', 'a'),
            ('</adf>a', 'a'),
            ('<asdf><asdf>e', 'e'),
            ('hi, <f x', 'hi, <f x'),
            ('234<235, right?', '234<235, right?'),
            ('a4<a5 right?', 'a4<a5 right?'),
            ('b7>b2!', 'b7>b2!'),
            ('</fe', '</fe'),
            ('<x>b<y>', 'b'),
            ('a<p onclick="alert(\'<test>\')">b</p>c', 'abc'),
            ('a<p a >b</p>c', 'abc'),
            ('d<a:b c:d>e</p>f', 'def'),
            ('<strong>foo</strong><a href="http://example.com">bar</a>', 'foobar'),
        )
        for value, output in items:
            self.check_output(f, value, output)

        # Test with more lengthy content (also catching performance regressions)
        for filename in ('strip_tags1.html', 'strip_tags2.txt'):
            path = os.path.join(os.path.dirname(upath(__file__)), 'files', filename)
            with open(path, 'r') as fp:
                content = force_text(fp.read())
                start = datetime.now()
                stripped = html.strip_tags(content)
                elapsed = datetime.now() - start
            self.assertEqual(elapsed.seconds, 0)
            self.assertIn("Please try again.", stripped)
            self.assertNotIn('<', stripped)

    def test_strip_spaces_between_tags(self):
        f = html.strip_spaces_between_tags
        # Strings that should come out untouched.
        items = (' <adf>', '<adf> ', ' </adf> ', ' <f> x</f>')
        for value in items:
            self.check_output(f, value)
        # Strings that have spaces to strip.
        items = (
            ('<d> </d>', '<d></d>'),
            ('<p>hello </p>\n<p> world</p>', '<p>hello </p><p> world</p>'),
            ('\n<p>\t</p>\n<p> </p>\n', '\n<p></p><p></p>\n'),
        )
        for value, output in items:
            self.check_output(f, value, output)

    def test_strip_entities(self):
        f = html.strip_entities
        # Strings that should come out untouched.
        values = ("&", "&a", "&a", "a&#a")
        for value in values:
            self.check_output(f, value)
        # Valid entities that should be stripped from the patterns.
        entities = ("&#1;", "&#12;", "&a;", "&fdasdfasdfasdf;")
        patterns = (
            ("asdf %(entity)s ", "asdf  "),
            ("%(entity)s%(entity)s", ""),
            ("&%(entity)s%(entity)s", "&"),
            ("%(entity)s3", "3"),
        )
        for entity in entities:
            for in_pattern, output in patterns:
                self.check_output(f, in_pattern % {'entity': entity}, output)

    def test_escapejs(self):
        f = html.escapejs
        items = (
            ('"double quotes" and \'single quotes\'', '\\u0022double quotes\\u0022 and \\u0027single quotes\\u0027'),
            (r'\ : backslashes, too', '\\u005C : backslashes, too'),
            ('and lots of whitespace: \r\n\t\v\f\b', 'and lots of whitespace: \\u000D\\u000A\\u0009\\u000B\\u000C\\u0008'),
            (r'<script>and this</script>', '\\u003Cscript\\u003Eand this\\u003C/script\\u003E'),
            ('paragraph separator:\u2029and line separator:\u2028', 'paragraph separator:\\u2029and line separator:\\u2028'),
        )
        for value, output in items:
            self.check_output(f, value, output)

    def test_remove_tags(self):
        f = html.remove_tags
        items = (
            ("<b><i>Yes</i></b>", "b i", "Yes"),
            ("<a>x</a> <p><b>y</b></p>", "a b", "x <p>y</p>"),
        )
        for value, tags, output in items:
            self.assertEqual(f(value, tags), output)

    def test_smart_urlquote(self):
        quote = html.smart_urlquote
        # Ensure that IDNs are properly quoted
        self.assertEqual(quote('http://öäü.com/'), 'http://xn--4ca9at.com/')
        self.assertEqual(quote('http://öäü.com/öäü/'), 'http://xn--4ca9at.com/%C3%B6%C3%A4%C3%BC/')
        # Ensure that everything unsafe is quoted, !*'();:@&=+$,/?#[]~ is considered safe as per RFC
        self.assertEqual(quote('http://example.com/path/öäü/'), 'http://example.com/path/%C3%B6%C3%A4%C3%BC/')
        self.assertEqual(quote('http://example.com/%C3%B6/ä/'), 'http://example.com/%C3%B6/%C3%A4/')
        self.assertEqual(quote('http://example.com/?x=1&y=2'), 'http://example.com/?x=1&y=2')

    def test_conditional_escape(self):
        s = '<h1>interop</h1>'
        self.assertEqual(html.conditional_escape(s),
                         '&lt;h1&gt;interop&lt;/h1&gt;')
        self.assertEqual(html.conditional_escape(safestring.mark_safe(s)), s)
Simplified smart_urlquote and added some basic tests. 2013-07-28 16:05:39 +08:00			`# -- coding: utf-8 --`
Fixed #18269 -- Applied unicode_literals for Python 3 compatibility. Thanks Vinay Sajip for the support of his django3 branch and Jannis Leidel for the review. 2012-06-08 00:08:47 +08:00			`from __future__ import unicode_literals`

Added more tests for strip_tags utility Refs #19237. 2013-04-01 22:42:31 +08:00			`from datetime import datetime`
			`import os`
Stopped using django.utils.unittest in the test suite. Refs #20680. 2013-07-01 20:22:27 +08:00			`from unittest import TestCase`
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 23:15:04 +08:00
Fixed #7261 -- support for __html__ for library interoperability The idea is that if an object implements __html__ which returns a string this is used as HTML representation (eg: on escaping). If the object is a str or unicode subclass and returns itself the object is a safe string type. This is an updated patch based on jbalogh and ivank patches. 2013-10-15 06:40:52 +08:00			`from django.utils import html, safestring`
Added more tests for strip_tags utility Refs #19237. 2013-04-01 22:42:31 +08:00			`from django.utils._os import upath`
Fixed #19237 -- Used HTML parser to strip tags The regex method used until now for the strip_tags utility is fast, but subject to flaws and security issues. Consensus and good practice lead use to use a slower but safer method. 2013-05-22 23:29:16 +08:00			`from django.utils.encoding import force_text`
Added more tests for strip_tags utility Refs #19237. 2013-04-01 22:42:31 +08:00
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 23:15:04 +08:00
Imported unittest from django.utils in util_tests Without this, the 'new' assertion methods are not present with Python 2.6. 2013-04-02 01:58:16 +08:00			`class TestUtilsHtml(TestCase):`
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 23:15:04 +08:00
			`def check_output(self, function, value, output=None):`
			`"""`
			`Check that function(value) equals output. If output is None,`
			`check that function(value) equals value.`
			`"""`
			`if output is None:`
			`output = value`
			`self.assertEqual(function(value), output)`

			`def test_escape(self):`
			`f = html.escape`
			`items = (`
Fix all violators of E231 2013-10-27 03:15:03 +08:00			`('&', '&'),`
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 23:15:04 +08:00			`('<', '<'),`
			`('>', '>'),`
			`('"', '"'),`
			`("'", '''),`
			`)`
			`# Substitution patterns for testing the above items.`
			`patterns = ("%s", "asdf%sfdsa", "%s1", "1%sb")`
			`for value, output in items:`
			`for pattern in patterns:`
			`self.check_output(f, pattern % value, pattern % output)`
			`# Check repeated values.`
			`self.check_output(f, value * 2, output * 2)`
			`# Verify it doesn't double replace &.`
			`self.check_output(f, '<&', '<&')`

Added 'format_html' utility for formatting HTML fragments safely 2012-07-01 01:54:38 +08:00			`def test_format_html(self):`
			`self.assertEqual(`
Removed u prefixes on unicode strings. They break Python 3. 2012-07-20 18:29:22 +08:00			`html.format_html("{0} {1} {third} {fourth}",`
			`"< Dangerous >",`
			`html.mark_safe("<b>safe</b>"),`
Added 'format_html' utility for formatting HTML fragments safely 2012-07-01 01:54:38 +08:00			`third="< dangerous again",`
Removed u prefixes on unicode strings. They break Python 3. 2012-07-20 18:29:22 +08:00			`fourth=html.mark_safe("<i>safe again</i>")`
Added 'format_html' utility for formatting HTML fragments safely 2012-07-01 01:54:38 +08:00			`),`
Removed u prefixes on unicode strings. They break Python 3. 2012-07-20 18:29:22 +08:00			`"< Dangerous > <b>safe</b> < dangerous again <i>safe again</i>"`
Fixed #21287 -- Fixed E123 pep8 warnings 2013-10-18 17:02:43 +08:00			`)`
Added 'format_html' utility for formatting HTML fragments safely 2012-07-01 01:54:38 +08:00
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 23:15:04 +08:00			`def test_linebreaks(self):`
			`f = html.linebreaks`
			`items = (`
			`("para1\n\npara2\r\rpara3", "<p>para1</p>\n\n<p>para2</p>\n\n<p>para3</p>"),`
			`("para1\nsub1\rsub2\n\npara2", "<p>para1<br />sub1<br />sub2</p>\n\n<p>para2</p>"),`
			`("para1\r\n\r\npara2\rsub1\r\rpara4", "<p>para1</p>\n\n<p>para2<br />sub1</p>\n\n<p>para4</p>"),`
			`("para1\tmore\n\npara2", "<p>para1\tmore</p>\n\n<p>para2</p>"),`
			`)`
			`for value, output in items:`
			`self.check_output(f, value, output)`

			`def test_strip_tags(self):`
			`f = html.strip_tags`
			`items = (`
Fixed #19237 -- Used HTML parser to strip tags The regex method used until now for the strip_tags utility is fast, but subject to flaws and security issues. Consensus and good practice lead use to use a slower but safer method. 2013-05-22 23:29:16 +08:00			`('<p>See: 'é is an apostrophe followed by e acute</p>',`
			`'See: 'é is an apostrophe followed by e acute'),`
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 23:15:04 +08:00			`('<adf>a', 'a'),`
			`('</adf>a', 'a'),`
			`('<asdf><asdf>e', 'e'),`
Fixed #19237 -- Used HTML parser to strip tags The regex method used until now for the strip_tags utility is fast, but subject to flaws and security issues. Consensus and good practice lead use to use a slower but safer method. 2013-05-22 23:29:16 +08:00			`('hi, <f x', 'hi, <f x'),`
Fixed #19237 (again) - Made strip_tags consistent between Python versions 2013-05-23 20:00:17 +08:00			`('234<235, right?', '234<235, right?'),`
			`('a4<a5 right?', 'a4<a5 right?'),`
			`('b7>b2!', 'b7>b2!'),`
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 23:15:04 +08:00			`('</fe', '</fe'),`
			`('<x>b<y>', 'b'),`
Fixed #19237 -- Improved strip_tags utility The previous pattern didn't properly addressed cases where '>' was present inside quoted tag content. 2012-11-24 19:10:25 +08:00			`('a<p onclick="alert(\'<test>\')">b</p>c', 'abc'),`
			`('a<p a >b</p>c', 'abc'),`
			`('d<a:b c:d>e</p>f', 'def'),`
Improved regex in strip_tags Thanks Pablo Recio for the report. Refs #19237. 2013-02-07 04:20:43 +08:00			`('<strong>foo</strong><a href="http://example.com">bar</a>', 'foobar'),`
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 23:15:04 +08:00			`)`
			`for value, output in items:`
			`self.check_output(f, value, output)`

Added more tests for strip_tags utility Refs #19237. 2013-04-01 22:42:31 +08:00			`# Test with more lengthy content (also catching performance regressions)`
			`for filename in ('strip_tags1.html', 'strip_tags2.txt'):`
			`path = os.path.join(os.path.dirname(upath(__file__)), 'files', filename)`
			`with open(path, 'r') as fp:`
Fixed #19237 -- Used HTML parser to strip tags The regex method used until now for the strip_tags utility is fast, but subject to flaws and security issues. Consensus and good practice lead use to use a slower but safer method. 2013-05-22 23:29:16 +08:00			`content = force_text(fp.read())`
Added more tests for strip_tags utility Refs #19237. 2013-04-01 22:42:31 +08:00			`start = datetime.now()`
Fixed #19237 -- Used HTML parser to strip tags The regex method used until now for the strip_tags utility is fast, but subject to flaws and security issues. Consensus and good practice lead use to use a slower but safer method. 2013-05-22 23:29:16 +08:00			`stripped = html.strip_tags(content)`
Added more tests for strip_tags utility Refs #19237. 2013-04-01 22:42:31 +08:00			`elapsed = datetime.now() - start`
			`self.assertEqual(elapsed.seconds, 0)`
			`self.assertIn("Please try again.", stripped)`
			`self.assertNotIn('<', stripped)`

Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 23:15:04 +08:00			`def test_strip_spaces_between_tags(self):`
			`f = html.strip_spaces_between_tags`
			`# Strings that should come out untouched.`
			`items = (' <adf>', '<adf> ', ' </adf> ', ' <f> x</f>')`
			`for value in items:`
			`self.check_output(f, value)`
			`# Strings that have spaces to strip.`
			`items = (`
			`('<d> </d>', '<d></d>'),`
			`('<p>hello </p>\n<p> world</p>', '<p>hello </p><p> world</p>'),`
			`('\n<p>\t</p>\n<p> </p>\n', '\n<p></p><p></p>\n'),`
			`)`
			`for value, output in items:`
			`self.check_output(f, value, output)`

			`def test_strip_entities(self):`
			`f = html.strip_entities`
			`# Strings that should come out untouched.`
			`values = ("&", "&a", "&a", "a&#a")`
			`for value in values:`
			`self.check_output(f, value)`
			`# Valid entities that should be stripped from the patterns.`
			`entities = ("", "", "&a;", "&fdasdfasdfasdf;")`
			`patterns = (`
			`("asdf %(entity)s ", "asdf "),`
			`("%(entity)s%(entity)s", ""),`
			`("&%(entity)s%(entity)s", "&"),`
			`("%(entity)s3", "3"),`
			`)`
			`for entity in entities:`
			`for in_pattern, output in patterns:`
			`self.check_output(f, in_pattern % {'entity': entity}, output)`

Fixed #2986 -- Made the JavaScript code that drives related model instance addition in a popup window handle a model representation containing new lines. Also, moved the escapejs functionality yoo django.utils.html so it can be used from Python code. Thanks andrewwatts for the patch. git-svn-id: http://code.djangoproject.com/svn/django/trunk@15131 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2011-01-03 01:34:52 +08:00			`def test_escapejs(self):`
			`f = html.escapejs`
			`items = (`
Fixed #18269 -- Applied unicode_literals for Python 3 compatibility. Thanks Vinay Sajip for the support of his django3 branch and Jannis Leidel for the review. 2012-06-08 00:08:47 +08:00			`('"double quotes" and \'single quotes\'', '\\u0022double quotes\\u0022 and \\u0027single quotes\\u0027'),`
			`(r'\ : backslashes, too', '\\u005C : backslashes, too'),`
			`('and lots of whitespace: \r\n\t\v\f\b', 'and lots of whitespace: \\u000D\\u000A\\u0009\\u000B\\u000C\\u0008'),`
			`(r'<script>and this</script>', '\\u003Cscript\\u003Eand this\\u003C/script\\u003E'),`
			`('paragraph separator:\u2029and line separator:\u2028', 'paragraph separator:\\u2029and line separator:\\u2028'),`
Fixed #2986 -- Made the JavaScript code that drives related model instance addition in a popup window handle a model representation containing new lines. Also, moved the escapejs functionality yoo django.utils.html so it can be used from Python code. Thanks andrewwatts for the patch. git-svn-id: http://code.djangoproject.com/svn/django/trunk@15131 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2011-01-03 01:34:52 +08:00			`)`
			`for value, output in items:`
			`self.check_output(f, value, output)`
Fixed #7267 - UnicodeDecodeError in clean_html Thanks to Nikolay for the report, and gav and aaugustin for the patch. git-svn-id: http://code.djangoproject.com/svn/django/trunk@16118 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2011-04-28 22:08:53 +08:00
Fixed #14516 -- Extract methods from removetags and slugify template filters Patch by @jphalip updated to apply, documentation and release notes added. I've documented strip_tags as well as remove_tags as the difference between the two wouldn't be immediately obvious. 2012-08-18 20:53:22 +08:00			`def test_remove_tags(self):`
			`f = html.remove_tags`
			`items = (`
			`("<b><i>Yes</i></b>", "b i", "Yes"),`
			`("<a>x</a> <p><b>y</b></p>", "a b", "x <p>y</p>"),`
			`)`
			`for value, tags, output in items:`
Replaced a deprecated assertEquals 2012-09-24 22:07:58 +08:00			`self.assertEqual(f(value, tags), output)`
Simplified smart_urlquote and added some basic tests. 2013-07-28 16:05:39 +08:00
			`def test_smart_urlquote(self):`
			`quote = html.smart_urlquote`
			`# Ensure that IDNs are properly quoted`
			`self.assertEqual(quote('http://öäü.com/'), 'http://xn--4ca9at.com/')`
			`self.assertEqual(quote('http://öäü.com/öäü/'), 'http://xn--4ca9at.com/%C3%B6%C3%A4%C3%BC/')`
			`# Ensure that everything unsafe is quoted, !*'();:@&=+$,/?#[]~ is considered safe as per RFC`
			`self.assertEqual(quote('http://example.com/path/öäü/'), 'http://example.com/path/%C3%B6%C3%A4%C3%BC/')`
			`self.assertEqual(quote('http://example.com/%C3%B6/ä/'), 'http://example.com/%C3%B6/%C3%A4/')`
			`self.assertEqual(quote('http://example.com/?x=1&y=2'), 'http://example.com/?x=1&y=2')`
Fixed #7261 -- support for __html__ for library interoperability The idea is that if an object implements __html__ which returns a string this is used as HTML representation (eg: on escaping). If the object is a str or unicode subclass and returns itself the object is a safe string type. This is an updated patch based on jbalogh and ivank patches. 2013-10-15 06:40:52 +08:00
			`def test_conditional_escape(self):`
			`s = '<h1>interop</h1>'`
			`self.assertEqual(html.conditional_escape(s),`
			`'<h1>interop</h1>')`
			`self.assertEqual(html.conditional_escape(safestring.mark_safe(s)), s)`