django/tests/utils_tests/test_html.py

from __future__ import unicode_literals

from datetime import datetime
import os
from unittest import TestCase

from django.utils import html
from django.utils._os import upath
from django.utils.encoding import force_text


class TestUtilsHtml(TestCase):

    def check_output(self, function, value, output=None):
        """
        Check that function(value) equals output.  If output is None,
        check that function(value) equals value.
        """
        if output is None:
            output = value
        self.assertEqual(function(value), output)

    def test_escape(self):
        f = html.escape
        items = (
            ('&','&amp;'),
            ('<', '&lt;'),
            ('>', '&gt;'),
            ('"', '&quot;'),
            ("'", '&#39;'),
        )
        # Substitution patterns for testing the above items.
        patterns = ("%s", "asdf%sfdsa", "%s1", "1%sb")
        for value, output in items:
            for pattern in patterns:
                self.check_output(f, pattern % value, pattern % output)
            # Check repeated values.
            self.check_output(f, value * 2, output * 2)
        # Verify it doesn't double replace &.
        self.check_output(f, '<&', '&lt;&amp;')

    def test_format_html(self):
        self.assertEqual(
            html.format_html("{0} {1} {third} {fourth}",
                             "< Dangerous >",
                             html.mark_safe("<b>safe</b>"),
                             third="< dangerous again",
                             fourth=html.mark_safe("<i>safe again</i>")
                             ),
            "&lt; Dangerous &gt; <b>safe</b> &lt; dangerous again <i>safe again</i>"
            )

    def test_linebreaks(self):
        f = html.linebreaks
        items = (
            ("para1\n\npara2\r\rpara3", "<p>para1</p>\n\n<p>para2</p>\n\n<p>para3</p>"),
            ("para1\nsub1\rsub2\n\npara2", "<p>para1<br />sub1<br />sub2</p>\n\n<p>para2</p>"),
            ("para1\r\n\r\npara2\rsub1\r\rpara4", "<p>para1</p>\n\n<p>para2<br />sub1</p>\n\n<p>para4</p>"),
            ("para1\tmore\n\npara2", "<p>para1\tmore</p>\n\n<p>para2</p>"),
        )
        for value, output in items:
            self.check_output(f, value, output)

    def test_strip_tags(self):
        f = html.strip_tags
        items = (
            ('<p>See: &#39;&eacute; is an apostrophe followed by e acute</p>',
             'See: &#39;&eacute; is an apostrophe followed by e acute'),
            ('<adf>a', 'a'),
            ('</adf>a', 'a'),
            ('<asdf><asdf>e', 'e'),
            ('hi, <f x', 'hi, <f x'),
            ('234<235, right?', '234<235, right?'),
            ('a4<a5 right?', 'a4<a5 right?'),
            ('b7>b2!', 'b7>b2!'),
            ('</fe', '</fe'),
            ('<x>b<y>', 'b'),
            ('a<p onclick="alert(\'<test>\')">b</p>c', 'abc'),
            ('a<p a >b</p>c', 'abc'),
            ('d<a:b c:d>e</p>f', 'def'),
            ('<strong>foo</strong><a href="http://example.com">bar</a>', 'foobar'),
        )
        for value, output in items:
            self.check_output(f, value, output)

        # Test with more lengthy content (also catching performance regressions)
        for filename in ('strip_tags1.html', 'strip_tags2.txt'):
            path = os.path.join(os.path.dirname(upath(__file__)), 'files', filename)
            with open(path, 'r') as fp:
                content = force_text(fp.read())
                start = datetime.now()
                stripped = html.strip_tags(content)
                elapsed = datetime.now() - start
            self.assertEqual(elapsed.seconds, 0)
            self.assertIn("Please try again.", stripped)
            self.assertNotIn('<', stripped)

    def test_strip_spaces_between_tags(self):
        f = html.strip_spaces_between_tags
        # Strings that should come out untouched.
        items = (' <adf>', '<adf> ', ' </adf> ', ' <f> x</f>')
        for value in items:
            self.check_output(f, value)
        # Strings that have spaces to strip.
        items = (
            ('<d> </d>', '<d></d>'),
            ('<p>hello </p>\n<p> world</p>', '<p>hello </p><p> world</p>'),
            ('\n<p>\t</p>\n<p> </p>\n', '\n<p></p><p></p>\n'),
        )
        for value, output in items:
            self.check_output(f, value, output)

    def test_strip_entities(self):
        f = html.strip_entities
        # Strings that should come out untouched.
        values = ("&", "&a", "&a", "a&#a")
        for value in values:
            self.check_output(f, value)
        # Valid entities that should be stripped from the patterns.
        entities = ("&#1;", "&#12;", "&a;", "&fdasdfasdfasdf;")
        patterns = (
            ("asdf %(entity)s ", "asdf  "),
            ("%(entity)s%(entity)s", ""),
            ("&%(entity)s%(entity)s", "&"),
            ("%(entity)s3", "3"),
        )
        for entity in entities:
            for in_pattern, output in patterns:
                self.check_output(f, in_pattern % {'entity': entity}, output)

    def test_fix_ampersands(self):
        f = html.fix_ampersands
        # Strings without ampersands or with ampersands already encoded.
        values = ("a&#1;", "b", "&a;", "&amp; &x; ", "asdf")
        patterns = (
            ("%s", "%s"),
            ("&%s", "&amp;%s"),
            ("&%s&", "&amp;%s&amp;"),
        )
        for value in values:
            for in_pattern, out_pattern in patterns:
                self.check_output(f, in_pattern % value, out_pattern % value)
        # Strings with ampersands that need encoding.
        items = (
            ("&#;", "&amp;#;"),
            ("&#875 ;", "&amp;#875 ;"),
            ("&#4abc;", "&amp;#4abc;"),
        )
        for value, output in items:
            self.check_output(f, value, output)

    def test_escapejs(self):
        f = html.escapejs
        items = (
            ('"double quotes" and \'single quotes\'', '\\u0022double quotes\\u0022 and \\u0027single quotes\\u0027'),
            (r'\ : backslashes, too', '\\u005C : backslashes, too'),
            ('and lots of whitespace: \r\n\t\v\f\b', 'and lots of whitespace: \\u000D\\u000A\\u0009\\u000B\\u000C\\u0008'),
            (r'<script>and this</script>', '\\u003Cscript\\u003Eand this\\u003C/script\\u003E'),
            ('paragraph separator:\u2029and line separator:\u2028', 'paragraph separator:\\u2029and line separator:\\u2028'),
        )
        for value, output in items:
            self.check_output(f, value, output)

    def test_clean_html(self):
        f = html.clean_html
        items = (
            ('<p>I <i>believe</i> in <b>semantic markup</b>!</p>', '<p>I <em>believe</em> in <strong>semantic markup</strong>!</p>'),
            ('I escape & I don\'t <a href="#" target="_blank">target</a>', 'I escape &amp; I don\'t <a href="#" >target</a>'),
            ('<p>I kill whitespace</p><br clear="all"><p>&nbsp;</p>', '<p>I kill whitespace</p>'),
            # also a regression test for #7267: this used to raise an UnicodeDecodeError
            ('<p>* foo</p><p>* bar</p>', '<ul>\n<li> foo</li><li> bar</li>\n</ul>'),
        )
        for value, output in items:
            self.check_output(f, value, output)

    def test_remove_tags(self):
        f = html.remove_tags
        items = (
            ("<b><i>Yes</i></b>", "b i", "Yes"),
            ("<a>x</a> <p><b>y</b></p>", "a b", "x <p>y</p>"),
        )
        for value, tags, output in items:
            self.assertEqual(f(value, tags), output)
Fixed #18269 -- Applied unicode_literals for Python 3 compatibility. Thanks Vinay Sajip for the support of his django3 branch and Jannis Leidel for the review. 2012-06-07 16:08:47 +00:00			`from __future__ import unicode_literals`

Added more tests for strip_tags utility Refs #19237. 2013-04-01 14:42:31 +00:00			`from datetime import datetime`
			`import os`
Stopped using django.utils.unittest in the test suite. Refs #20680. 2013-07-01 12:22:27 +00:00			`from unittest import TestCase`
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 15:15:04 +00:00
			`from django.utils import html`
Added more tests for strip_tags utility Refs #19237. 2013-04-01 14:42:31 +00:00			`from django.utils._os import upath`
Fixed #19237 -- Used HTML parser to strip tags The regex method used until now for the strip_tags utility is fast, but subject to flaws and security issues. Consensus and good practice lead use to use a slower but safer method. 2013-05-22 15:29:16 +00:00			`from django.utils.encoding import force_text`
Added more tests for strip_tags utility Refs #19237. 2013-04-01 14:42:31 +00:00
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 15:15:04 +00:00
Imported unittest from django.utils in util_tests Without this, the 'new' assertion methods are not present with Python 2.6. 2013-04-01 17:58:16 +00:00			`class TestUtilsHtml(TestCase):`
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 15:15:04 +00:00
			`def check_output(self, function, value, output=None):`
			`"""`
			`Check that function(value) equals output. If output is None,`
			`check that function(value) equals value.`
			`"""`
			`if output is None:`
			`output = value`
			`self.assertEqual(function(value), output)`

			`def test_escape(self):`
			`f = html.escape`
			`items = (`
			`('&','&'),`
			`('<', '<'),`
			`('>', '>'),`
			`('"', '"'),`
			`("'", '''),`
			`)`
			`# Substitution patterns for testing the above items.`
			`patterns = ("%s", "asdf%sfdsa", "%s1", "1%sb")`
			`for value, output in items:`
			`for pattern in patterns:`
			`self.check_output(f, pattern % value, pattern % output)`
			`# Check repeated values.`
			`self.check_output(f, value * 2, output * 2)`
			`# Verify it doesn't double replace &.`
			`self.check_output(f, '<&', '<&')`

Added 'format_html' utility for formatting HTML fragments safely 2012-06-30 17:54:38 +00:00			`def test_format_html(self):`
			`self.assertEqual(`
Removed u prefixes on unicode strings. They break Python 3. 2012-07-20 10:29:22 +00:00			`html.format_html("{0} {1} {third} {fourth}",`
			`"< Dangerous >",`
			`html.mark_safe("<b>safe</b>"),`
Added 'format_html' utility for formatting HTML fragments safely 2012-06-30 17:54:38 +00:00			`third="< dangerous again",`
Removed u prefixes on unicode strings. They break Python 3. 2012-07-20 10:29:22 +00:00			`fourth=html.mark_safe("<i>safe again</i>")`
Added 'format_html' utility for formatting HTML fragments safely 2012-06-30 17:54:38 +00:00			`),`
Removed u prefixes on unicode strings. They break Python 3. 2012-07-20 10:29:22 +00:00			`"< Dangerous > <b>safe</b> < dangerous again <i>safe again</i>"`
Added 'format_html' utility for formatting HTML fragments safely 2012-06-30 17:54:38 +00:00			`)`

Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 15:15:04 +00:00			`def test_linebreaks(self):`
			`f = html.linebreaks`
			`items = (`
			`("para1\n\npara2\r\rpara3", "<p>para1</p>\n\n<p>para2</p>\n\n<p>para3</p>"),`
			`("para1\nsub1\rsub2\n\npara2", "<p>para1<br />sub1<br />sub2</p>\n\n<p>para2</p>"),`
			`("para1\r\n\r\npara2\rsub1\r\rpara4", "<p>para1</p>\n\n<p>para2<br />sub1</p>\n\n<p>para4</p>"),`
			`("para1\tmore\n\npara2", "<p>para1\tmore</p>\n\n<p>para2</p>"),`
			`)`
			`for value, output in items:`
			`self.check_output(f, value, output)`

			`def test_strip_tags(self):`
			`f = html.strip_tags`
			`items = (`
Fixed #19237 -- Used HTML parser to strip tags The regex method used until now for the strip_tags utility is fast, but subject to flaws and security issues. Consensus and good practice lead use to use a slower but safer method. 2013-05-22 15:29:16 +00:00			`('<p>See: 'é is an apostrophe followed by e acute</p>',`
			`'See: 'é is an apostrophe followed by e acute'),`
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 15:15:04 +00:00			`('<adf>a', 'a'),`
			`('</adf>a', 'a'),`
			`('<asdf><asdf>e', 'e'),`
Fixed #19237 -- Used HTML parser to strip tags The regex method used until now for the strip_tags utility is fast, but subject to flaws and security issues. Consensus and good practice lead use to use a slower but safer method. 2013-05-22 15:29:16 +00:00			`('hi, <f x', 'hi, <f x'),`
Fixed #19237 (again) - Made strip_tags consistent between Python versions 2013-05-23 12:00:17 +00:00			`('234<235, right?', '234<235, right?'),`
			`('a4<a5 right?', 'a4<a5 right?'),`
			`('b7>b2!', 'b7>b2!'),`
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 15:15:04 +00:00			`('</fe', '</fe'),`
			`('<x>b<y>', 'b'),`
Fixed #19237 -- Improved strip_tags utility The previous pattern didn't properly addressed cases where '>' was present inside quoted tag content. 2012-11-24 11:10:25 +00:00			`('a<p onclick="alert(\'<test>\')">b</p>c', 'abc'),`
			`('a<p a >b</p>c', 'abc'),`
			`('d<a:b c:d>e</p>f', 'def'),`
Improved regex in strip_tags Thanks Pablo Recio for the report. Refs #19237. 2013-02-06 20:20:43 +00:00			`('<strong>foo</strong><a href="http://example.com">bar</a>', 'foobar'),`
Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 15:15:04 +00:00			`)`
			`for value, output in items:`
			`self.check_output(f, value, output)`

Added more tests for strip_tags utility Refs #19237. 2013-04-01 14:42:31 +00:00			`# Test with more lengthy content (also catching performance regressions)`
			`for filename in ('strip_tags1.html', 'strip_tags2.txt'):`
			`path = os.path.join(os.path.dirname(upath(__file__)), 'files', filename)`
			`with open(path, 'r') as fp:`
Fixed #19237 -- Used HTML parser to strip tags The regex method used until now for the strip_tags utility is fast, but subject to flaws and security issues. Consensus and good practice lead use to use a slower but safer method. 2013-05-22 15:29:16 +00:00			`content = force_text(fp.read())`
Added more tests for strip_tags utility Refs #19237. 2013-04-01 14:42:31 +00:00			`start = datetime.now()`
Fixed #19237 -- Used HTML parser to strip tags The regex method used until now for the strip_tags utility is fast, but subject to flaws and security issues. Consensus and good practice lead use to use a slower but safer method. 2013-05-22 15:29:16 +00:00			`stripped = html.strip_tags(content)`
Added more tests for strip_tags utility Refs #19237. 2013-04-01 14:42:31 +00:00			`elapsed = datetime.now() - start`
			`self.assertEqual(elapsed.seconds, 0)`
			`self.assertIn("Please try again.", stripped)`
			`self.assertNotIn('<', stripped)`

Reorganized utils tests so it's all in separate modules. Thanks to Stephan Jaekel. git-svn-id: http://code.djangoproject.com/svn/django/trunk@13889 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2010-09-27 15:15:04 +00:00			`def test_strip_spaces_between_tags(self):`
			`f = html.strip_spaces_between_tags`
			`# Strings that should come out untouched.`
			`items = (' <adf>', '<adf> ', ' </adf> ', ' <f> x</f>')`
			`for value in items:`
			`self.check_output(f, value)`
			`# Strings that have spaces to strip.`
			`items = (`
			`('<d> </d>', '<d></d>'),`
			`('<p>hello </p>\n<p> world</p>', '<p>hello </p><p> world</p>'),`
			`('\n<p>\t</p>\n<p> </p>\n', '\n<p></p><p></p>\n'),`
			`)`
			`for value, output in items:`
			`self.check_output(f, value, output)`

			`def test_strip_entities(self):`
			`f = html.strip_entities`
			`# Strings that should come out untouched.`
			`values = ("&", "&a", "&a", "a&#a")`
			`for value in values:`
			`self.check_output(f, value)`
			`# Valid entities that should be stripped from the patterns.`
			`entities = ("", "", "&a;", "&fdasdfasdfasdf;")`
			`patterns = (`
			`("asdf %(entity)s ", "asdf "),`
			`("%(entity)s%(entity)s", ""),`
			`("&%(entity)s%(entity)s", "&"),`
			`("%(entity)s3", "3"),`
			`)`
			`for entity in entities:`
			`for in_pattern, output in patterns:`
			`self.check_output(f, in_pattern % {'entity': entity}, output)`

			`def test_fix_ampersands(self):`
			`f = html.fix_ampersands`
			`# Strings without ampersands or with ampersands already encoded.`
			`values = ("a", "b", "&a;", "& &x; ", "asdf")`
			`patterns = (`
			`("%s", "%s"),`
			`("&%s", "&%s"),`
			`("&%s&", "&%s&"),`
			`)`
			`for value in values:`
			`for in_pattern, out_pattern in patterns:`
			`self.check_output(f, in_pattern % value, out_pattern % value)`
			`# Strings with ampersands that need encoding.`
			`items = (`
			`("&#;", "&#;"),`
			`("&#875 ;", "&#875 ;"),`
			`("&#4abc;", "&#4abc;"),`
			`)`
			`for value, output in items:`
			`self.check_output(f, value, output)`
Fixed #2986 -- Made the JavaScript code that drives related model instance addition in a popup window handle a model representation containing new lines. Also, moved the escapejs functionality yoo django.utils.html so it can be used from Python code. Thanks andrewwatts for the patch. git-svn-id: http://code.djangoproject.com/svn/django/trunk@15131 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2011-01-02 17:34:52 +00:00
			`def test_escapejs(self):`
			`f = html.escapejs`
			`items = (`
Fixed #18269 -- Applied unicode_literals for Python 3 compatibility. Thanks Vinay Sajip for the support of his django3 branch and Jannis Leidel for the review. 2012-06-07 16:08:47 +00:00			`('"double quotes" and \'single quotes\'', '\\u0022double quotes\\u0022 and \\u0027single quotes\\u0027'),`
			`(r'\ : backslashes, too', '\\u005C : backslashes, too'),`
			`('and lots of whitespace: \r\n\t\v\f\b', 'and lots of whitespace: \\u000D\\u000A\\u0009\\u000B\\u000C\\u0008'),`
			`(r'<script>and this</script>', '\\u003Cscript\\u003Eand this\\u003C/script\\u003E'),`
			`('paragraph separator:\u2029and line separator:\u2028', 'paragraph separator:\\u2029and line separator:\\u2028'),`
Fixed #2986 -- Made the JavaScript code that drives related model instance addition in a popup window handle a model representation containing new lines. Also, moved the escapejs functionality yoo django.utils.html so it can be used from Python code. Thanks andrewwatts for the patch. git-svn-id: http://code.djangoproject.com/svn/django/trunk@15131 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2011-01-02 17:34:52 +00:00			`)`
			`for value, output in items:`
			`self.check_output(f, value, output)`
Fixed #7267 - UnicodeDecodeError in clean_html Thanks to Nikolay for the report, and gav and aaugustin for the patch. git-svn-id: http://code.djangoproject.com/svn/django/trunk@16118 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2011-04-28 14:08:53 +00:00
			`def test_clean_html(self):`
			`f = html.clean_html`
			`items = (`
Fixed #18269 -- Applied unicode_literals for Python 3 compatibility. Thanks Vinay Sajip for the support of his django3 branch and Jannis Leidel for the review. 2012-06-07 16:08:47 +00:00			`('<p>I <i>believe</i> in <b>semantic markup</b>!</p>', '<p>I <em>believe</em> in <strong>semantic markup</strong>!</p>'),`
			`('I escape & I don\'t <a href="#" target="_blank">target</a>', 'I escape & I don\'t <a href="#" >target</a>'),`
			`('<p>I kill whitespace</p><br clear="all"><p> </p>', '<p>I kill whitespace</p>'),`
Fixed #7267 - UnicodeDecodeError in clean_html Thanks to Nikolay for the report, and gav and aaugustin for the patch. git-svn-id: http://code.djangoproject.com/svn/django/trunk@16118 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2011-04-28 14:08:53 +00:00			`# also a regression test for #7267: this used to raise an UnicodeDecodeError`
Fixed #18269 -- Applied unicode_literals for Python 3 compatibility. Thanks Vinay Sajip for the support of his django3 branch and Jannis Leidel for the review. 2012-06-07 16:08:47 +00:00			`('<p>* foo</p><p>* bar</p>', '<ul>\n<li> foo</li><li> bar</li>\n</ul>'),`
Fixed #7267 - UnicodeDecodeError in clean_html Thanks to Nikolay for the report, and gav and aaugustin for the patch. git-svn-id: http://code.djangoproject.com/svn/django/trunk@16118 bcc190cf-cafb-0310-a4f2-bffc1f526a37 2011-04-28 14:08:53 +00:00			`)`
			`for value, output in items:`
			`self.check_output(f, value, output)`
Fixed #14516 -- Extract methods from removetags and slugify template filters Patch by @jphalip updated to apply, documentation and release notes added. I've documented strip_tags as well as remove_tags as the difference between the two wouldn't be immediately obvious. 2012-08-18 12:53:22 +00:00
			`def test_remove_tags(self):`
			`f = html.remove_tags`
			`items = (`
			`("<b><i>Yes</i></b>", "b i", "Yes"),`
			`("<a>x</a> <p><b>y</b></p>", "a b", "x <p>y</p>"),`
			`)`
			`for value, tags, output in items:`
Replaced a deprecated assertEquals 2012-09-24 14:07:58 +00:00			`self.assertEqual(f(value, tags), output)`