""" tests for encutils.py """ import http.client from io import StringIO try: import cssutils.encutils as encutils except ImportError: import encutils # helper log log = encutils.buildlog(stream=StringIO()) class TestAutoEncoding: def _fakeRes(self, content): "build a fake HTTP response" class FakeRes: def __init__(self, content): self._info = http.client.HTTPMessage() # Adjust to testdata. items = content.split(':') if len(items) > 1: # Get the type by just # using the data at the end. t = items[-1].strip() self._info.set_type(t) def info(self): return self._info def read(self): return content return FakeRes(content) def test_getTextTypeByMediaType(self): "encutils._getTextTypeByMediaType" tests = { 'application/xml': encutils._XML_APPLICATION_TYPE, 'application/xml-dtd': encutils._XML_APPLICATION_TYPE, 'application/xml-external-parsed-entity': encutils._XML_APPLICATION_TYPE, 'application/xhtml+xml': encutils._XML_APPLICATION_TYPE, 'text/xml': encutils._XML_TEXT_TYPE, 'text/xml-external-parsed-entity': encutils._XML_TEXT_TYPE, 'text/xhtml+xml': encutils._XML_TEXT_TYPE, 'text/html': encutils._HTML_TEXT_TYPE, 'text/css': encutils._TEXT_UTF8, 'text/plain': encutils._TEXT_TYPE, 'x/x': encutils._OTHER_TYPE, 'ANYTHING': encutils._OTHER_TYPE, } for test, exp in list(tests.items()): assert exp == encutils._getTextTypeByMediaType(test, log=log) def test_getTextType(self): "encutils._getTextType" tests = { '\x00\x00\xFE\xFF""": ( None, None, ), """""": ( None, None, ), """""": ( 'text/html', None, ), """""": ( 'text/html', None, ), """""": ( 'text/html', 'ascii', ), """""": ( 'text/html', 'ascii', ), """""": ( 'text/html', 'iso-8859-1', ), """""": ( 'text/html', 'ascii', ), """""": ( 'text/html', 'ascii', ), """""": ( 'text/html', 'ascii', ), """""": ( 'text/html', 'ascii', ), """""": ( 'text/html', 'ascii', ), """""": ( 'text/html', 'ascii', ), """""": ( 'text/html', 'ascii', ), """""": ( 'text/html', 'ascii', ), """""": ( 'text/html', 'ascii', ), """""": ( 'text/html', 'ascii', ), """raises exception: """: (None, None), """ """: ( 'text/html', 'ascii', ), """ """: ( 'text/html', 'ascii', ), # py 2.7.3 fixed HTMLParser so: (None, None) """ """: ( 'text/html', None, ), } for test, exp in list(tests.items()): assert exp == encutils.getMetaInfo(test, log=log) def test_detectXMLEncoding(self): "encutils.detectXMLEncoding" tests = ( # BOM (('utf_32_be'), '\x00\x00\xFE\xFFanything'), (('utf_32_le'), '\xFF\xFE\x00\x00anything'), (('utf_16_be'), '\xFE\xFFanything'), (('utf_16_le'), '\xFF\xFEanything'), (('utf-8'), '\xef\xbb\xbfanything'), # encoding= (('ascii'), ''), (('ascii'), ""), (('iso-8859-1'), ""), # default (('utf-8'), ''), (('utf-8'), ''), ) for exp, test in tests: assert exp == encutils.detectXMLEncoding(test, log=log) def test_tryEncodings(self): "encutils.tryEncodings" try: tests = [ ('ascii', b'abc'), ('windows-1252', '€'.encode('windows-1252')), ('ascii', b'1'), ] except ImportError: tests = [ ('ascii', b'abc'), ('windows-1252', '€'.encode('windows-1252')), ('iso-8859-1', 'äöüß'.encode('iso-8859-1')), ('iso-8859-1', 'äöüß'.encode('windows-1252')), # ('utf-8', u'\u1111'.encode('utf-8')) ] for exp, test in tests: assert exp == encutils.tryEncodings(test).lower() def test_getEncodingInfo(self): "encutils.getEncodingInfo" # (expectedencoding, expectedmismatch): (httpheader, filecontent) tests = [ # --- application/xhtml+xml --- # header default and XML default ( ('utf-8', False), ( '''Content-Type: application/xhtml+xml''', ''' ''', ), ), # XML default ( ('utf-8', False), ( None, ''' ''', ), ), # meta is ignored! ( ('utf-8', False), ( '''Content-Type: application/xhtml+xml''', ''' ''', ), ), # header enc and XML default ( ('iso-h', True), ( '''Content-Type: application/xhtml+xml;charset=iso-H''', ''' ''', ), ), # mismatch header and XML explicit, header wins ( ('iso-h', True), ( '''Content-Type: application/xhtml+xml;charset=iso-H''', ''' ''', ), ), # header == XML, meta ignored! ( ('iso-h', False), ( '''Content-Type: application/xhtml+xml;charset=iso-H''', ''' ''', ), ), # XML only, meta ignored! ( ('iso-x', False), ( '''Content-Type: application/xhtml+xml''', ''' ''', ), ), # no text or not enough text: (('iso-h', False), ('Content-Type: application/xml;charset=iso-h', '1')), (('utf-8', False), ('Content-Type: application/xml', None)), ((None, False), ('Content-Type: application/xml', '1')), # --- text/xml --- # default enc ( ('ascii', False), ( '''Content-Type: text/xml''', ''' ''', ), ), # default as XML ignored and meta completely ignored ( ('ascii', False), ( '''Content-Type: text/xml''', ''' ''', ), ), (('ascii', False), ('Content-Type: text/xml', '1')), (('ascii', False), ('Content-Type: text/xml', None)), # header enc ( ('iso-h', False), ( '''Content-Type: text/xml;charset=iso-H''', ''' ''', ), ), # header only, XML and meta ignored! ( ('iso-h', False), ( '''Content-Type: text/xml;charset=iso-H''', ''' ''', ), ), ( ('iso-h', False), ( '''Content-Type: text/xml;charset=iso-H''', ''' ''', ), ), # --- text/html --- # default enc ( ('iso-8859-1', False), ( 'Content-Type: text/html;', '''''', ), ), (('iso-8859-1', False), ('Content-Type: text/html;', None)), # header enc ( ('iso-h', False), ( 'Content-Type: text/html;charset=iso-H', '''''', ), ), # meta enc ( ('iso-m', False), ( 'Content-Type: text/html', '''''', ), ), # mismatch header and meta, header wins ( ('iso-h', True), ( 'Content-Type: text/html;charset=iso-H', '''''', ), ), # no header: ( (None, False), ( None, '''''', ), ), # no encoding at all ( (None, False), ( None, '''''', ), ), ((None, False), (None, '''text''')), # --- no header --- ((None, False), (None, '')), (('iso-8859-1', False), ('''NoContentType''', '''OnlyText''')), (('iso-8859-1', False), ('Content-Type: text/html;', None)), (('iso-8859-1', False), ('Content-Type: text/html;', '1')), # XML (('utf-8', False), (None, '''''')), # meta ignored ( ('utf-8', False), ( None, ''' ''', ), ), (('utf-8', False), ('Content-Type: text/css;', '1')), (('iso-h', False), ('Content-Type: text/css;charset=iso-h', '1')), # only header is used by encutils (('utf-8', False), ('Content-Type: text/css', '@charset "ascii";')), ] for exp, test in tests: header, text = test if header: res = encutils.getEncodingInfo(self._fakeRes(header), text) else: res = encutils.getEncodingInfo(text=text) res = (res.encoding, res.mismatch) assert exp == res