diff --git a/glymur/_uuid_io/Exif.py b/glymur/_uuid_io/Exif.py index f70a090..3fb24b1 100644 --- a/glymur/_uuid_io/Exif.py +++ b/glymur/_uuid_io/Exif.py @@ -1,7 +1,8 @@ # -*- coding: utf-8 -*- """ -Handlers for various UUID types. +Handlers for Exif UUIDs. Be nice if we would find a standard for this. """ +import pprint import struct import sys import warnings @@ -12,7 +13,7 @@ if sys.hexversion < 0x02070000: else: from collections import OrderedDict -class _Exif(object): +class UUIDExif(object): """ Attributes ---------- @@ -25,10 +26,10 @@ class _Exif(object): def __init__(self, read_buffer): """Interpret raw buffer consisting of Exif IFD. """ - self.exif_image = None - self.exif_photo = None - self.exif_gpsinfo = None - self.exif_iop = None + exif_image = None + exif_photo = None + exif_gpsinfo = None + exif_iop = None self.read_buffer = read_buffer @@ -45,24 +46,39 @@ class _Exif(object): # This is the 'Exif Image' portion. exif = _ExifImageIfd(self.endian, read_buffer[6:], offset) - self.exif_image = exif.processed_ifd + exif_image = exif.processed_ifd - if 'ExifTag' in self.exif_image.keys(): - offset = self.exif_image['ExifTag'] - photo = _ExifPhotoIfd(self.endian, read_buffer[6:], offset) - self.exif_photo = photo.processed_ifd + if 'ExifTag' in exif_image.keys(): + offset = exif_image['ExifTag'] + photo_ifd = _ExifPhotoIfd(self.endian, read_buffer[6:], offset) + exif_photo = photo_ifd.processed_ifd - if 'InteroperabilityTag' in self.exif_photo.keys(): - offset = self.exif_photo['InteroperabilityTag'] + if 'InteroperabilityTag' in exif_photo.keys(): + offset = exif_photo['InteroperabilityTag'] interop = _ExifInteroperabilityIfd(self.endian, read_buffer[6:], offset) - self.iop = interop.processed_ifd + iop = interop.processed_ifd - if 'GPSTag' in self.exif_image.keys(): - offset = self.exif_image['GPSTag'] + if 'GPSTag' in exif_image.keys(): + offset = exif_image['GPSTag'] gps = _ExifGPSInfoIfd(self.endian, read_buffer[6:], offset) - self.exif_gpsinfo = gps.processed_ifd + exif_gpsinfo = gps.processed_ifd + + self.ifds = OrderedDict() + self.ifds['Image'] = exif_image + self.ifds['Photo'] = exif_photo + self.ifds['GPSInfo'] = exif_gpsinfo + self.ifds['Iop'] = exif_iop + + def __str__(self): + # 2.7 has trouble pretty-printing ordered dicts, so print them + # as regular dicts. Not ideal, but at least it's good on 3.3+. + if sys.hexversion < 0x03000000: + data = dict(self.ifds) + else: + data = self.ifds + return '\n' + pprint.pformat(data) class _Ifd(object): diff --git a/glymur/_uuid_io/XMP.py b/glymur/_uuid_io/XMP.py new file mode 100644 index 0000000..0451a92 --- /dev/null +++ b/glymur/_uuid_io/XMP.py @@ -0,0 +1,46 @@ +# -*- coding: utf-8 -*- + +""" +Handler for a UUID for XMP. +""" + +import sys +from xml.etree import cElementTree as ET + +from ..core import _pretty_print_xml + +class UUIDXMP(object): + """ + Handler for a UUID for XMP. + + Attributes + ---------- + packet : ElementTree + XML conforming to the XMP specifications. + + References + ---------- + .. [XMP] International Organization for Standardication. ISO/IEC + 16684-1:2012 - Graphic technology -- Extensible metadata platform (XMP) + specification -- Part 1: Data model, serialization and core properties + """ + def __init__(self, read_buffer): + """ + Parameters + ---------- + read_buffer : byte array + sequence of bytes that can be decoded into an XMP packet. + """ + + # XMP data. Parse as XML. + if sys.hexversion < 0x03000000: + # 2.x strings same as bytes + elt = ET.fromstring(read_buffer) + else: + # 3.x takes strings, not bytes. + text = read_buffer.decode('utf-8') + elt = ET.fromstring(text) + self.packet = ET.ElementTree(elt) + + def __str__(self): + return _pretty_print_xml(self.packet) diff --git a/glymur/_uuid_io/__init__.py b/glymur/_uuid_io/__init__.py index 721ee36..5545351 100644 --- a/glymur/_uuid_io/__init__.py +++ b/glymur/_uuid_io/__init__.py @@ -1 +1,4 @@ -from .Exif import _Exif +from .Exif import UUIDExif +from .XMP import UUIDXMP +from .generic import UUIDGeneric + diff --git a/glymur/_uuid_io/generic.py b/glymur/_uuid_io/generic.py new file mode 100644 index 0000000..bad68a2 --- /dev/null +++ b/glymur/_uuid_io/generic.py @@ -0,0 +1,27 @@ +# -*- coding: utf-8 -*- + +""" +Handler for a generic UUID. +""" + +class UUIDGeneric(object): + """ + Handler for a generic UUID that is not currently recognized. + + Attributes + ---------- + data : byte array + Sequence of uninterpreted bytes as read from the file. + """ + def __init__(self, read_buffer): + """ + Parameters + ---------- + read_buffer : byte array + sequence of bytes as read from the file. + """ + self.data = read_buffer + + def __str__(self): + return '{0} bytes'.format(len(self.data)) + diff --git a/glymur/core.py b/glymur/core.py index 22b5a19..620d0e3 100644 --- a/glymur/core.py +++ b/glymur/core.py @@ -1,5 +1,8 @@ """Core definitions to be shared amongst the modules. """ +import copy +import xml.etree.cElementTree as ET + # Progression order LRCP = 0 RLCP = 1 @@ -73,3 +76,45 @@ _CAPABILITIES_DISPLAY = { 1: '0', 2: '1', 3: '3'} + + +def _pretty_print_xml(xml, level=0): + """Pretty print XML data. + """ + xml = copy.deepcopy(xml) + _indent(xml.getroot(), level=level) + xmltext = ET.tostring(xml.getroot(), encoding='utf-8').decode('utf-8') + + # Indent it a bit. + lst = [(' ' + x) for x in xmltext.split('\n')] + try: + xml = '\n'.join(lst) + return '\n{0}'.format(xml) + except UnicodeEncodeError: + # This can happen on python 2.x if the character set contains certain + # non-ascii characters. Just print out the corresponding xml char + # entities instead. + xml = u'\n'.join(lst) + text = u'\n{0}'.format(xml) + text = text.encode('ascii', 'xmlcharrefreplace') + return text + + +def _indent(elem, level=0): + """Recipe for pretty printing XML. Please see + + http://effbot.org/zone/element-lib.htm#prettyprint + """ + i = "\n" + level * " " + if len(elem): + if not elem.text or not elem.text.strip(): + elem.text = i + " " + if not elem.tail or not elem.tail.strip(): + elem.tail = i + for elem in elem: + _indent(elem, level + 1) + if not elem.tail or not elem.tail.strip(): + elem.tail = i + else: + if level and (not elem.tail or not elem.tail.strip()): + elem.tail = i diff --git a/glymur/jp2box.py b/glymur/jp2box.py index 3ea911e..681cf51 100644 --- a/glymur/jp2box.py +++ b/glymur/jp2box.py @@ -13,7 +13,6 @@ References # pylint: disable=C0302,R0903,R0913 -import copy import datetime import math import os @@ -38,8 +37,9 @@ from .core import _COLORSPACE_MAP_DISPLAY from .core import _COLOR_TYPE_MAP_DISPLAY from .core import ENUMERATED_COLORSPACE, RESTRICTED_ICC_PROFILE from .core import ANY_ICC_PROFILE, VENDOR_COLOR_METHOD +from .core import _pretty_print_xml -from ._uuid_io import _Exif +from . import _uuid_io _METHOD_DISPLAY = { ENUMERATED_COLORSPACE: 'enumerated colorspace', @@ -2065,8 +2065,11 @@ class UUIDBox(Jp2kBox): ---------- the_uuid : uuid.UUID Identifies the type of UUID box. + data : object + Specific to each type of UUID. There are handlers for XMP, Exif, + and unknown UUIDs. raw_data : byte array - This is the "payload" of data for the specified UUID. + Sequence of uninterpreted bytes as read from the file. length : int length of the box in bytes. offset : int @@ -2076,59 +2079,33 @@ class UUIDBox(Jp2kBox): self.uuid = the_uuid if the_uuid == uuid.UUID('be7acfcb-97a9-42e8-9c71-999491e3afac'): - # XMP data. Parse as XML. Seems to be a difference between - # ElementTree in version 2.7 and 3.3. - if sys.hexversion < 0x03000000: - elt = ET.fromstring(raw_data) - else: - text = raw_data.decode('utf-8') - elt = ET.fromstring(text) - self.data = ET.ElementTree(elt) + self.data = _uuid_io.UUIDXMP(raw_data) self._type = 'XMP' elif the_uuid.bytes == b'JpgTiffExif->JP2': - exif_obj = _Exif(raw_data) - ifds = OrderedDict() - ifds['Image'] = exif_obj.exif_image - ifds['Photo'] = exif_obj.exif_photo - ifds['GPSInfo'] = exif_obj.exif_gpsinfo - ifds['Iop'] = exif_obj.exif_iop - self.data = ifds + self.data = _uuid_io.UUIDExif(raw_data) self._type = 'Exif' else: - self.data = raw_data + self.data = _uuid_io.UUIDGeneric(raw_data) self._type = 'unknown' + + self.raw_data = raw_data self.length = length self.offset = offset def __str__(self): msg = '{0}\n' - msg += ' UUID: {1}{2}\n' + msg += ' UUID: {1} ({2})\n' msg += ' UUID Data: {3}' - if self.uuid == uuid.UUID('be7acfcb-97a9-42e8-9c71-999491e3afac'): - uuid_type = ' (XMP)' - uuid_data = _pretty_print_xml(self.data) - elif self.uuid.bytes == b'JpgTiffExif->JP2': - uuid_type = ' (Exif)' - # 2.7 has trouble pretty-printing ordered dicts, so print them - # as regular dicts. Not ideal, but at least it's good on 3.3+. - if sys.hexversion < 0x03000000: - data = dict(self.data) - else: - data = self.data - uuid_data = '\n' + pprint.pformat(data) - else: - uuid_type = '' - uuid_data = '{0} bytes'.format(len(self.data)) - msg = msg.format(Jp2kBox.__str__(self), self.uuid, - uuid_type, - uuid_data) + self._type, + str(self.data)) return msg + def write(self, fptr): """Write a UUID box box to file. """ @@ -2196,45 +2173,3 @@ _BOX_WITH_ID = { 'url ': DataEntryURLBox, 'uuid': UUIDBox, 'xml ': XMLBox} - - -def _indent(elem, level=0): - """Recipe for pretty printing XML. Please see - - http://effbot.org/zone/element-lib.htm#prettyprint - """ - i = "\n" + level * " " - if len(elem): - if not elem.text or not elem.text.strip(): - elem.text = i + " " - if not elem.tail or not elem.tail.strip(): - elem.tail = i - for elem in elem: - _indent(elem, level + 1) - if not elem.tail or not elem.tail.strip(): - elem.tail = i - else: - if level and (not elem.tail or not elem.tail.strip()): - elem.tail = i - - -def _pretty_print_xml(xml, level=0): - """Pretty print XML data. - """ - xml = copy.deepcopy(xml) - _indent(xml.getroot(), level=level) - xmltext = ET.tostring(xml.getroot(), encoding='utf-8').decode('utf-8') - - # Indent it a bit. - lst = [(' ' + x) for x in xmltext.split('\n')] - try: - xml = '\n'.join(lst) - return '\n{0}'.format(xml) - except UnicodeEncodeError: - # This can happen on python 2.x if the character set contains certain - # non-ascii characters. Just print out the corresponding xml char - # entities instead. - xml = u'\n'.join(lst) - text = u'\n{0}'.format(xml) - text = text.encode('ascii', 'xmlcharrefreplace') - return text diff --git a/glymur/test/test_jp2k.py b/glymur/test/test_jp2k.py index c9e750a..026b129 100644 --- a/glymur/test/test_jp2k.py +++ b/glymur/test/test_jp2k.py @@ -353,7 +353,7 @@ class TestJp2k(unittest.TestCase): def test_xmp_attribute(self): """Verify the XMP packet in the shipping example file can be read.""" j = Jp2k(self.jp2file) - xmp = j.box[3].data + xmp = j.box[3].data.packet ns0 = '{http://www.w3.org/1999/02/22-rdf-syntax-ns#}' ns2 = '{http://ns.adobe.com/xap/1.0/}' name = '{0}RDF/{0}Description/{1}CreatorTool'.format(ns0, ns2) diff --git a/glymur/test/test_printing.py b/glymur/test/test_printing.py index 5824e0e..a1f0c55 100644 --- a/glymur/test/test_printing.py +++ b/glymur/test/test_printing.py @@ -636,6 +636,7 @@ class TestPrinting(unittest.TestCase): actual = fake_out.getvalue().strip() expected = nemo_xmp_box + self.maxDiff = None self.assertEqual(actual, expected) def test_codestream(self): @@ -1024,7 +1025,7 @@ class TestPrinting(unittest.TestCase): print(jp2.box[4]) actual = fake_out.getvalue().strip() lines = ['UUID Box (uuid) @ (1544, 25)', - ' UUID: 3a0d0218-0ae9-4115-b376-4bca41ce0e71', + ' UUID: 3a0d0218-0ae9-4115-b376-4bca41ce0e71 (unknown)', ' UUID Data: 1 bytes'] expected = '\n'.join(lines)