Refactored UUID handling. #104

New classes for each type in _uuid_io sub package.
This commit is contained in:
jevans 2013-10-26 16:51:07 -04:00
commit 3474e49fce
8 changed files with 173 additions and 100 deletions

View file

@ -1,7 +1,8 @@
# -*- coding: utf-8 -*-
"""
Handlers for various UUID types.
Handlers for Exif UUIDs. Be nice if we would find a standard for this.
"""
import pprint
import struct
import sys
import warnings
@ -12,7 +13,7 @@ if sys.hexversion < 0x02070000:
else:
from collections import OrderedDict
class _Exif(object):
class UUIDExif(object):
"""
Attributes
----------
@ -25,10 +26,10 @@ class _Exif(object):
def __init__(self, read_buffer):
"""Interpret raw buffer consisting of Exif IFD.
"""
self.exif_image = None
self.exif_photo = None
self.exif_gpsinfo = None
self.exif_iop = None
exif_image = None
exif_photo = None
exif_gpsinfo = None
exif_iop = None
self.read_buffer = read_buffer
@ -45,24 +46,39 @@ class _Exif(object):
# This is the 'Exif Image' portion.
exif = _ExifImageIfd(self.endian, read_buffer[6:], offset)
self.exif_image = exif.processed_ifd
exif_image = exif.processed_ifd
if 'ExifTag' in self.exif_image.keys():
offset = self.exif_image['ExifTag']
photo = _ExifPhotoIfd(self.endian, read_buffer[6:], offset)
self.exif_photo = photo.processed_ifd
if 'ExifTag' in exif_image.keys():
offset = exif_image['ExifTag']
photo_ifd = _ExifPhotoIfd(self.endian, read_buffer[6:], offset)
exif_photo = photo_ifd.processed_ifd
if 'InteroperabilityTag' in self.exif_photo.keys():
offset = self.exif_photo['InteroperabilityTag']
if 'InteroperabilityTag' in exif_photo.keys():
offset = exif_photo['InteroperabilityTag']
interop = _ExifInteroperabilityIfd(self.endian,
read_buffer[6:],
offset)
self.iop = interop.processed_ifd
iop = interop.processed_ifd
if 'GPSTag' in self.exif_image.keys():
offset = self.exif_image['GPSTag']
if 'GPSTag' in exif_image.keys():
offset = exif_image['GPSTag']
gps = _ExifGPSInfoIfd(self.endian, read_buffer[6:], offset)
self.exif_gpsinfo = gps.processed_ifd
exif_gpsinfo = gps.processed_ifd
self.ifds = OrderedDict()
self.ifds['Image'] = exif_image
self.ifds['Photo'] = exif_photo
self.ifds['GPSInfo'] = exif_gpsinfo
self.ifds['Iop'] = exif_iop
def __str__(self):
# 2.7 has trouble pretty-printing ordered dicts, so print them
# as regular dicts. Not ideal, but at least it's good on 3.3+.
if sys.hexversion < 0x03000000:
data = dict(self.ifds)
else:
data = self.ifds
return '\n' + pprint.pformat(data)
class _Ifd(object):

46
glymur/_uuid_io/XMP.py Normal file
View file

@ -0,0 +1,46 @@
# -*- coding: utf-8 -*-
"""
Handler for a UUID for XMP.
"""
import sys
from xml.etree import cElementTree as ET
from ..core import _pretty_print_xml
class UUIDXMP(object):
"""
Handler for a UUID for XMP.
Attributes
----------
packet : ElementTree
XML conforming to the XMP specifications.
References
----------
.. [XMP] International Organization for Standardication. ISO/IEC
16684-1:2012 - Graphic technology -- Extensible metadata platform (XMP)
specification -- Part 1: Data model, serialization and core properties
"""
def __init__(self, read_buffer):
"""
Parameters
----------
read_buffer : byte array
sequence of bytes that can be decoded into an XMP packet.
"""
# XMP data. Parse as XML.
if sys.hexversion < 0x03000000:
# 2.x strings same as bytes
elt = ET.fromstring(read_buffer)
else:
# 3.x takes strings, not bytes.
text = read_buffer.decode('utf-8')
elt = ET.fromstring(text)
self.packet = ET.ElementTree(elt)
def __str__(self):
return _pretty_print_xml(self.packet)

View file

@ -1 +1,4 @@
from .Exif import _Exif
from .Exif import UUIDExif
from .XMP import UUIDXMP
from .generic import UUIDGeneric

View file

@ -0,0 +1,27 @@
# -*- coding: utf-8 -*-
"""
Handler for a generic UUID.
"""
class UUIDGeneric(object):
"""
Handler for a generic UUID that is not currently recognized.
Attributes
----------
data : byte array
Sequence of uninterpreted bytes as read from the file.
"""
def __init__(self, read_buffer):
"""
Parameters
----------
read_buffer : byte array
sequence of bytes as read from the file.
"""
self.data = read_buffer
def __str__(self):
return '{0} bytes'.format(len(self.data))

View file

@ -1,5 +1,8 @@
"""Core definitions to be shared amongst the modules.
"""
import copy
import xml.etree.cElementTree as ET
# Progression order
LRCP = 0
RLCP = 1
@ -73,3 +76,45 @@ _CAPABILITIES_DISPLAY = {
1: '0',
2: '1',
3: '3'}
def _pretty_print_xml(xml, level=0):
"""Pretty print XML data.
"""
xml = copy.deepcopy(xml)
_indent(xml.getroot(), level=level)
xmltext = ET.tostring(xml.getroot(), encoding='utf-8').decode('utf-8')
# Indent it a bit.
lst = [(' ' + x) for x in xmltext.split('\n')]
try:
xml = '\n'.join(lst)
return '\n{0}'.format(xml)
except UnicodeEncodeError:
# This can happen on python 2.x if the character set contains certain
# non-ascii characters. Just print out the corresponding xml char
# entities instead.
xml = u'\n'.join(lst)
text = u'\n{0}'.format(xml)
text = text.encode('ascii', 'xmlcharrefreplace')
return text
def _indent(elem, level=0):
"""Recipe for pretty printing XML. Please see
http://effbot.org/zone/element-lib.htm#prettyprint
"""
i = "\n" + level * " "
if len(elem):
if not elem.text or not elem.text.strip():
elem.text = i + " "
if not elem.tail or not elem.tail.strip():
elem.tail = i
for elem in elem:
_indent(elem, level + 1)
if not elem.tail or not elem.tail.strip():
elem.tail = i
else:
if level and (not elem.tail or not elem.tail.strip()):
elem.tail = i

View file

@ -13,7 +13,6 @@ References
# pylint: disable=C0302,R0903,R0913
import copy
import datetime
import math
import os
@ -38,8 +37,9 @@ from .core import _COLORSPACE_MAP_DISPLAY
from .core import _COLOR_TYPE_MAP_DISPLAY
from .core import ENUMERATED_COLORSPACE, RESTRICTED_ICC_PROFILE
from .core import ANY_ICC_PROFILE, VENDOR_COLOR_METHOD
from .core import _pretty_print_xml
from ._uuid_io import _Exif
from . import _uuid_io
_METHOD_DISPLAY = {
ENUMERATED_COLORSPACE: 'enumerated colorspace',
@ -2065,8 +2065,11 @@ class UUIDBox(Jp2kBox):
----------
the_uuid : uuid.UUID
Identifies the type of UUID box.
data : object
Specific to each type of UUID. There are handlers for XMP, Exif,
and unknown UUIDs.
raw_data : byte array
This is the "payload" of data for the specified UUID.
Sequence of uninterpreted bytes as read from the file.
length : int
length of the box in bytes.
offset : int
@ -2076,59 +2079,33 @@ class UUIDBox(Jp2kBox):
self.uuid = the_uuid
if the_uuid == uuid.UUID('be7acfcb-97a9-42e8-9c71-999491e3afac'):
# XMP data. Parse as XML. Seems to be a difference between
# ElementTree in version 2.7 and 3.3.
if sys.hexversion < 0x03000000:
elt = ET.fromstring(raw_data)
else:
text = raw_data.decode('utf-8')
elt = ET.fromstring(text)
self.data = ET.ElementTree(elt)
self.data = _uuid_io.UUIDXMP(raw_data)
self._type = 'XMP'
elif the_uuid.bytes == b'JpgTiffExif->JP2':
exif_obj = _Exif(raw_data)
ifds = OrderedDict()
ifds['Image'] = exif_obj.exif_image
ifds['Photo'] = exif_obj.exif_photo
ifds['GPSInfo'] = exif_obj.exif_gpsinfo
ifds['Iop'] = exif_obj.exif_iop
self.data = ifds
self.data = _uuid_io.UUIDExif(raw_data)
self._type = 'Exif'
else:
self.data = raw_data
self.data = _uuid_io.UUIDGeneric(raw_data)
self._type = 'unknown'
self.raw_data = raw_data
self.length = length
self.offset = offset
def __str__(self):
msg = '{0}\n'
msg += ' UUID: {1}{2}\n'
msg += ' UUID: {1} ({2})\n'
msg += ' UUID Data: {3}'
if self.uuid == uuid.UUID('be7acfcb-97a9-42e8-9c71-999491e3afac'):
uuid_type = ' (XMP)'
uuid_data = _pretty_print_xml(self.data)
elif self.uuid.bytes == b'JpgTiffExif->JP2':
uuid_type = ' (Exif)'
# 2.7 has trouble pretty-printing ordered dicts, so print them
# as regular dicts. Not ideal, but at least it's good on 3.3+.
if sys.hexversion < 0x03000000:
data = dict(self.data)
else:
data = self.data
uuid_data = '\n' + pprint.pformat(data)
else:
uuid_type = ''
uuid_data = '{0} bytes'.format(len(self.data))
msg = msg.format(Jp2kBox.__str__(self),
self.uuid,
uuid_type,
uuid_data)
self._type,
str(self.data))
return msg
def write(self, fptr):
"""Write a UUID box box to file.
"""
@ -2196,45 +2173,3 @@ _BOX_WITH_ID = {
'url ': DataEntryURLBox,
'uuid': UUIDBox,
'xml ': XMLBox}
def _indent(elem, level=0):
"""Recipe for pretty printing XML. Please see
http://effbot.org/zone/element-lib.htm#prettyprint
"""
i = "\n" + level * " "
if len(elem):
if not elem.text or not elem.text.strip():
elem.text = i + " "
if not elem.tail or not elem.tail.strip():
elem.tail = i
for elem in elem:
_indent(elem, level + 1)
if not elem.tail or not elem.tail.strip():
elem.tail = i
else:
if level and (not elem.tail or not elem.tail.strip()):
elem.tail = i
def _pretty_print_xml(xml, level=0):
"""Pretty print XML data.
"""
xml = copy.deepcopy(xml)
_indent(xml.getroot(), level=level)
xmltext = ET.tostring(xml.getroot(), encoding='utf-8').decode('utf-8')
# Indent it a bit.
lst = [(' ' + x) for x in xmltext.split('\n')]
try:
xml = '\n'.join(lst)
return '\n{0}'.format(xml)
except UnicodeEncodeError:
# This can happen on python 2.x if the character set contains certain
# non-ascii characters. Just print out the corresponding xml char
# entities instead.
xml = u'\n'.join(lst)
text = u'\n{0}'.format(xml)
text = text.encode('ascii', 'xmlcharrefreplace')
return text

View file

@ -353,7 +353,7 @@ class TestJp2k(unittest.TestCase):
def test_xmp_attribute(self):
"""Verify the XMP packet in the shipping example file can be read."""
j = Jp2k(self.jp2file)
xmp = j.box[3].data
xmp = j.box[3].data.packet
ns0 = '{http://www.w3.org/1999/02/22-rdf-syntax-ns#}'
ns2 = '{http://ns.adobe.com/xap/1.0/}'
name = '{0}RDF/{0}Description/{1}CreatorTool'.format(ns0, ns2)

View file

@ -636,6 +636,7 @@ class TestPrinting(unittest.TestCase):
actual = fake_out.getvalue().strip()
expected = nemo_xmp_box
self.maxDiff = None
self.assertEqual(actual, expected)
def test_codestream(self):
@ -1024,7 +1025,7 @@ class TestPrinting(unittest.TestCase):
print(jp2.box[4])
actual = fake_out.getvalue().strip()
lines = ['UUID Box (uuid) @ (1544, 25)',
' UUID: 3a0d0218-0ae9-4115-b376-4bca41ce0e71',
' UUID: 3a0d0218-0ae9-4115-b376-4bca41ce0e71 (unknown)',
' UUID Data: 1 bytes']
expected = '\n'.join(lines)