Libraries for Subliminal
This commit is contained in:
@@ -0,0 +1,56 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2003-2006 Thomas Schueppel <stain@acm.org>
|
||||
# Copyright (C) 2003-2006 Dirk Meyer <dischi@freevo.org>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
import mimetypes
|
||||
import os
|
||||
import sys
|
||||
from exceptions import *
|
||||
|
||||
PARSERS = [('asf', ['video/asf'], ['asf', 'wmv', 'wma']),
|
||||
('flv', ['video/flv'], ['flv']),
|
||||
('mkv', ['video/x-matroska', 'application/mkv'], ['mkv', 'mka', 'webm']),
|
||||
('mp4', ['video/quicktime', 'video/mp4'], ['mov', 'qt', 'mp4', 'mp4a', '3gp', '3gp2', '3g2', 'mk2']),
|
||||
('mpeg', ['video/mpeg'], ['mpeg', 'mpg', 'mp4', 'ts']),
|
||||
('ogm', ['application/ogg'], ['ogm', 'ogg', 'ogv']),
|
||||
('real', ['video/real'], ['rm', 'ra', 'ram']),
|
||||
('riff', ['video/avi'], ['wav', 'avi'])]
|
||||
|
||||
|
||||
def parse(path):
|
||||
if not os.path.isfile(path):
|
||||
raise ValueError('Invalid path')
|
||||
extension = os.path.splitext(path)[1][1:]
|
||||
mimetype = mimetypes.guess_type(path)[0]
|
||||
parser_ext = None
|
||||
parser_mime = None
|
||||
for (parser_name, parser_mimetypes, parser_extensions) in PARSERS:
|
||||
if mimetype in parser_mimetypes:
|
||||
parser_mime = parser_name
|
||||
if extension in parser_extensions:
|
||||
parser_ext = parser_name
|
||||
parser = parser_mime or parser_ext
|
||||
if not parser:
|
||||
raise NoParserError()
|
||||
mod = __import__(parser, globals=globals(), locals=locals(), fromlist=[], level=-1)
|
||||
with open(path, 'rb') as f:
|
||||
p = mod.Parser(f)
|
||||
return p
|
||||
@@ -0,0 +1,391 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2003-2006 Thomas Schueppel <stain@acm.org>
|
||||
# Copyright (C) 2003-2006 Dirk Meyer <dischi@freevo.org>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__all__ = ['Parser']
|
||||
|
||||
import struct
|
||||
import string
|
||||
import logging
|
||||
from exceptions import *
|
||||
import core
|
||||
|
||||
# get logging object
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
def _guid(input):
|
||||
# Remove any '-'
|
||||
s = string.join(string.split(input,'-'), '')
|
||||
r = ''
|
||||
if len(s) != 32:
|
||||
return ''
|
||||
x = ''
|
||||
for i in range(0,16):
|
||||
r+=chr(int(s[2*i:2*i+2],16))
|
||||
guid = struct.unpack('>IHHBB6s',r)
|
||||
return guid
|
||||
|
||||
GUIDS = {
|
||||
'ASF_Header_Object' : _guid('75B22630-668E-11CF-A6D9-00AA0062CE6C'),
|
||||
'ASF_Data_Object' : _guid('75B22636-668E-11CF-A6D9-00AA0062CE6C'),
|
||||
'ASF_Simple_Index_Object' : _guid('33000890-E5B1-11CF-89F4-00A0C90349CB'),
|
||||
'ASF_Index_Object' : _guid('D6E229D3-35DA-11D1-9034-00A0C90349BE'),
|
||||
'ASF_Media_Object_Index_Object' : _guid('FEB103F8-12AD-4C64-840F-2A1D2F7AD48C'),
|
||||
'ASF_Timecode_Index_Object' : _guid('3CB73FD0-0C4A-4803-953D-EDF7B6228F0C'),
|
||||
|
||||
'ASF_File_Properties_Object' : _guid('8CABDCA1-A947-11CF-8EE4-00C00C205365'),
|
||||
'ASF_Stream_Properties_Object' : _guid('B7DC0791-A9B7-11CF-8EE6-00C00C205365'),
|
||||
'ASF_Header_Extension_Object' : _guid('5FBF03B5-A92E-11CF-8EE3-00C00C205365'),
|
||||
'ASF_Codec_List_Object' : _guid('86D15240-311D-11D0-A3A4-00A0C90348F6'),
|
||||
'ASF_Script_Command_Object' : _guid('1EFB1A30-0B62-11D0-A39B-00A0C90348F6'),
|
||||
'ASF_Marker_Object' : _guid('F487CD01-A951-11CF-8EE6-00C00C205365'),
|
||||
'ASF_Bitrate_Mutual_Exclusion_Object' : _guid('D6E229DC-35DA-11D1-9034-00A0C90349BE'),
|
||||
'ASF_Error_Correction_Object' : _guid('75B22635-668E-11CF-A6D9-00AA0062CE6C'),
|
||||
'ASF_Content_Description_Object' : _guid('75B22633-668E-11CF-A6D9-00AA0062CE6C'),
|
||||
'ASF_Extended_Content_Description_Object' : _guid('D2D0A440-E307-11D2-97F0-00A0C95EA850'),
|
||||
'ASF_Content_Branding_Object' : _guid('2211B3FA-BD23-11D2-B4B7-00A0C955FC6E'),
|
||||
'ASF_Stream_Bitrate_Properties_Object' : _guid('7BF875CE-468D-11D1-8D82-006097C9A2B2'),
|
||||
'ASF_Content_Encryption_Object' : _guid('2211B3FB-BD23-11D2-B4B7-00A0C955FC6E'),
|
||||
'ASF_Extended_Content_Encryption_Object' : _guid('298AE614-2622-4C17-B935-DAE07EE9289C'),
|
||||
'ASF_Alt_Extended_Content_Encryption_Obj' : _guid('FF889EF1-ADEE-40DA-9E71-98704BB928CE'),
|
||||
'ASF_Digital_Signature_Object' : _guid('2211B3FC-BD23-11D2-B4B7-00A0C955FC6E'),
|
||||
'ASF_Padding_Object' : _guid('1806D474-CADF-4509-A4BA-9AABCB96AAE8'),
|
||||
|
||||
'ASF_Extended_Stream_Properties_Object' : _guid('14E6A5CB-C672-4332-8399-A96952065B5A'),
|
||||
'ASF_Advanced_Mutual_Exclusion_Object' : _guid('A08649CF-4775-4670-8A16-6E35357566CD'),
|
||||
'ASF_Group_Mutual_Exclusion_Object' : _guid('D1465A40-5A79-4338-B71B-E36B8FD6C249'),
|
||||
'ASF_Stream_Prioritization_Object' : _guid('D4FED15B-88D3-454F-81F0-ED5C45999E24'),
|
||||
'ASF_Bandwidth_Sharing_Object' : _guid('A69609E6-517B-11D2-B6AF-00C04FD908E9'),
|
||||
'ASF_Language_List_Object' : _guid('7C4346A9-EFE0-4BFC-B229-393EDE415C85'),
|
||||
'ASF_Metadata_Object' : _guid('C5F8CBEA-5BAF-4877-8467-AA8C44FA4CCA'),
|
||||
'ASF_Metadata_Library_Object' : _guid('44231C94-9498-49D1-A141-1D134E457054'),
|
||||
'ASF_Index_Parameters_Object' : _guid('D6E229DF-35DA-11D1-9034-00A0C90349BE'),
|
||||
'ASF_Media_Object_Index_Parameters_Obj' : _guid('6B203BAD-3F11-4E84-ACA8-D7613DE2CFA7'),
|
||||
'ASF_Timecode_Index_Parameters_Object' : _guid('F55E496D-9797-4B5D-8C8B-604DFE9BFB24'),
|
||||
|
||||
'ASF_Audio_Media' : _guid('F8699E40-5B4D-11CF-A8FD-00805F5C442B'),
|
||||
'ASF_Video_Media' : _guid('BC19EFC0-5B4D-11CF-A8FD-00805F5C442B'),
|
||||
'ASF_Command_Media' : _guid('59DACFC0-59E6-11D0-A3AC-00A0C90348F6'),
|
||||
'ASF_JFIF_Media' : _guid('B61BE100-5B4E-11CF-A8FD-00805F5C442B'),
|
||||
'ASF_Degradable_JPEG_Media' : _guid('35907DE0-E415-11CF-A917-00805F5C442B'),
|
||||
'ASF_File_Transfer_Media' : _guid('91BD222C-F21C-497A-8B6D-5AA86BFC0185'),
|
||||
'ASF_Binary_Media' : _guid('3AFB65E2-47EF-40F2-AC2C-70A90D71D343'),
|
||||
|
||||
'ASF_Web_Stream_Media_Subtype' : _guid('776257D4-C627-41CB-8F81-7AC7FF1C40CC'),
|
||||
'ASF_Web_Stream_Format' : _guid('DA1E6B13-8359-4050-B398-388E965BF00C'),
|
||||
|
||||
'ASF_No_Error_Correction' : _guid('20FB5700-5B55-11CF-A8FD-00805F5C442B'),
|
||||
'ASF_Audio_Spread' : _guid('BFC3CD50-618F-11CF-8BB2-00AA00B4E220'),
|
||||
}
|
||||
|
||||
|
||||
class Asf(core.AVContainer):
|
||||
"""
|
||||
ASF video parser. The ASF format is also used for Microsft Windows
|
||||
Media files like wmv.
|
||||
"""
|
||||
def __init__(self, file):
|
||||
core.AVContainer.__init__(self)
|
||||
self.mime = 'video/x-ms-asf'
|
||||
self.type = 'asf format'
|
||||
self._languages = []
|
||||
self._extinfo = {}
|
||||
|
||||
h = file.read(30)
|
||||
if len(h) < 30:
|
||||
raise ParseError()
|
||||
|
||||
(guidstr, objsize, objnum, reserved1, \
|
||||
reserved2) = struct.unpack('<16sQIBB',h)
|
||||
guid = self._parseguid(guidstr)
|
||||
|
||||
if (guid != GUIDS['ASF_Header_Object']):
|
||||
raise ParseError()
|
||||
if reserved1 != 0x01 or reserved2 != 0x02:
|
||||
raise ParseError()
|
||||
|
||||
log.debug("asf header size: %d / %d objects" % (objsize,objnum))
|
||||
header = file.read(objsize-30)
|
||||
for i in range(0,objnum):
|
||||
h = self._getnextheader(header)
|
||||
header = header[h[1]:]
|
||||
|
||||
del self._languages
|
||||
del self._extinfo
|
||||
|
||||
|
||||
def _findstream(self, id):
|
||||
for stream in self.video + self.audio:
|
||||
if stream.id == id:
|
||||
return stream
|
||||
|
||||
def _apply_extinfo(self, streamid):
|
||||
stream = self._findstream(streamid)
|
||||
if not stream or streamid not in self._extinfo:
|
||||
return
|
||||
stream.bitrate, stream.fps, langid, metadata = self._extinfo[streamid]
|
||||
if langid is not None and langid >= 0 and langid < len(self._languages):
|
||||
stream.language = self._languages[langid]
|
||||
if metadata:
|
||||
stream._appendtable('ASFMETADATA', metadata)
|
||||
|
||||
|
||||
def _parseguid(self,string):
|
||||
return struct.unpack('<IHHBB6s', string[:16])
|
||||
|
||||
|
||||
def _parsekv(self,s):
|
||||
pos = 0
|
||||
(descriptorlen,) = struct.unpack('<H', s[pos:pos+2])
|
||||
pos += 2
|
||||
descriptorname = s[pos:pos+descriptorlen]
|
||||
pos += descriptorlen
|
||||
descriptortype, valuelen = struct.unpack('<HH', s[pos:pos+4])
|
||||
pos += 4
|
||||
descriptorvalue = s[pos:pos+valuelen]
|
||||
pos += valuelen
|
||||
value = None
|
||||
if descriptortype == 0x0000:
|
||||
# Unicode string
|
||||
value = descriptorvalue
|
||||
elif descriptortype == 0x0001:
|
||||
# Byte Array
|
||||
value = descriptorvalue
|
||||
elif descriptortype == 0x0002:
|
||||
# Bool (?)
|
||||
value = struct.unpack('<I', descriptorvalue)[0] != 0
|
||||
elif descriptortype == 0x0003:
|
||||
# DWORD
|
||||
value = struct.unpack('<I', descriptorvalue)[0]
|
||||
elif descriptortype == 0x0004:
|
||||
# QWORD
|
||||
value = struct.unpack('<Q', descriptorvalue)[0]
|
||||
elif descriptortype == 0x0005:
|
||||
# WORD
|
||||
value = struct.unpack('<H', descriptorvalue)[0]
|
||||
else:
|
||||
log.debug("Unknown Descriptor Type %d" % descriptortype)
|
||||
return (pos,descriptorname,value)
|
||||
|
||||
|
||||
def _parsekv2(self,s):
|
||||
pos = 0
|
||||
strno, descriptorlen, descriptortype, valuelen = struct.unpack('<2xHHHI', s[pos:pos+12])
|
||||
pos += 12
|
||||
descriptorname = s[pos:pos+descriptorlen]
|
||||
pos += descriptorlen
|
||||
descriptorvalue = s[pos:pos+valuelen]
|
||||
pos += valuelen
|
||||
value = None
|
||||
|
||||
if descriptortype == 0x0000:
|
||||
# Unicode string
|
||||
value = descriptorvalue
|
||||
elif descriptortype == 0x0001:
|
||||
# Byte Array
|
||||
value = descriptorvalue
|
||||
elif descriptortype == 0x0002:
|
||||
# Bool
|
||||
value = struct.unpack('<H', descriptorvalue)[0] != 0
|
||||
pass
|
||||
elif descriptortype == 0x0003:
|
||||
# DWORD
|
||||
value = struct.unpack('<I', descriptorvalue)[0]
|
||||
elif descriptortype == 0x0004:
|
||||
# QWORD
|
||||
value = struct.unpack('<Q', descriptorvalue)[0]
|
||||
elif descriptortype == 0x0005:
|
||||
# WORD
|
||||
value = struct.unpack('<H', descriptorvalue)[0]
|
||||
else:
|
||||
log.debug("Unknown Descriptor Type %d" % descriptortype)
|
||||
return (pos,descriptorname,value,strno)
|
||||
|
||||
|
||||
def _getnextheader(self,s):
|
||||
r = struct.unpack('<16sQ',s[:24])
|
||||
(guidstr,objsize) = r
|
||||
guid = self._parseguid(guidstr)
|
||||
if guid == GUIDS['ASF_File_Properties_Object']:
|
||||
log.debug("File Properties Object")
|
||||
val = struct.unpack('<16s6Q4I',s[24:24+80])
|
||||
(fileid, size, date, packetcount, duration, \
|
||||
senddur, preroll, flags, minpack, maxpack, maxbr) = \
|
||||
val
|
||||
# FIXME: parse date to timestamp
|
||||
self.length = duration/10000000.0
|
||||
|
||||
elif guid == GUIDS['ASF_Stream_Properties_Object']:
|
||||
log.debug("Stream Properties Object [%d]" % objsize)
|
||||
streamtype = self._parseguid(s[24:40])
|
||||
errortype = self._parseguid(s[40:56])
|
||||
offset, typelen, errorlen, flags = struct.unpack('<QIIH', s[56:74])
|
||||
strno = flags & 0x7f
|
||||
encrypted = flags >> 15
|
||||
if encrypted:
|
||||
self._set('encrypted', True)
|
||||
if streamtype == GUIDS['ASF_Video_Media']:
|
||||
vi = core.VideoStream()
|
||||
vi.width, vi.height, depth, codec, = struct.unpack('<4xII2xH4s', s[89:89+20])
|
||||
vi.codec = codec
|
||||
vi.id = strno
|
||||
self.video.append(vi)
|
||||
elif streamtype == GUIDS['ASF_Audio_Media']:
|
||||
ai = core.AudioStream()
|
||||
twocc, ai.channels, ai.samplerate, bitrate, block, \
|
||||
ai.samplebits, = struct.unpack('<HHIIHH', s[78:78+16])
|
||||
ai.bitrate = 8*bitrate
|
||||
ai.codec = twocc
|
||||
ai.id = strno
|
||||
self.audio.append(ai)
|
||||
|
||||
self._apply_extinfo(strno)
|
||||
|
||||
elif guid == GUIDS['ASF_Extended_Stream_Properties_Object']:
|
||||
streamid, langid, frametime = struct.unpack('<HHQ',s[72:84])
|
||||
(bitrate,) = struct.unpack('<I', s[40:40+4])
|
||||
if streamid not in self._extinfo:
|
||||
self._extinfo[streamid] = [None, None, None, {}]
|
||||
if frametime == 0:
|
||||
# Problaby VFR, report as 1000fps (which is what MPlayer does)
|
||||
frametime = 10000.0
|
||||
self._extinfo[streamid][:3] = [bitrate, 10000000.0 / frametime, langid]
|
||||
self._apply_extinfo(streamid)
|
||||
|
||||
elif guid == GUIDS['ASF_Header_Extension_Object']:
|
||||
log.debug("ASF_Header_Extension_Object %d" % objsize)
|
||||
size = struct.unpack('<I',s[42:46])[0]
|
||||
data = s[46:46+size]
|
||||
while len(data):
|
||||
log.debug("Sub:")
|
||||
h = self._getnextheader(data)
|
||||
data = data[h[1]:]
|
||||
|
||||
elif guid == GUIDS['ASF_Codec_List_Object']:
|
||||
log.debug("List Object")
|
||||
pass
|
||||
|
||||
elif guid == GUIDS['ASF_Error_Correction_Object']:
|
||||
log.debug("Error Correction")
|
||||
pass
|
||||
|
||||
elif guid == GUIDS['ASF_Content_Description_Object']:
|
||||
log.debug("Content Description Object")
|
||||
val = struct.unpack('<5H', s[24:24+10])
|
||||
pos = 34
|
||||
strings = []
|
||||
for i in val:
|
||||
ss = s[pos:pos+i].replace('\0', '').lstrip().rstrip()
|
||||
strings.append(ss)
|
||||
pos+=i
|
||||
|
||||
# Set empty strings to None
|
||||
strings = [x or None for x in strings]
|
||||
self.title, self.artist, self.copyright, self.caption, rating = strings
|
||||
|
||||
elif guid == GUIDS['ASF_Extended_Content_Description_Object']:
|
||||
(count,) = struct.unpack('<H', s[24:26])
|
||||
pos = 26
|
||||
descriptor = {}
|
||||
for i in range(0, count):
|
||||
# Read additional content descriptors
|
||||
d = self._parsekv(s[pos:])
|
||||
pos += d[0]
|
||||
descriptor[d[1]] = d[2]
|
||||
self._appendtable('ASFDESCRIPTOR', descriptor)
|
||||
|
||||
elif guid == GUIDS['ASF_Metadata_Object']:
|
||||
(count,) = struct.unpack('<H', s[24:26])
|
||||
pos = 26
|
||||
streams = {}
|
||||
for i in range(0, count):
|
||||
# Read additional content descriptors
|
||||
size,key,value,strno = self._parsekv2(s[pos:])
|
||||
if strno not in streams:
|
||||
streams[strno] = {}
|
||||
streams[strno][key] = value
|
||||
pos += size
|
||||
|
||||
for strno, metadata in streams.items():
|
||||
if strno not in self._extinfo:
|
||||
self._extinfo[strno] = [None, None, None, {}]
|
||||
self._extinfo[strno][3].update(metadata)
|
||||
self._apply_extinfo(strno)
|
||||
|
||||
elif guid == GUIDS['ASF_Language_List_Object']:
|
||||
count = struct.unpack('<H', s[24:26])[0]
|
||||
pos = 26
|
||||
for i in range(0, count):
|
||||
idlen = struct.unpack('<B', s[pos:pos+1])[0]
|
||||
idstring = s[pos+1:pos+1+idlen]
|
||||
idstring = unicode(idstring, 'utf-16').replace('\0', '')
|
||||
log.debug("Language: %d/%d: %s" % (i+1, count, idstring))
|
||||
self._languages.append(idstring)
|
||||
pos += 1+idlen
|
||||
|
||||
elif guid == GUIDS['ASF_Stream_Bitrate_Properties_Object']:
|
||||
# This record contains stream bitrate with payload overhead. For
|
||||
# audio streams, we should have the average bitrate from
|
||||
# ASF_Stream_Properties_Object. For video streams, we get it from
|
||||
# ASF_Extended_Stream_Properties_Object. So this record is not
|
||||
# used.
|
||||
pass
|
||||
|
||||
elif guid == GUIDS['ASF_Content_Encryption_Object'] or \
|
||||
guid == GUIDS['ASF_Extended_Content_Encryption_Object']:
|
||||
self._set('encrypted', True)
|
||||
else:
|
||||
# Just print the type:
|
||||
for h in GUIDS.keys():
|
||||
if GUIDS[h] == guid:
|
||||
log.debug("Unparsed %s [%d]" % (h,objsize))
|
||||
break
|
||||
else:
|
||||
u = "%.8X-%.4X-%.4X-%.2X%.2X-%s" % guid
|
||||
log.debug("unknown: len=%d [%d]" % (len(u), objsize))
|
||||
return r
|
||||
|
||||
|
||||
class AsfAudio(core.AudioStream):
|
||||
"""
|
||||
ASF audio parser for wma files.
|
||||
"""
|
||||
def __init__(self):
|
||||
core.AudioStream.__init__(self)
|
||||
self.mime = 'audio/x-ms-asf'
|
||||
self.type = 'asf format'
|
||||
|
||||
|
||||
def Parser(file):
|
||||
"""
|
||||
Wrapper around audio and av content.
|
||||
"""
|
||||
asf = Asf(file)
|
||||
if not len(asf.audio) or len(asf.video):
|
||||
# AV container
|
||||
return asf
|
||||
# No video but audio streams. Handle has audio core
|
||||
audio = AsfAudio()
|
||||
for key in audio._keys:
|
||||
if key in asf._keys:
|
||||
if not getattr(audio, key, None):
|
||||
setattr(audio, key, getattr(asf, key))
|
||||
return audio
|
||||
@@ -0,0 +1,457 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2003-2006 Thomas Schueppel <stain@acm.org>
|
||||
# Copyright (C) 2003-2006 Dirk Meyer <dischi@freevo.org>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
|
||||
import re
|
||||
import logging
|
||||
import fourcc
|
||||
import language
|
||||
from exceptions import *
|
||||
from strutils import str_to_unicode, unicode_to_str
|
||||
|
||||
UNPRINTABLE_KEYS = ['thumbnail', 'url', 'codec_private']
|
||||
EXTENSION_DEVICE = 'device'
|
||||
EXTENSION_DIRECTORY = 'directory'
|
||||
EXTENSION_STREAM = 'stream'
|
||||
MEDIACORE = ['title', 'caption', 'comment', 'size', 'type', 'subtype', 'timestamp',
|
||||
'keywords', 'country', 'language', 'langcode', 'url', 'artist',
|
||||
'mime', 'datetime', 'tags', 'hash']
|
||||
AUDIOCORE = ['channels', 'samplerate', 'length', 'encoder', 'codec', 'format',
|
||||
'samplebits', 'bitrate', 'fourcc', 'trackno', 'id', 'userdate',
|
||||
'enabled', 'default', 'codec_private']
|
||||
MUSICCORE = ['trackof', 'album', 'genre', 'discs', 'thumbnail']
|
||||
VIDEOCORE = ['length', 'encoder', 'bitrate', 'samplerate', 'codec', 'format',
|
||||
'samplebits', 'width', 'height', 'fps', 'aspect', 'trackno',
|
||||
'fourcc', 'id', 'enabled', 'default', 'codec_private']
|
||||
AVCORE = ['length', 'encoder', 'trackno', 'trackof', 'copyright', 'product',
|
||||
'genre', 'writer', 'producer', 'studio', 'rating', 'actors', 'thumbnail',
|
||||
'delay', 'image', 'video', 'audio', 'subtitles', 'chapters', 'software',
|
||||
'summary', 'synopsis', 'season', 'episode', 'series']
|
||||
|
||||
# get logging object
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Media(object):
|
||||
media = None
|
||||
|
||||
"""
|
||||
Media is the base class to all Media Metadata Containers. It defines
|
||||
the basic structures that handle metadata. Media and its derivates
|
||||
contain a common set of metadata attributes that is listed in keys.
|
||||
Specific derivates contain additional keys to the dublin core set that is
|
||||
defined in Media.
|
||||
"""
|
||||
_keys = MEDIACORE
|
||||
table_mapping = {}
|
||||
|
||||
def __init__(self, hash=None):
|
||||
if hash is not None:
|
||||
# create Media based on dict
|
||||
for key, value in hash.items():
|
||||
if isinstance(value, list) and value and isinstance(value[0], dict):
|
||||
value = [Media(x) for x in value]
|
||||
self._set(key, value)
|
||||
return
|
||||
|
||||
self._keys = self._keys[:]
|
||||
self.tables = {}
|
||||
# Tags, unlike tables, are more well-defined dicts whose values are
|
||||
# either Tag objects, other dicts (for nested tags), or lists of either
|
||||
# (for multiple instances of the tag, e.g. actor). Where possible,
|
||||
# parsers should transform tag names to conform to the Official
|
||||
# Matroska tags defined at http://www.matroska.org/technical/specs/tagging/index.html
|
||||
# All tag names will be lower-cased.
|
||||
self.tags = Tags()
|
||||
for key in set(self._keys) - set(['media', 'tags']):
|
||||
setattr(self, key, None)
|
||||
|
||||
#
|
||||
# unicode and string convertion for debugging
|
||||
#
|
||||
#TODO: Fix that mess
|
||||
def __unicode__(self):
|
||||
result = u''
|
||||
|
||||
# print normal attributes
|
||||
lists = []
|
||||
for key in self._keys:
|
||||
value = getattr(self, key, None)
|
||||
if value == None or key == 'url':
|
||||
continue
|
||||
if isinstance(value, list):
|
||||
if not value:
|
||||
continue
|
||||
elif isinstance(value[0], basestring):
|
||||
# Just a list of strings (keywords?), so don't treat it specially.
|
||||
value = u', '.join(value)
|
||||
else:
|
||||
lists.append((key, value))
|
||||
continue
|
||||
elif isinstance(value, dict):
|
||||
# Tables or tags treated separately.
|
||||
continue
|
||||
if key in UNPRINTABLE_KEYS:
|
||||
value = '<unprintable data, size=%d>' % len(value)
|
||||
result += u'| %10s: %s\n' % (unicode(key), unicode(value))
|
||||
|
||||
# print tags (recursively, to support nested tags).
|
||||
def print_tags(tags, suffix, show_label):
|
||||
result = ''
|
||||
for n, (name, tag) in enumerate(tags.items()):
|
||||
result += u'| %12s%s%s = ' % (u'tags: ' if n == 0 and show_label else '', suffix, name)
|
||||
if isinstance(tag, list):
|
||||
# TODO: doesn't support lists/dicts within lists.
|
||||
result += u'%s\n' % ', '.join(subtag.value for subtag in tag)
|
||||
else:
|
||||
result += u'%s\n' % (tag.value or '')
|
||||
if isinstance(tag, dict):
|
||||
result += print_tags(tag, ' ', False)
|
||||
return result
|
||||
result += print_tags(self.tags, '', True)
|
||||
|
||||
# print lists
|
||||
for key, l in lists:
|
||||
for n, item in enumerate(l):
|
||||
label = '+-- ' + key.rstrip('s').capitalize()
|
||||
if key not in ['tracks', 'subtitles', 'chapters']:
|
||||
label += ' Track'
|
||||
result += u'%s #%d\n' % (label, n + 1)
|
||||
result += '| ' + re.sub(r'\n(.)', r'\n| \1', unicode(item))
|
||||
|
||||
# print tables
|
||||
if log.level >= 10:
|
||||
for name, table in self.tables.items():
|
||||
result += '+-- Table %s\n' % str(name)
|
||||
for key, value in table.items():
|
||||
try:
|
||||
value = unicode(value)
|
||||
if len(value) > 50:
|
||||
value = u'<unprintable data, size=%d>' % len(value)
|
||||
except (UnicodeDecodeError, TypeError), e:
|
||||
try:
|
||||
value = u'<unprintable data, size=%d>' % len(value)
|
||||
except AttributeError:
|
||||
value = u'<unprintable data>'
|
||||
result += u'| | %s: %s\n' % (unicode(key), value)
|
||||
return result
|
||||
|
||||
def __str__(self):
|
||||
return unicode(self).encode()
|
||||
|
||||
def __repr__(self):
|
||||
if hasattr(self, 'url'):
|
||||
return '<%s %s>' % (str(self.__class__)[8:-2], self.url)
|
||||
else:
|
||||
return '<%s>' % (str(self.__class__)[8:-2])
|
||||
|
||||
#
|
||||
# internal functions
|
||||
#
|
||||
def _appendtable(self, name, hashmap):
|
||||
"""
|
||||
Appends a tables of additional metadata to the Object.
|
||||
If such a table already exists, the given tables items are
|
||||
added to the existing one.
|
||||
"""
|
||||
if name not in self.tables:
|
||||
self.tables[name] = hashmap
|
||||
else:
|
||||
# Append to the already existing table
|
||||
for k in hashmap.keys():
|
||||
self.tables[name][k] = hashmap[k]
|
||||
|
||||
def _set(self, key, value):
|
||||
"""
|
||||
Set key to value and add the key to the internal keys list if
|
||||
missing.
|
||||
"""
|
||||
if value is None and getattr(self, key, None) is None:
|
||||
return
|
||||
if isinstance(value, str):
|
||||
value = str_to_unicode(value)
|
||||
setattr(self, key, value)
|
||||
if not key in self._keys:
|
||||
self._keys.append(key)
|
||||
|
||||
def _set_url(self, url):
|
||||
"""
|
||||
Set the URL of the source
|
||||
"""
|
||||
self.url = url
|
||||
|
||||
def _finalize(self):
|
||||
"""
|
||||
Correct same data based on specific rules
|
||||
"""
|
||||
# make sure all strings are unicode
|
||||
for key in self._keys:
|
||||
if key in UNPRINTABLE_KEYS:
|
||||
continue
|
||||
value = getattr(self, key)
|
||||
if value is None:
|
||||
continue
|
||||
if key == 'image':
|
||||
if isinstance(value, unicode):
|
||||
setattr(self, key, unicode_to_str(value))
|
||||
continue
|
||||
if isinstance(value, str):
|
||||
setattr(self, key, str_to_unicode(value))
|
||||
if isinstance(value, unicode):
|
||||
setattr(self, key, value.strip().rstrip().replace(u'\0', u''))
|
||||
if isinstance(value, list) and value and isinstance(value[0], Media):
|
||||
for submenu in value:
|
||||
submenu._finalize()
|
||||
|
||||
# copy needed tags from tables
|
||||
for name, table in self.tables.items():
|
||||
mapping = self.table_mapping.get(name, {})
|
||||
for tag, attr in mapping.items():
|
||||
if self.get(attr):
|
||||
continue
|
||||
value = table.get(tag, None)
|
||||
if value is not None:
|
||||
if not isinstance(value, (str, unicode)):
|
||||
value = str_to_unicode(str(value))
|
||||
elif isinstance(value, str):
|
||||
value = str_to_unicode(value)
|
||||
value = value.strip().rstrip().replace(u'\0', u'')
|
||||
setattr(self, attr, value)
|
||||
|
||||
if 'fourcc' in self._keys and 'codec' in self._keys and self.codec is not None:
|
||||
# Codec may be a fourcc, in which case we resolve it to its actual
|
||||
# name and set the fourcc attribute.
|
||||
self.fourcc, self.codec = fourcc.resolve(self.codec)
|
||||
if 'language' in self._keys:
|
||||
self.langcode, self.language = language.resolve(self.language)
|
||||
|
||||
#
|
||||
# data access
|
||||
#
|
||||
def __contains__(self, key):
|
||||
"""
|
||||
Test if key exists in the dict
|
||||
"""
|
||||
return hasattr(self, key)
|
||||
|
||||
def get(self, attr, default=None):
|
||||
"""
|
||||
Returns the given attribute. If the attribute is not set by
|
||||
the parser return 'default'.
|
||||
"""
|
||||
return getattr(self, attr, default)
|
||||
|
||||
def __getitem__(self, attr):
|
||||
"""
|
||||
Get the value of the given attribute
|
||||
"""
|
||||
return getattr(self, attr, None)
|
||||
|
||||
def __setitem__(self, key, value):
|
||||
"""
|
||||
Set the value of 'key' to 'value'
|
||||
"""
|
||||
setattr(self, key, value)
|
||||
|
||||
def has_key(self, key):
|
||||
"""
|
||||
Check if the object has an attribute 'key'
|
||||
"""
|
||||
return hasattr(self, key)
|
||||
|
||||
def convert(self):
|
||||
"""
|
||||
Convert Media to dict.
|
||||
"""
|
||||
result = {}
|
||||
for k in self._keys:
|
||||
value = getattr(self, k, None)
|
||||
if isinstance(value, list) and value and isinstance(value[0], Media):
|
||||
value = [x.convert() for x in value]
|
||||
result[k] = value
|
||||
return result
|
||||
|
||||
def keys(self):
|
||||
"""
|
||||
Return all keys for the attributes set by the parser.
|
||||
"""
|
||||
return self._keys
|
||||
|
||||
|
||||
class Collection(Media):
|
||||
"""
|
||||
Collection of Digial Media like CD, DVD, Directory, Playlist
|
||||
"""
|
||||
_keys = Media._keys + ['id', 'tracks']
|
||||
|
||||
def __init__(self):
|
||||
Media.__init__(self)
|
||||
self.tracks = []
|
||||
|
||||
|
||||
class Tag(object):
|
||||
"""
|
||||
An individual tag, which will be a value stored in a Tags object.
|
||||
|
||||
Tag values are strings (for binary data), unicode objects, or datetime
|
||||
objects for tags that represent dates or times.
|
||||
"""
|
||||
def __init__(self, value=None, langcode='und', binary=False):
|
||||
super(Tag, self).__init__()
|
||||
self.value = value
|
||||
self.langcode = langcode
|
||||
self.binary = binary
|
||||
|
||||
def __unicode__(self):
|
||||
return unicode(self.value)
|
||||
|
||||
def __str__(self):
|
||||
return str(self.value)
|
||||
|
||||
def __repr__(self):
|
||||
if not self.binary:
|
||||
return '<Tag object: %s>' % repr(self.value)
|
||||
else:
|
||||
return '<Binary Tag object: size=%d>' % len(self.value)
|
||||
|
||||
@property
|
||||
def langcode(self):
|
||||
return self._langcode
|
||||
|
||||
@langcode.setter
|
||||
def langcode(self, code):
|
||||
self._langcode, self.language = language.resolve(code)
|
||||
|
||||
|
||||
class Tags(dict, Tag):
|
||||
"""
|
||||
A dictionary containing Tag objects. Values can be other Tags objects
|
||||
(for nested tags), lists, or Tag objects.
|
||||
|
||||
A Tags object is more or less a dictionary but it also contains a value.
|
||||
This is necessary in order to represent this kind of tag specification
|
||||
(e.g. for Matroska)::
|
||||
|
||||
<Simple>
|
||||
<Name>LAW_RATING</Name>
|
||||
<String>PG</String>
|
||||
<Simple>
|
||||
<Name>COUNTRY</Name>
|
||||
<String>US</String>
|
||||
</Simple>
|
||||
</Simple>
|
||||
|
||||
The attribute RATING has a value (PG), but it also has a child tag
|
||||
COUNTRY that specifies the country code the rating belongs to.
|
||||
"""
|
||||
def __init__(self, value=None, langcode='und', binary=False):
|
||||
super(Tags, self).__init__()
|
||||
self.value = value
|
||||
self.langcode = langcode
|
||||
self.binary = False
|
||||
|
||||
|
||||
class AudioStream(Media):
|
||||
"""
|
||||
Audio Tracks in a Multiplexed Container.
|
||||
"""
|
||||
_keys = Media._keys + AUDIOCORE
|
||||
|
||||
|
||||
class Music(AudioStream):
|
||||
"""
|
||||
Digital Music.
|
||||
"""
|
||||
_keys = AudioStream._keys + MUSICCORE
|
||||
|
||||
def _finalize(self):
|
||||
"""
|
||||
Correct same data based on specific rules
|
||||
"""
|
||||
AudioStream._finalize(self)
|
||||
if self.trackof:
|
||||
try:
|
||||
# XXX Why is this needed anyway?
|
||||
if int(self.trackno) < 10:
|
||||
self.trackno = u'0%s' % int(self.trackno)
|
||||
except (AttributeError, ValueError):
|
||||
pass
|
||||
|
||||
|
||||
class VideoStream(Media):
|
||||
"""
|
||||
Video Tracks in a Multiplexed Container.
|
||||
"""
|
||||
_keys = Media._keys + VIDEOCORE
|
||||
|
||||
|
||||
class Chapter(Media):
|
||||
"""
|
||||
Chapter in a Multiplexed Container.
|
||||
"""
|
||||
_keys = ['enabled', 'name', 'pos', 'id']
|
||||
|
||||
def __init__(self, name=None, pos=0):
|
||||
Media.__init__(self)
|
||||
self.name = name
|
||||
self.pos = pos
|
||||
self.enabled = True
|
||||
|
||||
|
||||
class Subtitle(Media):
|
||||
"""
|
||||
Subtitle Tracks in a Multiplexed Container.
|
||||
"""
|
||||
_keys = ['enabled', 'default', 'langcode', 'language', 'trackno', 'title',
|
||||
'id', 'codec']
|
||||
|
||||
def __init__(self, language=None):
|
||||
Media.__init__(self)
|
||||
self.language = language
|
||||
|
||||
|
||||
class AVContainer(Media):
|
||||
"""
|
||||
Container for Audio and Video streams. This is the Container Type for
|
||||
all media, that contain more than one stream.
|
||||
"""
|
||||
_keys = Media._keys + AVCORE
|
||||
|
||||
def __init__(self):
|
||||
Media.__init__(self)
|
||||
self.audio = []
|
||||
self.video = []
|
||||
self.subtitles = []
|
||||
self.chapters = []
|
||||
|
||||
def _finalize(self):
|
||||
"""
|
||||
Correct same data based on specific rules
|
||||
"""
|
||||
Media._finalize(self)
|
||||
if not self.length and len(self.video) and self.video[0].length:
|
||||
self.length = 0
|
||||
# Length not specified for container, so use the largest length
|
||||
# of its tracks as container length.
|
||||
for track in self.video + self.audio:
|
||||
if track.length:
|
||||
self.length = max(self.length, track.length)
|
||||
@@ -0,0 +1,31 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
|
||||
class Error(Exception):
|
||||
pass
|
||||
|
||||
|
||||
class NoParserError(Error):
|
||||
pass
|
||||
|
||||
|
||||
class ParseError(Error):
|
||||
pass
|
||||
@@ -0,0 +1,182 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2003-2006 Dirk Meyer <dischi@freevo.org>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__all__ = ['Parser']
|
||||
|
||||
from exceptions import *
|
||||
import core
|
||||
import logging
|
||||
import struct
|
||||
|
||||
# get logging object
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
FLV_TAG_TYPE_AUDIO = 0x08
|
||||
FLV_TAG_TYPE_VIDEO = 0x09
|
||||
FLV_TAG_TYPE_META = 0x12
|
||||
|
||||
# audio flags
|
||||
FLV_AUDIO_CHANNEL_MASK = 0x01
|
||||
FLV_AUDIO_SAMPLERATE_MASK = 0x0c
|
||||
FLV_AUDIO_CODECID_MASK = 0xf0
|
||||
|
||||
FLV_AUDIO_SAMPLERATE_OFFSET = 2
|
||||
FLV_AUDIO_CODECID_OFFSET = 4
|
||||
FLV_AUDIO_CODECID = (0x0001, 0x0002, 0x0055, 0x0001)
|
||||
|
||||
# video flags
|
||||
FLV_VIDEO_CODECID_MASK = 0x0f
|
||||
FLV_VIDEO_CODECID = ( 'FLV1', 'MSS1', 'VP60') # wild guess
|
||||
|
||||
FLV_DATA_TYPE_NUMBER = 0x00
|
||||
FLV_DATA_TYPE_BOOL = 0x01
|
||||
FLV_DATA_TYPE_STRING = 0x02
|
||||
FLV_DATA_TYPE_OBJECT = 0x03
|
||||
FLC_DATA_TYPE_CLIP = 0x04
|
||||
FLV_DATA_TYPE_REFERENCE = 0x07
|
||||
FLV_DATA_TYPE_ECMARRAY = 0x08
|
||||
FLV_DATA_TYPE_ENDOBJECT = 0x09
|
||||
FLV_DATA_TYPE_ARRAY = 0x0a
|
||||
FLV_DATA_TYPE_DATE = 0x0b
|
||||
FLV_DATA_TYPE_LONGSTRING = 0x0c
|
||||
|
||||
FLVINFO = {
|
||||
'creator': 'copyright',
|
||||
}
|
||||
|
||||
class FlashVideo(core.AVContainer):
|
||||
"""
|
||||
Experimental parser for Flash videos. It requires certain flags to
|
||||
be set to report video resolutions and in most cases it does not
|
||||
provide that information.
|
||||
"""
|
||||
table_mapping = { 'FLVINFO' : FLVINFO }
|
||||
|
||||
def __init__(self,file):
|
||||
core.AVContainer.__init__(self)
|
||||
self.mime = 'video/flv'
|
||||
self.type = 'Flash Video'
|
||||
data = file.read(13)
|
||||
if len(data) < 13 or struct.unpack('>3sBBII', data)[0] != 'FLV':
|
||||
raise ParseError()
|
||||
|
||||
for i in range(10):
|
||||
if self.audio and self.video:
|
||||
break
|
||||
data = file.read(11)
|
||||
if len(data) < 11:
|
||||
break
|
||||
chunk = struct.unpack('>BH4BI', data)
|
||||
size = (chunk[1] << 8) + chunk[2]
|
||||
|
||||
if chunk[0] == FLV_TAG_TYPE_AUDIO:
|
||||
flags = ord(file.read(1))
|
||||
if not self.audio:
|
||||
a = core.AudioStream()
|
||||
a.channels = (flags & FLV_AUDIO_CHANNEL_MASK) + 1
|
||||
srate = (flags & FLV_AUDIO_SAMPLERATE_MASK)
|
||||
a.samplerate = (44100 << (srate >> FLV_AUDIO_SAMPLERATE_OFFSET) >> 3)
|
||||
codec = (flags & FLV_AUDIO_CODECID_MASK) >> FLV_AUDIO_CODECID_OFFSET
|
||||
if codec < len(FLV_AUDIO_CODECID):
|
||||
a.codec = FLV_AUDIO_CODECID[codec]
|
||||
self.audio.append(a)
|
||||
|
||||
file.seek(size - 1, 1)
|
||||
|
||||
elif chunk[0] == FLV_TAG_TYPE_VIDEO:
|
||||
flags = ord(file.read(1))
|
||||
if not self.video:
|
||||
v = core.VideoStream()
|
||||
codec = (flags & FLV_VIDEO_CODECID_MASK) - 2
|
||||
if codec < len(FLV_VIDEO_CODECID):
|
||||
v.codec = FLV_VIDEO_CODECID[codec]
|
||||
# width and height are in the meta packet, but I have
|
||||
# no file with such a packet inside. So maybe we have
|
||||
# to decode some parts of the video.
|
||||
self.video.append(v)
|
||||
|
||||
file.seek(size - 1, 1)
|
||||
|
||||
elif chunk[0] == FLV_TAG_TYPE_META:
|
||||
log.info('metadata %s', str(chunk))
|
||||
metadata = file.read(size)
|
||||
try:
|
||||
while metadata:
|
||||
length, value = self._parse_value(metadata)
|
||||
if isinstance(value, dict):
|
||||
log.info('metadata: %s', value)
|
||||
if value.get('creator'):
|
||||
self.copyright = value.get('creator')
|
||||
if value.get('width'):
|
||||
self.width = value.get('width')
|
||||
if value.get('height'):
|
||||
self.height = value.get('height')
|
||||
if value.get('duration'):
|
||||
self.length = value.get('duration')
|
||||
self._appendtable('FLVINFO', value)
|
||||
if not length:
|
||||
# parse error
|
||||
break
|
||||
metadata = metadata[length:]
|
||||
except (IndexError, struct.error, TypeError):
|
||||
pass
|
||||
else:
|
||||
log.info('unkown %s', str(chunk))
|
||||
file.seek(size, 1)
|
||||
|
||||
file.seek(4, 1)
|
||||
|
||||
def _parse_value(self, data):
|
||||
"""
|
||||
Parse the next metadata value.
|
||||
"""
|
||||
if ord(data[0]) == FLV_DATA_TYPE_NUMBER:
|
||||
value = struct.unpack('>d', data[1:9])[0]
|
||||
return 9, value
|
||||
|
||||
if ord(data[0]) == FLV_DATA_TYPE_BOOL:
|
||||
return 2, bool(data[1])
|
||||
|
||||
if ord(data[0]) == FLV_DATA_TYPE_STRING:
|
||||
length = (ord(data[1]) << 8) + ord(data[2])
|
||||
return length + 3, data[3:length+3]
|
||||
|
||||
if ord(data[0]) == FLV_DATA_TYPE_ECMARRAY:
|
||||
init_length = len(data)
|
||||
num = struct.unpack('>I', data[1:5])[0]
|
||||
data = data[5:]
|
||||
result = {}
|
||||
for i in range(num):
|
||||
length = (ord(data[0]) << 8) + ord(data[1])
|
||||
key = data[2:length+2]
|
||||
data = data[length + 2:]
|
||||
length, value = self._parse_value(data)
|
||||
if not length:
|
||||
return 0, result
|
||||
result[key] = value
|
||||
data = data[length:]
|
||||
return init_length - len(data), result
|
||||
|
||||
log.info('unknown code: %x. Stop metadata parser', ord(data[0]))
|
||||
return 0, None
|
||||
|
||||
|
||||
Parser = FlashVideo
|
||||
@@ -0,0 +1,853 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2003-2006 Dirk Meyer <dischi@freevo.org>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
|
||||
import string
|
||||
import re
|
||||
import struct
|
||||
|
||||
__all__ = ['resolve']
|
||||
|
||||
|
||||
def resolve(code):
|
||||
"""
|
||||
Transform a twocc or fourcc code into a name. Returns a 2-tuple of (cc,
|
||||
codec) where both are strings and cc is a string in the form '0xXX' if it's
|
||||
a twocc, or 'ABCD' if it's a fourcc. If the given code is not a known
|
||||
twocc or fourcc, the return value will be (None, 'Unknown'), unless the
|
||||
code is otherwise a printable string in which case it will be returned as
|
||||
the codec.
|
||||
"""
|
||||
if isinstance(code, basestring):
|
||||
codec = u'Unknown'
|
||||
# Check for twocc
|
||||
if re.match(r'^0x[\da-f]{1,4}$', code, re.I):
|
||||
# Twocc in hex form
|
||||
return code, TWOCC.get(int(code, 16), codec)
|
||||
elif code.isdigit() and 0 <= int(code) <= 0xff:
|
||||
# Twocc in decimal form
|
||||
return hex(int(code)), TWOCC.get(int(code), codec)
|
||||
elif len(code) == 2:
|
||||
code = struct.unpack('H', code)[0]
|
||||
return hex(code), TWOCC.get(code, codec)
|
||||
elif len(code) != 4 and len([x for x in code if x not in string.printable]) == 0:
|
||||
# Code is a printable string.
|
||||
codec = unicode(code)
|
||||
|
||||
if code[:2] == 'MS' and code[2:].upper() in FOURCC:
|
||||
code = code[2:]
|
||||
|
||||
if code.upper() in FOURCC:
|
||||
return code.upper(), unicode(FOURCC[code.upper()])
|
||||
return None, codec
|
||||
elif isinstance(code, (int, long)):
|
||||
return hex(code), TWOCC.get(code, u'Unknown')
|
||||
|
||||
return None, u'Unknown'
|
||||
|
||||
|
||||
TWOCC = {
|
||||
0x0000: 'Unknown Wave Format',
|
||||
0x0001: 'PCM',
|
||||
0x0002: 'Microsoft ADPCM',
|
||||
0x0003: 'IEEE Float',
|
||||
0x0004: 'Compaq Computer VSELP',
|
||||
0x0005: 'IBM CVSD',
|
||||
0x0006: 'A-Law',
|
||||
0x0007: 'mu-Law',
|
||||
0x0008: 'Microsoft DTS',
|
||||
0x0009: 'Microsoft DRM',
|
||||
0x0010: 'OKI ADPCM',
|
||||
0x0011: 'Intel DVI/IMA ADPCM',
|
||||
0x0012: 'Videologic MediaSpace ADPCM',
|
||||
0x0013: 'Sierra Semiconductor ADPCM',
|
||||
0x0014: 'Antex Electronics G.723 ADPCM',
|
||||
0x0015: 'DSP Solutions DigiSTD',
|
||||
0x0016: 'DSP Solutions DigiFIX',
|
||||
0x0017: 'Dialogic OKI ADPCM',
|
||||
0x0018: 'MediaVision ADPCM',
|
||||
0x0019: 'Hewlett-Packard CU',
|
||||
0x0020: 'Yamaha ADPCM',
|
||||
0x0021: 'Speech Compression Sonarc',
|
||||
0x0022: 'DSP Group TrueSpeech',
|
||||
0x0023: 'Echo Speech EchoSC1',
|
||||
0x0024: 'Audiofile AF36',
|
||||
0x0025: 'Audio Processing Technology APTX',
|
||||
0x0026: 'AudioFile AF10',
|
||||
0x0027: 'Prosody 1612',
|
||||
0x0028: 'LRC',
|
||||
0x0030: 'Dolby AC2',
|
||||
0x0031: 'Microsoft GSM 6.10',
|
||||
0x0032: 'MSNAudio',
|
||||
0x0033: 'Antex Electronics ADPCME',
|
||||
0x0034: 'Control Resources VQLPC',
|
||||
0x0035: 'DSP Solutions DigiREAL',
|
||||
0x0036: 'DSP Solutions DigiADPCM',
|
||||
0x0037: 'Control Resources CR10',
|
||||
0x0038: 'Natural MicroSystems VBXADPCM',
|
||||
0x0039: 'Crystal Semiconductor IMA ADPCM',
|
||||
0x003A: 'EchoSC3',
|
||||
0x003B: 'Rockwell ADPCM',
|
||||
0x003C: 'Rockwell Digit LK',
|
||||
0x003D: 'Xebec',
|
||||
0x0040: 'Antex Electronics G.721 ADPCM',
|
||||
0x0041: 'G.728 CELP',
|
||||
0x0042: 'MSG723',
|
||||
0x0043: 'IBM AVC ADPCM',
|
||||
0x0045: 'ITU-T G.726 ADPCM',
|
||||
0x0050: 'MPEG 1, Layer 1,2',
|
||||
0x0052: 'RT24',
|
||||
0x0053: 'PAC',
|
||||
0x0055: 'MPEG Layer 3',
|
||||
0x0059: 'Lucent G.723',
|
||||
0x0060: 'Cirrus',
|
||||
0x0061: 'ESPCM',
|
||||
0x0062: 'Voxware',
|
||||
0x0063: 'Canopus Atrac',
|
||||
0x0064: 'G.726 ADPCM',
|
||||
0x0065: 'G.722 ADPCM',
|
||||
0x0066: 'DSAT',
|
||||
0x0067: 'DSAT Display',
|
||||
0x0069: 'Voxware Byte Aligned',
|
||||
0x0070: 'Voxware AC8',
|
||||
0x0071: 'Voxware AC10',
|
||||
0x0072: 'Voxware AC16',
|
||||
0x0073: 'Voxware AC20',
|
||||
0x0074: 'Voxware MetaVoice',
|
||||
0x0075: 'Voxware MetaSound',
|
||||
0x0076: 'Voxware RT29HW',
|
||||
0x0077: 'Voxware VR12',
|
||||
0x0078: 'Voxware VR18',
|
||||
0x0079: 'Voxware TQ40',
|
||||
0x0080: 'Softsound',
|
||||
0x0081: 'Voxware TQ60',
|
||||
0x0082: 'MSRT24',
|
||||
0x0083: 'G.729A',
|
||||
0x0084: 'MVI MV12',
|
||||
0x0085: 'DF G.726',
|
||||
0x0086: 'DF GSM610',
|
||||
0x0088: 'ISIAudio',
|
||||
0x0089: 'Onlive',
|
||||
0x0091: 'SBC24',
|
||||
0x0092: 'Dolby AC3 SPDIF',
|
||||
0x0093: 'MediaSonic G.723',
|
||||
0x0094: 'Aculab PLC Prosody 8KBPS',
|
||||
0x0097: 'ZyXEL ADPCM',
|
||||
0x0098: 'Philips LPCBB',
|
||||
0x0099: 'Packed',
|
||||
0x00A0: 'Malden Electronics PHONYTALK',
|
||||
0x00FF: 'AAC',
|
||||
0x0100: 'Rhetorex ADPCM',
|
||||
0x0101: 'IBM mu-law',
|
||||
0x0102: 'IBM A-law',
|
||||
0x0103: 'IBM AVC Adaptive Differential Pulse Code Modulation',
|
||||
0x0111: 'Vivo G.723',
|
||||
0x0112: 'Vivo Siren',
|
||||
0x0123: 'Digital G.723',
|
||||
0x0125: 'Sanyo LD ADPCM',
|
||||
0x0130: 'Sipro Lab Telecom ACELP.net',
|
||||
0x0131: 'Sipro Lab Telecom ACELP.4800',
|
||||
0x0132: 'Sipro Lab Telecom ACELP.8V3',
|
||||
0x0133: 'Sipro Lab Telecom ACELP.G.729',
|
||||
0x0134: 'Sipro Lab Telecom ACELP.G.729A',
|
||||
0x0135: 'Sipro Lab Telecom ACELP.KELVIN',
|
||||
0x0140: 'Windows Media Video V8',
|
||||
0x0150: 'Qualcomm PureVoice',
|
||||
0x0151: 'Qualcomm HalfRate',
|
||||
0x0155: 'Ring Zero Systems TUB GSM',
|
||||
0x0160: 'Windows Media Audio V1 / DivX audio (WMA)',
|
||||
0x0161: 'Windows Media Audio V7 / V8 / V9',
|
||||
0x0162: 'Windows Media Audio Professional V9',
|
||||
0x0163: 'Windows Media Audio Lossless V9',
|
||||
0x0170: 'UNISYS NAP ADPCM',
|
||||
0x0171: 'UNISYS NAP ULAW',
|
||||
0x0172: 'UNISYS NAP ALAW',
|
||||
0x0173: 'UNISYS NAP 16K',
|
||||
0x0200: 'Creative Labs ADPCM',
|
||||
0x0202: 'Creative Labs Fastspeech8',
|
||||
0x0203: 'Creative Labs Fastspeech10',
|
||||
0x0210: 'UHER Informatic ADPCM',
|
||||
0x0215: 'Ulead DV ACM',
|
||||
0x0216: 'Ulead DV ACM',
|
||||
0x0220: 'Quarterdeck',
|
||||
0x0230: 'I-link Worldwide ILINK VC',
|
||||
0x0240: 'Aureal Semiconductor RAW SPORT',
|
||||
0x0241: 'ESST AC3',
|
||||
0x0250: 'Interactive Products HSX',
|
||||
0x0251: 'Interactive Products RPELP',
|
||||
0x0260: 'Consistent Software CS2',
|
||||
0x0270: 'Sony ATRAC3 (SCX, same as MiniDisk LP2)',
|
||||
0x0300: 'Fujitsu FM Towns Snd',
|
||||
0x0400: 'BTV Digital',
|
||||
0x0401: 'Intel Music Coder (IMC)',
|
||||
0x0402: 'Ligos Indeo Audio',
|
||||
0x0450: 'QDesign Music',
|
||||
0x0680: 'VME VMPCM',
|
||||
0x0681: 'AT&T Labs TPC',
|
||||
0x0700: 'YMPEG Alpha',
|
||||
0x08AE: 'ClearJump LiteWave',
|
||||
0x1000: 'Olivetti GSM',
|
||||
0x1001: 'Olivetti ADPCM',
|
||||
0x1002: 'Olivetti CELP',
|
||||
0x1003: 'Olivetti SBC',
|
||||
0x1004: 'Olivetti OPR',
|
||||
0x1100: 'Lernout & Hauspie LH Codec',
|
||||
0x1101: 'Lernout & Hauspie CELP codec',
|
||||
0x1102: 'Lernout & Hauspie SBC codec',
|
||||
0x1103: 'Lernout & Hauspie SBC codec',
|
||||
0x1104: 'Lernout & Hauspie SBC codec',
|
||||
0x1400: 'Norris',
|
||||
0x1401: 'AT&T ISIAudio',
|
||||
0x1500: 'Soundspace Music Compression',
|
||||
0x181C: 'VoxWare RT24 speech codec',
|
||||
0x181E: 'Lucent elemedia AX24000P Music codec',
|
||||
0x1C07: 'Lucent SX8300P speech codec',
|
||||
0x1C0C: 'Lucent SX5363S G.723 compliant codec',
|
||||
0x1F03: 'CUseeMe DigiTalk (ex-Rocwell)',
|
||||
0x1FC4: 'NCT Soft ALF2CD ACM',
|
||||
0x2000: 'AC3',
|
||||
0x2001: 'Dolby DTS (Digital Theater System)',
|
||||
0x2002: 'RealAudio 1 / 2 14.4',
|
||||
0x2003: 'RealAudio 1 / 2 28.8',
|
||||
0x2004: 'RealAudio G2 / 8 Cook (low bitrate)',
|
||||
0x2005: 'RealAudio 3 / 4 / 5 Music (DNET)',
|
||||
0x2006: 'RealAudio 10 AAC (RAAC)',
|
||||
0x2007: 'RealAudio 10 AAC+ (RACP)',
|
||||
0x3313: 'makeAVIS',
|
||||
0x4143: 'Divio MPEG-4 AAC audio',
|
||||
0x434C: 'LEAD Speech',
|
||||
0x564C: 'LEAD Vorbis',
|
||||
0x674F: 'Ogg Vorbis (mode 1)',
|
||||
0x6750: 'Ogg Vorbis (mode 2)',
|
||||
0x6751: 'Ogg Vorbis (mode 3)',
|
||||
0x676F: 'Ogg Vorbis (mode 1+)',
|
||||
0x6770: 'Ogg Vorbis (mode 2+)',
|
||||
0x6771: 'Ogg Vorbis (mode 3+)',
|
||||
0x7A21: 'GSM-AMR (CBR, no SID)',
|
||||
0x7A22: 'GSM-AMR (VBR, including SID)',
|
||||
0xDFAC: 'DebugMode SonicFoundry Vegas FrameServer ACM Codec',
|
||||
0xF1AC: 'Free Lossless Audio Codec FLAC',
|
||||
0xFFFE: 'Extensible wave format',
|
||||
0xFFFF: 'development'
|
||||
}
|
||||
|
||||
|
||||
FOURCC = {
|
||||
'1978': 'A.M.Paredes predictor (LossLess)',
|
||||
'2VUY': 'Optibase VideoPump 8-bit 4:2:2 Component YCbCr',
|
||||
'3IV0': 'MPEG4-based codec 3ivx',
|
||||
'3IV1': '3ivx v1',
|
||||
'3IV2': '3ivx v2',
|
||||
'3IVD': 'FFmpeg DivX ;-) (MS MPEG-4 v3)',
|
||||
'3IVX': 'MPEG4-based codec 3ivx',
|
||||
'8BPS': 'Apple QuickTime Planar RGB with Alpha-channel',
|
||||
'AAS4': 'Autodesk Animator codec (RLE)',
|
||||
'AASC': 'Autodesk Animator',
|
||||
'ABYR': 'Kensington ABYR',
|
||||
'ACTL': 'Streambox ACT-L2',
|
||||
'ADV1': 'Loronix WaveCodec',
|
||||
'ADVJ': 'Avid M-JPEG Avid Technology Also known as AVRn',
|
||||
'AEIK': 'Intel Indeo Video 3.2',
|
||||
'AEMI': 'Array VideoONE MPEG1-I Capture',
|
||||
'AFLC': 'Autodesk Animator FLC',
|
||||
'AFLI': 'Autodesk Animator FLI',
|
||||
'AHDV': 'CineForm 10-bit Visually Perfect HD',
|
||||
'AJPG': '22fps JPEG-based codec for digital cameras',
|
||||
'AMPG': 'Array VideoONE MPEG',
|
||||
'ANIM': 'Intel RDX (ANIM)',
|
||||
'AP41': 'AngelPotion Definitive',
|
||||
'AP42': 'AngelPotion Definitive',
|
||||
'ASLC': 'AlparySoft Lossless Codec',
|
||||
'ASV1': 'Asus Video v1',
|
||||
'ASV2': 'Asus Video v2',
|
||||
'ASVX': 'Asus Video 2.0 (audio)',
|
||||
'ATM4': 'Ahead Nero Digital MPEG-4 Codec',
|
||||
'AUR2': 'Aura 2 Codec - YUV 4:2:2',
|
||||
'AURA': 'Aura 1 Codec - YUV 4:1:1',
|
||||
'AV1X': 'Avid 1:1x (Quick Time)',
|
||||
'AVC1': 'H.264 AVC',
|
||||
'AVD1': 'Avid DV (Quick Time)',
|
||||
'AVDJ': 'Avid Meridien JFIF with Alpha-channel',
|
||||
'AVDN': 'Avid DNxHD (Quick Time)',
|
||||
'AVDV': 'Avid DV',
|
||||
'AVI1': 'MainConcept Motion JPEG Codec',
|
||||
'AVI2': 'MainConcept Motion JPEG Codec',
|
||||
'AVID': 'Avid Motion JPEG',
|
||||
'AVIS': 'Wrapper for AviSynth',
|
||||
'AVMP': 'Avid IMX (Quick Time)',
|
||||
'AVR ': 'Avid ABVB/NuVista MJPEG with Alpha-channel',
|
||||
'AVRN': 'Avid Motion JPEG',
|
||||
'AVUI': 'Avid Meridien Uncompressed with Alpha-channel',
|
||||
'AVUP': 'Avid 10bit Packed (Quick Time)',
|
||||
'AYUV': '4:4:4 YUV (AYUV)',
|
||||
'AZPR': 'Quicktime Apple Video',
|
||||
'AZRP': 'Quicktime Apple Video',
|
||||
'BGR ': 'Uncompressed BGR32 8:8:8:8',
|
||||
'BGR(15)': 'Uncompressed BGR15 5:5:5',
|
||||
'BGR(16)': 'Uncompressed BGR16 5:6:5',
|
||||
'BGR(24)': 'Uncompressed BGR24 8:8:8',
|
||||
'BHIV': 'BeHere iVideo',
|
||||
'BINK': 'RAD Game Tools Bink Video',
|
||||
'BIT ': 'BI_BITFIELDS (Raw RGB)',
|
||||
'BITM': 'Microsoft H.261',
|
||||
'BLOX': 'Jan Jezabek BLOX MPEG Codec',
|
||||
'BLZ0': 'DivX for Blizzard Decoder Filter',
|
||||
'BT20': 'Conexant Prosumer Video',
|
||||
'BTCV': 'Conexant Composite Video Codec',
|
||||
'BTVC': 'Conexant Composite Video',
|
||||
'BW00': 'BergWave (Wavelet)',
|
||||
'BW10': 'Data Translation Broadway MPEG Capture',
|
||||
'BXBG': 'BOXX BGR',
|
||||
'BXRG': 'BOXX RGB',
|
||||
'BXY2': 'BOXX 10-bit YUV',
|
||||
'BXYV': 'BOXX YUV',
|
||||
'CC12': 'Intel YUV12',
|
||||
'CDV5': 'Canopus SD50/DVHD',
|
||||
'CDVC': 'Canopus DV',
|
||||
'CDVH': 'Canopus SD50/DVHD',
|
||||
'CFCC': 'Digital Processing Systems DPS Perception',
|
||||
'CFHD': 'CineForm 10-bit Visually Perfect HD',
|
||||
'CGDI': 'Microsoft Office 97 Camcorder Video',
|
||||
'CHAM': 'Winnov Caviara Champagne',
|
||||
'CJPG': 'Creative WebCam JPEG',
|
||||
'CLJR': 'Cirrus Logic YUV 4 pixels',
|
||||
'CLLC': 'Canopus LossLess',
|
||||
'CLPL': 'YV12',
|
||||
'CMYK': 'Common Data Format in Printing',
|
||||
'COL0': 'FFmpeg DivX ;-) (MS MPEG-4 v3)',
|
||||
'COL1': 'FFmpeg DivX ;-) (MS MPEG-4 v3)',
|
||||
'CPLA': 'Weitek 4:2:0 YUV Planar',
|
||||
'CRAM': 'Microsoft Video 1 (CRAM)',
|
||||
'CSCD': 'RenderSoft CamStudio lossless Codec',
|
||||
'CTRX': 'Citrix Scalable Video Codec',
|
||||
'CUVC': 'Canopus HQ',
|
||||
'CVID': 'Radius Cinepak',
|
||||
'CWLT': 'Microsoft Color WLT DIB',
|
||||
'CYUV': 'Creative Labs YUV',
|
||||
'CYUY': 'ATI YUV',
|
||||
'D261': 'H.261',
|
||||
'D263': 'H.263',
|
||||
'DAVC': 'Dicas MPEGable H.264/MPEG-4 AVC base profile codec',
|
||||
'DC25': 'MainConcept ProDV Codec',
|
||||
'DCAP': 'Pinnacle DV25 Codec',
|
||||
'DCL1': 'Data Connection Conferencing Codec',
|
||||
'DCT0': 'WniWni Codec',
|
||||
'DFSC': 'DebugMode FrameServer VFW Codec',
|
||||
'DIB ': 'Full Frames (Uncompressed)',
|
||||
'DIV1': 'FFmpeg-4 V1 (hacked MS MPEG-4 V1)',
|
||||
'DIV2': 'MS MPEG-4 V2',
|
||||
'DIV3': 'DivX v3 MPEG-4 Low-Motion',
|
||||
'DIV4': 'DivX v3 MPEG-4 Fast-Motion',
|
||||
'DIV5': 'DIV5',
|
||||
'DIV6': 'DivX MPEG-4',
|
||||
'DIVX': 'DivX',
|
||||
'DM4V': 'Dicas MPEGable MPEG-4',
|
||||
'DMB1': 'Matrox Rainbow Runner hardware MJPEG',
|
||||
'DMB2': 'Paradigm MJPEG',
|
||||
'DMK2': 'ViewSonic V36 PDA Video',
|
||||
'DP02': 'DynaPel MPEG-4',
|
||||
'DPS0': 'DPS Reality Motion JPEG',
|
||||
'DPSC': 'DPS PAR Motion JPEG',
|
||||
'DRWX': 'Pinnacle DV25 Codec',
|
||||
'DSVD': 'DSVD',
|
||||
'DTMT': 'Media-100 Codec',
|
||||
'DTNT': 'Media-100 Codec',
|
||||
'DUCK': 'Duck True Motion 1.0',
|
||||
'DV10': 'BlueFish444 (lossless RGBA, YUV 10-bit)',
|
||||
'DV25': 'Matrox DVCPRO codec',
|
||||
'DV50': 'Matrox DVCPRO50 codec',
|
||||
'DVAN': 'DVAN',
|
||||
'DVC ': 'Apple QuickTime DV (DVCPRO NTSC)',
|
||||
'DVCP': 'Apple QuickTime DV (DVCPRO PAL)',
|
||||
'DVCS': 'MainConcept DV Codec',
|
||||
'DVE2': 'InSoft DVE-2 Videoconferencing',
|
||||
'DVH1': 'Pinnacle DVHD100',
|
||||
'DVHD': 'DV 1125 lines at 30.00 Hz or 1250 lines at 25.00 Hz',
|
||||
'DVIS': 'VSYNC DualMoon Iris DV codec',
|
||||
'DVL ': 'Radius SoftDV 16:9 NTSC',
|
||||
'DVLP': 'Radius SoftDV 16:9 PAL',
|
||||
'DVMA': 'Darim Vision DVMPEG',
|
||||
'DVOR': 'BlueFish444 (lossless RGBA, YUV 10-bit)',
|
||||
'DVPN': 'Apple QuickTime DV (DV NTSC)',
|
||||
'DVPP': 'Apple QuickTime DV (DV PAL)',
|
||||
'DVR1': 'TARGA2000 Codec',
|
||||
'DVRS': 'VSYNC DualMoon Iris DV codec',
|
||||
'DVSD': 'DV',
|
||||
'DVSL': 'DV compressed in SD (SDL)',
|
||||
'DVX1': 'DVX1000SP Video Decoder',
|
||||
'DVX2': 'DVX2000S Video Decoder',
|
||||
'DVX3': 'DVX3000S Video Decoder',
|
||||
'DX50': 'DivX v5',
|
||||
'DXGM': 'Electronic Arts Game Video codec',
|
||||
'DXSB': 'DivX Subtitles Codec',
|
||||
'DXT1': 'Microsoft DirectX Compressed Texture (DXT1)',
|
||||
'DXT2': 'Microsoft DirectX Compressed Texture (DXT2)',
|
||||
'DXT3': 'Microsoft DirectX Compressed Texture (DXT3)',
|
||||
'DXT4': 'Microsoft DirectX Compressed Texture (DXT4)',
|
||||
'DXT5': 'Microsoft DirectX Compressed Texture (DXT5)',
|
||||
'DXTC': 'Microsoft DirectX Compressed Texture (DXTC)',
|
||||
'DXTN': 'Microsoft DirectX Compressed Texture (DXTn)',
|
||||
'EKQ0': 'Elsa EKQ0',
|
||||
'ELK0': 'Elsa ELK0',
|
||||
'EM2V': 'Etymonix MPEG-2 I-frame',
|
||||
'EQK0': 'Elsa graphics card quick codec',
|
||||
'ESCP': 'Eidos Escape',
|
||||
'ETV1': 'eTreppid Video ETV1',
|
||||
'ETV2': 'eTreppid Video ETV2',
|
||||
'ETVC': 'eTreppid Video ETVC',
|
||||
'FFDS': 'FFDShow supported',
|
||||
'FFV1': 'FFDShow supported',
|
||||
'FFVH': 'FFVH codec',
|
||||
'FLIC': 'Autodesk FLI/FLC Animation',
|
||||
'FLJP': 'D-Vision Field Encoded Motion JPEG',
|
||||
'FLV1': 'FLV1 codec',
|
||||
'FMJP': 'D-Vision fieldbased ISO MJPEG',
|
||||
'FRLE': 'SoftLab-NSK Y16 + Alpha RLE',
|
||||
'FRWA': 'SoftLab-Nsk Forward Motion JPEG w/ alpha channel',
|
||||
'FRWD': 'SoftLab-Nsk Forward Motion JPEG',
|
||||
'FRWT': 'SoftLab-NSK Vision Forward Motion JPEG with Alpha-channel',
|
||||
'FRWU': 'SoftLab-NSK Vision Forward Uncompressed',
|
||||
'FVF1': 'Iterated Systems Fractal Video Frame',
|
||||
'FVFW': 'ff MPEG-4 based on XviD codec',
|
||||
'GEPJ': 'White Pine (ex Paradigm Matrix) Motion JPEG Codec',
|
||||
'GJPG': 'Grand Tech GT891x Codec',
|
||||
'GLCC': 'GigaLink AV Capture codec',
|
||||
'GLZW': 'Motion LZW',
|
||||
'GPEG': 'Motion JPEG',
|
||||
'GPJM': 'Pinnacle ReelTime MJPEG Codec',
|
||||
'GREY': 'Apparently a duplicate of Y800',
|
||||
'GWLT': 'Microsoft Greyscale WLT DIB',
|
||||
'H260': 'H.260',
|
||||
'H261': 'H.261',
|
||||
'H262': 'H.262',
|
||||
'H263': 'H.263',
|
||||
'H264': 'H.264 AVC',
|
||||
'H265': 'H.265',
|
||||
'H266': 'H.266',
|
||||
'H267': 'H.267',
|
||||
'H268': 'H.268',
|
||||
'H269': 'H.269',
|
||||
'HD10': 'BlueFish444 (lossless RGBA, YUV 10-bit)',
|
||||
'HDX4': 'Jomigo HDX4',
|
||||
'HFYU': 'Huffman Lossless Codec',
|
||||
'HMCR': 'Rendition Motion Compensation Format (HMCR)',
|
||||
'HMRR': 'Rendition Motion Compensation Format (HMRR)',
|
||||
'I263': 'Intel ITU H.263 Videoconferencing (i263)',
|
||||
'I420': 'Intel Indeo 4',
|
||||
'IAN ': 'Intel RDX',
|
||||
'ICLB': 'InSoft CellB Videoconferencing',
|
||||
'IDM0': 'IDM Motion Wavelets 2.0',
|
||||
'IF09': 'Microsoft H.261',
|
||||
'IGOR': 'Power DVD',
|
||||
'IJPG': 'Intergraph JPEG',
|
||||
'ILVC': 'Intel Layered Video',
|
||||
'ILVR': 'ITU-T H.263+',
|
||||
'IMC1': 'IMC1',
|
||||
'IMC2': 'IMC2',
|
||||
'IMC3': 'IMC3',
|
||||
'IMC4': 'IMC4',
|
||||
'IMJG': 'Accom SphereOUS MJPEG with Alpha-channel',
|
||||
'IPDV': 'I-O Data Device Giga AVI DV Codec',
|
||||
'IPJ2': 'Image Power JPEG2000',
|
||||
'IR21': 'Intel Indeo 2.1',
|
||||
'IRAW': 'Intel YUV Uncompressed',
|
||||
'IUYV': 'Interlaced version of UYVY (line order 0,2,4 then 1,3,5 etc)',
|
||||
'IV30': 'Ligos Indeo 3.0',
|
||||
'IV31': 'Ligos Indeo 3.1',
|
||||
'IV32': 'Ligos Indeo 3.2',
|
||||
'IV33': 'Ligos Indeo 3.3',
|
||||
'IV34': 'Ligos Indeo 3.4',
|
||||
'IV35': 'Ligos Indeo 3.5',
|
||||
'IV36': 'Ligos Indeo 3.6',
|
||||
'IV37': 'Ligos Indeo 3.7',
|
||||
'IV38': 'Ligos Indeo 3.8',
|
||||
'IV39': 'Ligos Indeo 3.9',
|
||||
'IV40': 'Ligos Indeo Interactive 4.0',
|
||||
'IV41': 'Ligos Indeo Interactive 4.1',
|
||||
'IV42': 'Ligos Indeo Interactive 4.2',
|
||||
'IV43': 'Ligos Indeo Interactive 4.3',
|
||||
'IV44': 'Ligos Indeo Interactive 4.4',
|
||||
'IV45': 'Ligos Indeo Interactive 4.5',
|
||||
'IV46': 'Ligos Indeo Interactive 4.6',
|
||||
'IV47': 'Ligos Indeo Interactive 4.7',
|
||||
'IV48': 'Ligos Indeo Interactive 4.8',
|
||||
'IV49': 'Ligos Indeo Interactive 4.9',
|
||||
'IV50': 'Ligos Indeo Interactive 5.0',
|
||||
'IY41': 'Interlaced version of Y41P (line order 0,2,4,...,1,3,5...)',
|
||||
'IYU1': '12 bit format used in mode 2 of the IEEE 1394 Digital Camera 1.04 spec',
|
||||
'IYU2': '24 bit format used in mode 2 of the IEEE 1394 Digital Camera 1.04 spec',
|
||||
'IYUV': 'Intel Indeo iYUV 4:2:0',
|
||||
'JBYR': 'Kensington JBYR',
|
||||
'JFIF': 'Motion JPEG (FFmpeg)',
|
||||
'JPEG': 'Still Image JPEG DIB',
|
||||
'JPG ': 'JPEG compressed',
|
||||
'JPGL': 'Webcam JPEG Light',
|
||||
'KMVC': 'Karl Morton\'s Video Codec',
|
||||
'KPCD': 'Kodak Photo CD',
|
||||
'L261': 'Lead Technologies H.261',
|
||||
'L263': 'Lead Technologies H.263',
|
||||
'LAGS': 'Lagarith LossLess',
|
||||
'LBYR': 'Creative WebCam codec',
|
||||
'LCMW': 'Lead Technologies Motion CMW Codec',
|
||||
'LCW2': 'LEADTools MCMW 9Motion Wavelet)',
|
||||
'LEAD': 'LEAD Video Codec',
|
||||
'LGRY': 'Lead Technologies Grayscale Image',
|
||||
'LJ2K': 'LEADTools JPEG2000',
|
||||
'LJPG': 'LEAD MJPEG Codec',
|
||||
'LMP2': 'LEADTools MPEG2',
|
||||
'LOCO': 'LOCO Lossless Codec',
|
||||
'LSCR': 'LEAD Screen Capture',
|
||||
'LSVM': 'Vianet Lighting Strike Vmail (Streaming)',
|
||||
'LZO1': 'LZO compressed (lossless codec)',
|
||||
'M261': 'Microsoft H.261',
|
||||
'M263': 'Microsoft H.263',
|
||||
'M4CC': 'ESS MPEG4 Divio codec',
|
||||
'M4S2': 'Microsoft MPEG-4 (M4S2)',
|
||||
'MC12': 'ATI Motion Compensation Format (MC12)',
|
||||
'MC24': 'MainConcept Motion JPEG Codec',
|
||||
'MCAM': 'ATI Motion Compensation Format (MCAM)',
|
||||
'MCZM': 'Theory MicroCosm Lossless 64bit RGB with Alpha-channel',
|
||||
'MDVD': 'Alex MicroDVD Video (hacked MS MPEG-4)',
|
||||
'MDVF': 'Pinnacle DV/DV50/DVHD100',
|
||||
'MHFY': 'A.M.Paredes mhuffyYUV (LossLess)',
|
||||
'MJ2C': 'Morgan Multimedia Motion JPEG2000',
|
||||
'MJPA': 'Pinnacle ReelTime MJPG hardware codec',
|
||||
'MJPB': 'Motion JPEG codec',
|
||||
'MJPG': 'Motion JPEG DIB',
|
||||
'MJPX': 'Pegasus PICVideo Motion JPEG',
|
||||
'MMES': 'Matrox MPEG-2 I-frame',
|
||||
'MNVD': 'MindBend MindVid LossLess',
|
||||
'MP2A': 'MPEG-2 Audio',
|
||||
'MP2T': 'MPEG-2 Transport Stream',
|
||||
'MP2V': 'MPEG-2 Video',
|
||||
'MP41': 'Microsoft MPEG-4 V1 (enhansed H263)',
|
||||
'MP42': 'Microsoft MPEG-4 (low-motion)',
|
||||
'MP43': 'Microsoft MPEG-4 (fast-motion)',
|
||||
'MP4A': 'MPEG-4 Audio',
|
||||
'MP4S': 'Microsoft MPEG-4 (MP4S)',
|
||||
'MP4T': 'MPEG-4 Transport Stream',
|
||||
'MP4V': 'Apple QuickTime MPEG-4 native',
|
||||
'MPEG': 'MPEG-1',
|
||||
'MPG1': 'FFmpeg-1',
|
||||
'MPG2': 'FFmpeg-1',
|
||||
'MPG3': 'Same as Low motion DivX MPEG-4',
|
||||
'MPG4': 'Microsoft MPEG-4 Video High Speed Compressor',
|
||||
'MPGI': 'Sigma Designs MPEG',
|
||||
'MPNG': 'Motion PNG codec',
|
||||
'MRCA': 'Martin Regen Codec',
|
||||
'MRLE': 'Run Length Encoding',
|
||||
'MSS1': 'Windows Screen Video',
|
||||
'MSS2': 'Windows Media 9',
|
||||
'MSUC': 'MSU LossLess',
|
||||
'MSVC': 'Microsoft Video 1',
|
||||
'MSZH': 'Lossless codec (ZIP compression)',
|
||||
'MTGA': 'Motion TGA images (24, 32 bpp)',
|
||||
'MTX1': 'Matrox MTX1',
|
||||
'MTX2': 'Matrox MTX2',
|
||||
'MTX3': 'Matrox MTX3',
|
||||
'MTX4': 'Matrox MTX4',
|
||||
'MTX5': 'Matrox MTX5',
|
||||
'MTX6': 'Matrox MTX6',
|
||||
'MTX7': 'Matrox MTX7',
|
||||
'MTX8': 'Matrox MTX8',
|
||||
'MTX9': 'Matrox MTX9',
|
||||
'MV12': 'MV12',
|
||||
'MVI1': 'Motion Pixels MVI',
|
||||
'MVI2': 'Motion Pixels MVI',
|
||||
'MWV1': 'Aware Motion Wavelets',
|
||||
'MYUV': 'Media-100 844/X Uncompressed',
|
||||
'NAVI': 'nAVI',
|
||||
'NDIG': 'Ahead Nero Digital MPEG-4 Codec',
|
||||
'NHVU': 'NVidia Texture Format (GEForce 3)',
|
||||
'NO16': 'Theory None16 64bit uncompressed RAW',
|
||||
'NT00': 'NewTek LigtWave HDTV YUV with Alpha-channel',
|
||||
'NTN1': 'Nogatech Video Compression 1',
|
||||
'NTN2': 'Nogatech Video Compression 2 (GrabBee hardware coder)',
|
||||
'NUV1': 'NuppelVideo',
|
||||
'NV12': '8-bit Y plane followed by an interleaved U/V plane with 2x2 subsampling',
|
||||
'NV21': 'As NV12 with U and V reversed in the interleaved plane',
|
||||
'NVDS': 'nVidia Texture Format',
|
||||
'NVHS': 'NVidia Texture Format (GEForce 3)',
|
||||
'NVS0': 'nVidia GeForce Texture',
|
||||
'NVS1': 'nVidia GeForce Texture',
|
||||
'NVS2': 'nVidia GeForce Texture',
|
||||
'NVS3': 'nVidia GeForce Texture',
|
||||
'NVS4': 'nVidia GeForce Texture',
|
||||
'NVS5': 'nVidia GeForce Texture',
|
||||
'NVT0': 'nVidia GeForce Texture',
|
||||
'NVT1': 'nVidia GeForce Texture',
|
||||
'NVT2': 'nVidia GeForce Texture',
|
||||
'NVT3': 'nVidia GeForce Texture',
|
||||
'NVT4': 'nVidia GeForce Texture',
|
||||
'NVT5': 'nVidia GeForce Texture',
|
||||
'PDVC': 'I-O Data Device Digital Video Capture DV codec',
|
||||
'PGVV': 'Radius Video Vision',
|
||||
'PHMO': 'IBM Photomotion',
|
||||
'PIM1': 'Pegasus Imaging',
|
||||
'PIM2': 'Pegasus Imaging',
|
||||
'PIMJ': 'Pegasus Imaging Lossless JPEG',
|
||||
'PIXL': 'MiroVideo XL (Motion JPEG)',
|
||||
'PNG ': 'Apple PNG',
|
||||
'PNG1': 'Corecodec.org CorePNG Codec',
|
||||
'PVEZ': 'Horizons Technology PowerEZ',
|
||||
'PVMM': 'PacketVideo Corporation MPEG-4',
|
||||
'PVW2': 'Pegasus Imaging Wavelet Compression',
|
||||
'PVWV': 'Pegasus Imaging Wavelet 2000',
|
||||
'PXLT': 'Apple Pixlet (Wavelet)',
|
||||
'Q1.0': 'Q-Team QPEG 1.0 (www.q-team.de)',
|
||||
'Q1.1': 'Q-Team QPEG 1.1 (www.q-team.de)',
|
||||
'QDGX': 'Apple QuickDraw GX',
|
||||
'QPEG': 'Q-Team QPEG 1.0',
|
||||
'QPEQ': 'Q-Team QPEG 1.1',
|
||||
'R210': 'BlackMagic YUV (Quick Time)',
|
||||
'R411': 'Radius DV NTSC YUV',
|
||||
'R420': 'Radius DV PAL YUV',
|
||||
'RAVI': 'GroupTRON ReferenceAVI codec (dummy for MPEG compressor)',
|
||||
'RAV_': 'GroupTRON ReferenceAVI codec (dummy for MPEG compressor)',
|
||||
'RAW ': 'Full Frames (Uncompressed)',
|
||||
'RGB ': 'Full Frames (Uncompressed)',
|
||||
'RGB(15)': 'Uncompressed RGB15 5:5:5',
|
||||
'RGB(16)': 'Uncompressed RGB16 5:6:5',
|
||||
'RGB(24)': 'Uncompressed RGB24 8:8:8',
|
||||
'RGB1': 'Uncompressed RGB332 3:3:2',
|
||||
'RGBA': 'Raw RGB with alpha',
|
||||
'RGBO': 'Uncompressed RGB555 5:5:5',
|
||||
'RGBP': 'Uncompressed RGB565 5:6:5',
|
||||
'RGBQ': 'Uncompressed RGB555X 5:5:5 BE',
|
||||
'RGBR': 'Uncompressed RGB565X 5:6:5 BE',
|
||||
'RGBT': 'Computer Concepts 32-bit support',
|
||||
'RL4 ': 'RLE 4bpp RGB',
|
||||
'RL8 ': 'RLE 8bpp RGB',
|
||||
'RLE ': 'Microsoft Run Length Encoder',
|
||||
'RLE4': 'Run Length Encoded 4',
|
||||
'RLE8': 'Run Length Encoded 8',
|
||||
'RMP4': 'REALmagic MPEG-4 Video Codec',
|
||||
'ROQV': 'Id RoQ File Video Decoder',
|
||||
'RPZA': 'Apple Video 16 bit "road pizza"',
|
||||
'RT21': 'Intel Real Time Video 2.1',
|
||||
'RTV0': 'NewTek VideoToaster',
|
||||
'RUD0': 'Rududu video codec',
|
||||
'RV10': 'RealVideo codec',
|
||||
'RV13': 'RealVideo codec',
|
||||
'RV20': 'RealVideo G2',
|
||||
'RV30': 'RealVideo 8',
|
||||
'RV40': 'RealVideo 9',
|
||||
'RVX ': 'Intel RDX (RVX )',
|
||||
'S263': 'Sorenson Vision H.263',
|
||||
'S422': 'Tekram VideoCap C210 YUV 4:2:2',
|
||||
'SAMR': 'Adaptive Multi-Rate (AMR) audio codec',
|
||||
'SAN3': 'MPEG-4 codec (direct copy of DivX 3.11a)',
|
||||
'SDCC': 'Sun Communication Digital Camera Codec',
|
||||
'SEDG': 'Samsung MPEG-4 codec',
|
||||
'SFMC': 'CrystalNet Surface Fitting Method',
|
||||
'SHR0': 'BitJazz SheerVideo',
|
||||
'SHR1': 'BitJazz SheerVideo',
|
||||
'SHR2': 'BitJazz SheerVideo',
|
||||
'SHR3': 'BitJazz SheerVideo',
|
||||
'SHR4': 'BitJazz SheerVideo',
|
||||
'SHR5': 'BitJazz SheerVideo',
|
||||
'SHR6': 'BitJazz SheerVideo',
|
||||
'SHR7': 'BitJazz SheerVideo',
|
||||
'SJPG': 'CUseeMe Networks Codec',
|
||||
'SL25': 'SoftLab-NSK DVCPRO',
|
||||
'SL50': 'SoftLab-NSK DVCPRO50',
|
||||
'SLDV': 'SoftLab-NSK Forward DV Draw codec',
|
||||
'SLIF': 'SoftLab-NSK MPEG2 I-frames',
|
||||
'SLMJ': 'SoftLab-NSK Forward MJPEG',
|
||||
'SMC ': 'Apple Graphics (SMC) codec (256 color)',
|
||||
'SMSC': 'Radius SMSC',
|
||||
'SMSD': 'Radius SMSD',
|
||||
'SMSV': 'WorldConnect Wavelet Video',
|
||||
'SNOW': 'SNOW codec',
|
||||
'SP40': 'SunPlus YUV',
|
||||
'SP44': 'SunPlus Aiptek MegaCam Codec',
|
||||
'SP53': 'SunPlus Aiptek MegaCam Codec',
|
||||
'SP54': 'SunPlus Aiptek MegaCam Codec',
|
||||
'SP55': 'SunPlus Aiptek MegaCam Codec',
|
||||
'SP56': 'SunPlus Aiptek MegaCam Codec',
|
||||
'SP57': 'SunPlus Aiptek MegaCam Codec',
|
||||
'SP58': 'SunPlus Aiptek MegaCam Codec',
|
||||
'SPIG': 'Radius Spigot',
|
||||
'SPLC': 'Splash Studios ACM Audio Codec',
|
||||
'SPRK': 'Sorenson Spark',
|
||||
'SQZ2': 'Microsoft VXTreme Video Codec V2',
|
||||
'STVA': 'ST CMOS Imager Data (Bayer)',
|
||||
'STVB': 'ST CMOS Imager Data (Nudged Bayer)',
|
||||
'STVC': 'ST CMOS Imager Data (Bunched)',
|
||||
'STVX': 'ST CMOS Imager Data (Extended CODEC Data Format)',
|
||||
'STVY': 'ST CMOS Imager Data (Extended CODEC Data Format with Correction Data)',
|
||||
'SV10': 'Sorenson Video R1',
|
||||
'SVQ1': 'Sorenson Video R3',
|
||||
'SVQ3': 'Sorenson Video 3 (Apple Quicktime 5)',
|
||||
'SWC1': 'MainConcept Motion JPEG Codec',
|
||||
'T420': 'Toshiba YUV 4:2:0',
|
||||
'TGA ': 'Apple TGA (with Alpha-channel)',
|
||||
'THEO': 'FFVFW Supported Codec',
|
||||
'TIFF': 'Apple TIFF (with Alpha-channel)',
|
||||
'TIM2': 'Pinnacle RAL DVI',
|
||||
'TLMS': 'TeraLogic Motion Intraframe Codec (TLMS)',
|
||||
'TLST': 'TeraLogic Motion Intraframe Codec (TLST)',
|
||||
'TM20': 'Duck TrueMotion 2.0',
|
||||
'TM2A': 'Duck TrueMotion Archiver 2.0',
|
||||
'TM2X': 'Duck TrueMotion 2X',
|
||||
'TMIC': 'TeraLogic Motion Intraframe Codec (TMIC)',
|
||||
'TMOT': 'Horizons Technology TrueMotion S',
|
||||
'TR20': 'Duck TrueMotion RealTime 2.0',
|
||||
'TRLE': 'Akula Alpha Pro Custom AVI (LossLess)',
|
||||
'TSCC': 'TechSmith Screen Capture Codec',
|
||||
'TV10': 'Tecomac Low-Bit Rate Codec',
|
||||
'TVJP': 'TrueVision Field Encoded Motion JPEG',
|
||||
'TVMJ': 'Truevision TARGA MJPEG Hardware Codec',
|
||||
'TY0N': 'Trident TY0N',
|
||||
'TY2C': 'Trident TY2C',
|
||||
'TY2N': 'Trident TY2N',
|
||||
'U263': 'UB Video StreamForce H.263',
|
||||
'U<Y ': 'Discreet UC YUV 4:2:2:4 10 bit',
|
||||
'U<YA': 'Discreet UC YUV 4:2:2:4 10 bit (with Alpha-channel)',
|
||||
'UCOD': 'eMajix.com ClearVideo',
|
||||
'ULTI': 'IBM Ultimotion',
|
||||
'UMP4': 'UB Video MPEG 4',
|
||||
'UYNV': 'UYVY',
|
||||
'UYVP': 'YCbCr 4:2:2',
|
||||
'UYVU': 'SoftLab-NSK Forward YUV codec',
|
||||
'UYVY': 'UYVY 4:2:2 byte ordering',
|
||||
'V210': 'Optibase VideoPump 10-bit 4:2:2 Component YCbCr',
|
||||
'V261': 'Lucent VX2000S',
|
||||
'V422': '24 bit YUV 4:2:2 Format',
|
||||
'V655': '16 bit YUV 4:2:2 Format',
|
||||
'VBLE': 'MarcFD VBLE Lossless Codec',
|
||||
'VCR1': 'ATI VCR 1.0',
|
||||
'VCR2': 'ATI VCR 2.0',
|
||||
'VCR3': 'ATI VCR 3.0',
|
||||
'VCR4': 'ATI VCR 4.0',
|
||||
'VCR5': 'ATI VCR 5.0',
|
||||
'VCR6': 'ATI VCR 6.0',
|
||||
'VCR7': 'ATI VCR 7.0',
|
||||
'VCR8': 'ATI VCR 8.0',
|
||||
'VCR9': 'ATI VCR 9.0',
|
||||
'VDCT': 'Video Maker Pro DIB',
|
||||
'VDOM': 'VDOnet VDOWave',
|
||||
'VDOW': 'VDOnet VDOLive (H.263)',
|
||||
'VDST': 'VirtualDub remote frameclient ICM driver',
|
||||
'VDTZ': 'Darim Vison VideoTizer YUV',
|
||||
'VGPX': 'VGPixel Codec',
|
||||
'VIDM': 'DivX 5.0 Pro Supported Codec',
|
||||
'VIDS': 'YUV 4:2:2 CCIR 601 for V422',
|
||||
'VIFP': 'VIFP',
|
||||
'VIV1': 'Vivo H.263',
|
||||
'VIV2': 'Vivo H.263',
|
||||
'VIVO': 'Vivo H.263 v2.00',
|
||||
'VIXL': 'Miro Video XL',
|
||||
'VLV1': 'Videologic VLCAP.DRV',
|
||||
'VP30': 'On2 VP3.0',
|
||||
'VP31': 'On2 VP3.1',
|
||||
'VP40': 'On2 TrueCast VP4',
|
||||
'VP50': 'On2 TrueCast VP5',
|
||||
'VP60': 'On2 TrueCast VP6',
|
||||
'VP61': 'On2 TrueCast VP6.1',
|
||||
'VP62': 'On2 TrueCast VP6.2',
|
||||
'VP70': 'On2 TrueMotion VP7',
|
||||
'VQC1': 'Vector-quantised codec 1',
|
||||
'VQC2': 'Vector-quantised codec 2',
|
||||
'VR21': 'BlackMagic YUV (Quick Time)',
|
||||
'VSSH': 'Vanguard VSS H.264',
|
||||
'VSSV': 'Vanguard Software Solutions Video Codec',
|
||||
'VSSW': 'Vanguard VSS H.264',
|
||||
'VTLP': 'Alaris VideoGramPixel Codec',
|
||||
'VX1K': 'VX1000S Video Codec',
|
||||
'VX2K': 'VX2000S Video Codec',
|
||||
'VXSP': 'VX1000SP Video Codec',
|
||||
'VYU9': 'ATI Technologies YUV',
|
||||
'VYUY': 'ATI Packed YUV Data',
|
||||
'WBVC': 'Winbond W9960',
|
||||
'WHAM': 'Microsoft Video 1 (WHAM)',
|
||||
'WINX': 'Winnov Software Compression',
|
||||
'WJPG': 'AverMedia Winbond JPEG',
|
||||
'WMV1': 'Windows Media Video V7',
|
||||
'WMV2': 'Windows Media Video V8',
|
||||
'WMV3': 'Windows Media Video V9',
|
||||
'WMVA': 'WMVA codec',
|
||||
'WMVP': 'Windows Media Video V9',
|
||||
'WNIX': 'WniWni Codec',
|
||||
'WNV1': 'Winnov Hardware Compression',
|
||||
'WNVA': 'Winnov hw compress',
|
||||
'WRLE': 'Apple QuickTime BMP Codec',
|
||||
'WRPR': 'VideoTools VideoServer Client Codec',
|
||||
'WV1F': 'WV1F codec',
|
||||
'WVLT': 'IllusionHope Wavelet 9/7',
|
||||
'WVP2': 'WVP2 codec',
|
||||
'X263': 'Xirlink H.263',
|
||||
'X264': 'XiWave GNU GPL x264 MPEG-4 Codec',
|
||||
'XLV0': 'NetXL Video Decoder',
|
||||
'XMPG': 'Xing MPEG (I-Frame only)',
|
||||
'XVID': 'XviD MPEG-4',
|
||||
'XVIX': 'Based on XviD MPEG-4 codec',
|
||||
'XWV0': 'XiWave Video Codec',
|
||||
'XWV1': 'XiWave Video Codec',
|
||||
'XWV2': 'XiWave Video Codec',
|
||||
'XWV3': 'XiWave Video Codec (Xi-3 Video)',
|
||||
'XWV4': 'XiWave Video Codec',
|
||||
'XWV5': 'XiWave Video Codec',
|
||||
'XWV6': 'XiWave Video Codec',
|
||||
'XWV7': 'XiWave Video Codec',
|
||||
'XWV8': 'XiWave Video Codec',
|
||||
'XWV9': 'XiWave Video Codec',
|
||||
'XXAN': 'XXAN',
|
||||
'XYZP': 'Extended PAL format XYZ palette',
|
||||
'Y211': 'YUV 2:1:1 Packed',
|
||||
'Y216': 'Pinnacle TARGA CineWave YUV (Quick Time)',
|
||||
'Y411': 'YUV 4:1:1 Packed',
|
||||
'Y41B': 'YUV 4:1:1 Planar',
|
||||
'Y41P': 'PC1 4:1:1',
|
||||
'Y41T': 'PC1 4:1:1 with transparency',
|
||||
'Y422': 'Y422',
|
||||
'Y42B': 'YUV 4:2:2 Planar',
|
||||
'Y42T': 'PCI 4:2:2 with transparency',
|
||||
'Y444': 'IYU2',
|
||||
'Y8 ': 'Grayscale video',
|
||||
'Y800': 'Simple grayscale video',
|
||||
'YC12': 'Intel YUV12 Codec',
|
||||
'YMPG': 'YMPEG Alpha',
|
||||
'YU12': 'ATI YV12 4:2:0 Planar',
|
||||
'YU92': 'Intel - YUV',
|
||||
'YUNV': 'YUNV',
|
||||
'YUV2': 'Apple Component Video (YUV 4:2:2)',
|
||||
'YUV8': 'Winnov Caviar YUV8',
|
||||
'YUV9': 'Intel YUV9',
|
||||
'YUVP': 'YCbCr 4:2:2',
|
||||
'YUY2': 'Uncompressed YUV 4:2:2',
|
||||
'YUYV': 'Canopus YUV',
|
||||
'YV12': 'YVU12 Planar',
|
||||
'YV16': 'Elecard YUV 4:2:2 Planar',
|
||||
'YV92': 'Intel Smart Video Recorder YVU9',
|
||||
'YVU9': 'Intel YVU9 Planar',
|
||||
'YVYU': 'YVYU 4:2:2 byte ordering',
|
||||
'ZLIB': 'ZLIB',
|
||||
'ZPEG': 'Metheus Video Zipper',
|
||||
'ZYGO': 'ZyGo Video Codec'
|
||||
}
|
||||
|
||||
# make it fool prove
|
||||
for code, value in FOURCC.items():
|
||||
if not code.upper() in FOURCC:
|
||||
FOURCC[code.upper()] = value
|
||||
if code.endswith(' '):
|
||||
FOURCC[code.strip().upper()] = value
|
||||
@@ -0,0 +1,26 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__title__ = 'enzyme'
|
||||
__description__ = 'Video metadata parser'
|
||||
__version__ = '0.1'
|
||||
__author__ = 'Antoine Bertin'
|
||||
__email__ = 'diaoulael@gmail.com'
|
||||
__license__ = 'GPLv3'
|
||||
@@ -0,0 +1,538 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2003-2006 Dirk Meyer <dischi@freevo.org>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
|
||||
import re
|
||||
|
||||
__all__ = ['resolve']
|
||||
|
||||
|
||||
def resolve(code):
|
||||
"""
|
||||
Transform the given (2- or 3-letter) language code to a human readable
|
||||
language name. The return value is a 2-tuple containing the given
|
||||
language code and the language name. If the language code cannot be
|
||||
resolved, name will be 'Unknown (<code>)'.
|
||||
"""
|
||||
if not code:
|
||||
return None, None
|
||||
if not isinstance(code, basestring):
|
||||
raise ValueError('Invalid language code specified by parser')
|
||||
|
||||
# Take up to 3 letters from the code.
|
||||
code = re.split(r'[^a-z]', code.lower())[0][:3]
|
||||
|
||||
for spec in codes:
|
||||
if code in spec[:-1]:
|
||||
return code, spec[-1]
|
||||
|
||||
return code, u'Unknown (%s)' % code
|
||||
|
||||
|
||||
# Parsed from http://www.loc.gov/standards/iso639-2/ISO-639-2_utf-8.txt
|
||||
codes = (
|
||||
('aar', 'aa', u'Afar'),
|
||||
('abk', 'ab', u'Abkhazian'),
|
||||
('ace', u'Achinese'),
|
||||
('ach', u'Acoli'),
|
||||
('ada', u'Adangme'),
|
||||
('ady', u'Adyghe'),
|
||||
('afa', u'Afro-Asiatic '),
|
||||
('afh', u'Afrihili'),
|
||||
('afr', 'af', u'Afrikaans'),
|
||||
('ain', u'Ainu'),
|
||||
('aka', 'ak', u'Akan'),
|
||||
('akk', u'Akkadian'),
|
||||
('alb', 'sq', u'Albanian'),
|
||||
('ale', u'Aleut'),
|
||||
('alg', u'Algonquian languages'),
|
||||
('alt', u'Southern Altai'),
|
||||
('amh', 'am', u'Amharic'),
|
||||
('ang', u'English, Old '),
|
||||
('anp', u'Angika'),
|
||||
('apa', u'Apache languages'),
|
||||
('ara', 'ar', u'Arabic'),
|
||||
('arc', u'Official Aramaic '),
|
||||
('arg', 'an', u'Aragonese'),
|
||||
('arm', 'hy', u'Armenian'),
|
||||
('arn', u'Mapudungun'),
|
||||
('arp', u'Arapaho'),
|
||||
('art', u'Artificial '),
|
||||
('arw', u'Arawak'),
|
||||
('asm', 'as', u'Assamese'),
|
||||
('ast', u'Asturian'),
|
||||
('ath', u'Athapascan languages'),
|
||||
('aus', u'Australian languages'),
|
||||
('ava', 'av', u'Avaric'),
|
||||
('ave', 'ae', u'Avestan'),
|
||||
('awa', u'Awadhi'),
|
||||
('aym', 'ay', u'Aymara'),
|
||||
('aze', 'az', u'Azerbaijani'),
|
||||
('bad', u'Banda languages'),
|
||||
('bai', u'Bamileke languages'),
|
||||
('bak', 'ba', u'Bashkir'),
|
||||
('bal', u'Baluchi'),
|
||||
('bam', 'bm', u'Bambara'),
|
||||
('ban', u'Balinese'),
|
||||
('baq', 'eu', u'Basque'),
|
||||
('bas', u'Basa'),
|
||||
('bat', u'Baltic '),
|
||||
('bej', u'Beja'),
|
||||
('bel', 'be', u'Belarusian'),
|
||||
('bem', u'Bemba'),
|
||||
('ben', 'bn', u'Bengali'),
|
||||
('ber', u'Berber '),
|
||||
('bho', u'Bhojpuri'),
|
||||
('bih', 'bh', u'Bihari'),
|
||||
('bik', u'Bikol'),
|
||||
('bin', u'Bini'),
|
||||
('bis', 'bi', u'Bislama'),
|
||||
('bla', u'Siksika'),
|
||||
('bnt', u'Bantu '),
|
||||
('bos', 'bs', u'Bosnian'),
|
||||
('bra', u'Braj'),
|
||||
('bre', 'br', u'Breton'),
|
||||
('btk', u'Batak languages'),
|
||||
('bua', u'Buriat'),
|
||||
('bug', u'Buginese'),
|
||||
('bul', 'bg', u'Bulgarian'),
|
||||
('bur', 'my', u'Burmese'),
|
||||
('byn', u'Blin'),
|
||||
('cad', u'Caddo'),
|
||||
('cai', u'Central American Indian '),
|
||||
('car', u'Galibi Carib'),
|
||||
('cat', 'ca', u'Catalan'),
|
||||
('cau', u'Caucasian '),
|
||||
('ceb', u'Cebuano'),
|
||||
('cel', u'Celtic '),
|
||||
('cha', 'ch', u'Chamorro'),
|
||||
('chb', u'Chibcha'),
|
||||
('che', 'ce', u'Chechen'),
|
||||
('chg', u'Chagatai'),
|
||||
('chi', 'zh', u'Chinese'),
|
||||
('chk', u'Chuukese'),
|
||||
('chm', u'Mari'),
|
||||
('chn', u'Chinook jargon'),
|
||||
('cho', u'Choctaw'),
|
||||
('chp', u'Chipewyan'),
|
||||
('chr', u'Cherokee'),
|
||||
('chu', 'cu', u'Church Slavic'),
|
||||
('chv', 'cv', u'Chuvash'),
|
||||
('chy', u'Cheyenne'),
|
||||
('cmc', u'Chamic languages'),
|
||||
('cop', u'Coptic'),
|
||||
('cor', 'kw', u'Cornish'),
|
||||
('cos', 'co', u'Corsican'),
|
||||
('cpe', u'Creoles and pidgins, English based '),
|
||||
('cpf', u'Creoles and pidgins, French-based '),
|
||||
('cpp', u'Creoles and pidgins, Portuguese-based '),
|
||||
('cre', 'cr', u'Cree'),
|
||||
('crh', u'Crimean Tatar'),
|
||||
('crp', u'Creoles and pidgins '),
|
||||
('csb', u'Kashubian'),
|
||||
('cus', u'Cushitic '),
|
||||
('cze', 'cs', u'Czech'),
|
||||
('dak', u'Dakota'),
|
||||
('dan', 'da', u'Danish'),
|
||||
('dar', u'Dargwa'),
|
||||
('day', u'Land Dayak languages'),
|
||||
('del', u'Delaware'),
|
||||
('den', u'Slave '),
|
||||
('dgr', u'Dogrib'),
|
||||
('din', u'Dinka'),
|
||||
('div', 'dv', u'Divehi'),
|
||||
('doi', u'Dogri'),
|
||||
('dra', u'Dravidian '),
|
||||
('dsb', u'Lower Sorbian'),
|
||||
('dua', u'Duala'),
|
||||
('dum', u'Dutch, Middle '),
|
||||
('dut', 'nl', u'Dutch'),
|
||||
('dyu', u'Dyula'),
|
||||
('dzo', 'dz', u'Dzongkha'),
|
||||
('efi', u'Efik'),
|
||||
('egy', u'Egyptian '),
|
||||
('eka', u'Ekajuk'),
|
||||
('elx', u'Elamite'),
|
||||
('eng', 'en', u'English'),
|
||||
('enm', u'English, Middle '),
|
||||
('epo', 'eo', u'Esperanto'),
|
||||
('est', 'et', u'Estonian'),
|
||||
('ewe', 'ee', u'Ewe'),
|
||||
('ewo', u'Ewondo'),
|
||||
('fan', u'Fang'),
|
||||
('fao', 'fo', u'Faroese'),
|
||||
('fat', u'Fanti'),
|
||||
('fij', 'fj', u'Fijian'),
|
||||
('fil', u'Filipino'),
|
||||
('fin', 'fi', u'Finnish'),
|
||||
('fiu', u'Finno-Ugrian '),
|
||||
('fon', u'Fon'),
|
||||
('fre', 'fr', u'French'),
|
||||
('frm', u'French, Middle '),
|
||||
('fro', u'French, Old '),
|
||||
('frr', u'Northern Frisian'),
|
||||
('frs', u'Eastern Frisian'),
|
||||
('fry', 'fy', u'Western Frisian'),
|
||||
('ful', 'ff', u'Fulah'),
|
||||
('fur', u'Friulian'),
|
||||
('gaa', u'Ga'),
|
||||
('gay', u'Gayo'),
|
||||
('gba', u'Gbaya'),
|
||||
('gem', u'Germanic '),
|
||||
('geo', 'ka', u'Georgian'),
|
||||
('ger', 'de', u'German'),
|
||||
('gez', u'Geez'),
|
||||
('gil', u'Gilbertese'),
|
||||
('gla', 'gd', u'Gaelic'),
|
||||
('gle', 'ga', u'Irish'),
|
||||
('glg', 'gl', u'Galician'),
|
||||
('glv', 'gv', u'Manx'),
|
||||
('gmh', u'German, Middle High '),
|
||||
('goh', u'German, Old High '),
|
||||
('gon', u'Gondi'),
|
||||
('gor', u'Gorontalo'),
|
||||
('got', u'Gothic'),
|
||||
('grb', u'Grebo'),
|
||||
('grc', u'Greek, Ancient '),
|
||||
('gre', 'el', u'Greek, Modern '),
|
||||
('grn', 'gn', u'Guarani'),
|
||||
('gsw', u'Swiss German'),
|
||||
('guj', 'gu', u'Gujarati'),
|
||||
('gwi', u"Gwich'in"),
|
||||
('hai', u'Haida'),
|
||||
('hat', 'ht', u'Haitian'),
|
||||
('hau', 'ha', u'Hausa'),
|
||||
('haw', u'Hawaiian'),
|
||||
('heb', 'he', u'Hebrew'),
|
||||
('her', 'hz', u'Herero'),
|
||||
('hil', u'Hiligaynon'),
|
||||
('him', u'Himachali'),
|
||||
('hin', 'hi', u'Hindi'),
|
||||
('hit', u'Hittite'),
|
||||
('hmn', u'Hmong'),
|
||||
('hmo', 'ho', u'Hiri Motu'),
|
||||
('hsb', u'Upper Sorbian'),
|
||||
('hun', 'hu', u'Hungarian'),
|
||||
('hup', u'Hupa'),
|
||||
('iba', u'Iban'),
|
||||
('ibo', 'ig', u'Igbo'),
|
||||
('ice', 'is', u'Icelandic'),
|
||||
('ido', 'io', u'Ido'),
|
||||
('iii', 'ii', u'Sichuan Yi'),
|
||||
('ijo', u'Ijo languages'),
|
||||
('iku', 'iu', u'Inuktitut'),
|
||||
('ile', 'ie', u'Interlingue'),
|
||||
('ilo', u'Iloko'),
|
||||
('ina', 'ia', u'Interlingua '),
|
||||
('inc', u'Indic '),
|
||||
('ind', 'id', u'Indonesian'),
|
||||
('ine', u'Indo-European '),
|
||||
('inh', u'Ingush'),
|
||||
('ipk', 'ik', u'Inupiaq'),
|
||||
('ira', u'Iranian '),
|
||||
('iro', u'Iroquoian languages'),
|
||||
('ita', 'it', u'Italian'),
|
||||
('jav', 'jv', u'Javanese'),
|
||||
('jbo', u'Lojban'),
|
||||
('jpn', 'ja', u'Japanese'),
|
||||
('jpr', u'Judeo-Persian'),
|
||||
('jrb', u'Judeo-Arabic'),
|
||||
('kaa', u'Kara-Kalpak'),
|
||||
('kab', u'Kabyle'),
|
||||
('kac', u'Kachin'),
|
||||
('kal', 'kl', u'Kalaallisut'),
|
||||
('kam', u'Kamba'),
|
||||
('kan', 'kn', u'Kannada'),
|
||||
('kar', u'Karen languages'),
|
||||
('kas', 'ks', u'Kashmiri'),
|
||||
('kau', 'kr', u'Kanuri'),
|
||||
('kaw', u'Kawi'),
|
||||
('kaz', 'kk', u'Kazakh'),
|
||||
('kbd', u'Kabardian'),
|
||||
('kha', u'Khasi'),
|
||||
('khi', u'Khoisan '),
|
||||
('khm', 'km', u'Central Khmer'),
|
||||
('kho', u'Khotanese'),
|
||||
('kik', 'ki', u'Kikuyu'),
|
||||
('kin', 'rw', u'Kinyarwanda'),
|
||||
('kir', 'ky', u'Kirghiz'),
|
||||
('kmb', u'Kimbundu'),
|
||||
('kok', u'Konkani'),
|
||||
('kom', 'kv', u'Komi'),
|
||||
('kon', 'kg', u'Kongo'),
|
||||
('kor', 'ko', u'Korean'),
|
||||
('kos', u'Kosraean'),
|
||||
('kpe', u'Kpelle'),
|
||||
('krc', u'Karachay-Balkar'),
|
||||
('krl', u'Karelian'),
|
||||
('kro', u'Kru languages'),
|
||||
('kru', u'Kurukh'),
|
||||
('kua', 'kj', u'Kuanyama'),
|
||||
('kum', u'Kumyk'),
|
||||
('kur', 'ku', u'Kurdish'),
|
||||
('kut', u'Kutenai'),
|
||||
('lad', u'Ladino'),
|
||||
('lah', u'Lahnda'),
|
||||
('lam', u'Lamba'),
|
||||
('lao', 'lo', u'Lao'),
|
||||
('lat', 'la', u'Latin'),
|
||||
('lav', 'lv', u'Latvian'),
|
||||
('lez', u'Lezghian'),
|
||||
('lim', 'li', u'Limburgan'),
|
||||
('lin', 'ln', u'Lingala'),
|
||||
('lit', 'lt', u'Lithuanian'),
|
||||
('lol', u'Mongo'),
|
||||
('loz', u'Lozi'),
|
||||
('ltz', 'lb', u'Luxembourgish'),
|
||||
('lua', u'Luba-Lulua'),
|
||||
('lub', 'lu', u'Luba-Katanga'),
|
||||
('lug', 'lg', u'Ganda'),
|
||||
('lui', u'Luiseno'),
|
||||
('lun', u'Lunda'),
|
||||
('luo', u'Luo '),
|
||||
('lus', u'Lushai'),
|
||||
('mac', 'mk', u'Macedonian'),
|
||||
('mad', u'Madurese'),
|
||||
('mag', u'Magahi'),
|
||||
('mah', 'mh', u'Marshallese'),
|
||||
('mai', u'Maithili'),
|
||||
('mak', u'Makasar'),
|
||||
('mal', 'ml', u'Malayalam'),
|
||||
('man', u'Mandingo'),
|
||||
('mao', 'mi', u'Maori'),
|
||||
('map', u'Austronesian '),
|
||||
('mar', 'mr', u'Marathi'),
|
||||
('mas', u'Masai'),
|
||||
('may', 'ms', u'Malay'),
|
||||
('mdf', u'Moksha'),
|
||||
('mdr', u'Mandar'),
|
||||
('men', u'Mende'),
|
||||
('mga', u'Irish, Middle '),
|
||||
('mic', u"Mi'kmaq"),
|
||||
('min', u'Minangkabau'),
|
||||
('mis', u'Uncoded languages'),
|
||||
('mkh', u'Mon-Khmer '),
|
||||
('mlg', 'mg', u'Malagasy'),
|
||||
('mlt', 'mt', u'Maltese'),
|
||||
('mnc', u'Manchu'),
|
||||
('mni', u'Manipuri'),
|
||||
('mno', u'Manobo languages'),
|
||||
('moh', u'Mohawk'),
|
||||
('mol', 'mo', u'Moldavian'),
|
||||
('mon', 'mn', u'Mongolian'),
|
||||
('mos', u'Mossi'),
|
||||
('mul', u'Multiple languages'),
|
||||
('mun', u'Munda languages'),
|
||||
('mus', u'Creek'),
|
||||
('mwl', u'Mirandese'),
|
||||
('mwr', u'Marwari'),
|
||||
('myn', u'Mayan languages'),
|
||||
('myv', u'Erzya'),
|
||||
('nah', u'Nahuatl languages'),
|
||||
('nai', u'North American Indian'),
|
||||
('nap', u'Neapolitan'),
|
||||
('nau', 'na', u'Nauru'),
|
||||
('nav', 'nv', u'Navajo'),
|
||||
('nbl', 'nr', u'Ndebele, South'),
|
||||
('nde', 'nd', u'Ndebele, North'),
|
||||
('ndo', 'ng', u'Ndonga'),
|
||||
('nds', u'Low German'),
|
||||
('nep', 'ne', u'Nepali'),
|
||||
('new', u'Nepal Bhasa'),
|
||||
('nia', u'Nias'),
|
||||
('nic', u'Niger-Kordofanian '),
|
||||
('niu', u'Niuean'),
|
||||
('nno', 'nn', u'Norwegian Nynorsk'),
|
||||
('nob', 'nb', u'Bokm\xe5l, Norwegian'),
|
||||
('nog', u'Nogai'),
|
||||
('non', u'Norse, Old'),
|
||||
('nor', 'no', u'Norwegian'),
|
||||
('nqo', u"N'Ko"),
|
||||
('nso', u'Pedi'),
|
||||
('nub', u'Nubian languages'),
|
||||
('nwc', u'Classical Newari'),
|
||||
('nya', 'ny', u'Chichewa'),
|
||||
('nym', u'Nyamwezi'),
|
||||
('nyn', u'Nyankole'),
|
||||
('nyo', u'Nyoro'),
|
||||
('nzi', u'Nzima'),
|
||||
('oci', 'oc', u'Occitan '),
|
||||
('oji', 'oj', u'Ojibwa'),
|
||||
('ori', 'or', u'Oriya'),
|
||||
('orm', 'om', u'Oromo'),
|
||||
('osa', u'Osage'),
|
||||
('oss', 'os', u'Ossetian'),
|
||||
('ota', u'Turkish, Ottoman '),
|
||||
('oto', u'Otomian languages'),
|
||||
('paa', u'Papuan '),
|
||||
('pag', u'Pangasinan'),
|
||||
('pal', u'Pahlavi'),
|
||||
('pam', u'Pampanga'),
|
||||
('pan', 'pa', u'Panjabi'),
|
||||
('pap', u'Papiamento'),
|
||||
('pau', u'Palauan'),
|
||||
('peo', u'Persian, Old '),
|
||||
('per', 'fa', u'Persian'),
|
||||
('phi', u'Philippine '),
|
||||
('phn', u'Phoenician'),
|
||||
('pli', 'pi', u'Pali'),
|
||||
('pol', 'pl', u'Polish'),
|
||||
('pon', u'Pohnpeian'),
|
||||
('por', 'pt', u'Portuguese'),
|
||||
('pra', u'Prakrit languages'),
|
||||
('pro', u'Proven\xe7al, Old '),
|
||||
('pus', 'ps', u'Pushto'),
|
||||
('qaa-qtz', u'Reserved for local use'),
|
||||
('que', 'qu', u'Quechua'),
|
||||
('raj', u'Rajasthani'),
|
||||
('rap', u'Rapanui'),
|
||||
('rar', u'Rarotongan'),
|
||||
('roa', u'Romance '),
|
||||
('roh', 'rm', u'Romansh'),
|
||||
('rom', u'Romany'),
|
||||
('rum', 'ro', u'Romanian'),
|
||||
('run', 'rn', u'Rundi'),
|
||||
('rup', u'Aromanian'),
|
||||
('rus', 'ru', u'Russian'),
|
||||
('sad', u'Sandawe'),
|
||||
('sag', 'sg', u'Sango'),
|
||||
('sah', u'Yakut'),
|
||||
('sai', u'South American Indian '),
|
||||
('sal', u'Salishan languages'),
|
||||
('sam', u'Samaritan Aramaic'),
|
||||
('san', 'sa', u'Sanskrit'),
|
||||
('sas', u'Sasak'),
|
||||
('sat', u'Santali'),
|
||||
('scc', 'sr', u'Serbian'),
|
||||
('scn', u'Sicilian'),
|
||||
('sco', u'Scots'),
|
||||
('scr', 'hr', u'Croatian'),
|
||||
('sel', u'Selkup'),
|
||||
('sem', u'Semitic '),
|
||||
('sga', u'Irish, Old '),
|
||||
('sgn', u'Sign Languages'),
|
||||
('shn', u'Shan'),
|
||||
('sid', u'Sidamo'),
|
||||
('sin', 'si', u'Sinhala'),
|
||||
('sio', u'Siouan languages'),
|
||||
('sit', u'Sino-Tibetan '),
|
||||
('sla', u'Slavic '),
|
||||
('slo', 'sk', u'Slovak'),
|
||||
('slv', 'sl', u'Slovenian'),
|
||||
('sma', u'Southern Sami'),
|
||||
('sme', 'se', u'Northern Sami'),
|
||||
('smi', u'Sami languages '),
|
||||
('smj', u'Lule Sami'),
|
||||
('smn', u'Inari Sami'),
|
||||
('smo', 'sm', u'Samoan'),
|
||||
('sms', u'Skolt Sami'),
|
||||
('sna', 'sn', u'Shona'),
|
||||
('snd', 'sd', u'Sindhi'),
|
||||
('snk', u'Soninke'),
|
||||
('sog', u'Sogdian'),
|
||||
('som', 'so', u'Somali'),
|
||||
('son', u'Songhai languages'),
|
||||
('sot', 'st', u'Sotho, Southern'),
|
||||
('spa', 'es', u'Spanish'),
|
||||
('srd', 'sc', u'Sardinian'),
|
||||
('srn', u'Sranan Tongo'),
|
||||
('srr', u'Serer'),
|
||||
('ssa', u'Nilo-Saharan '),
|
||||
('ssw', 'ss', u'Swati'),
|
||||
('suk', u'Sukuma'),
|
||||
('sun', 'su', u'Sundanese'),
|
||||
('sus', u'Susu'),
|
||||
('sux', u'Sumerian'),
|
||||
('swa', 'sw', u'Swahili'),
|
||||
('swe', 'sv', u'Swedish'),
|
||||
('syc', u'Classical Syriac'),
|
||||
('syr', u'Syriac'),
|
||||
('tah', 'ty', u'Tahitian'),
|
||||
('tai', u'Tai '),
|
||||
('tam', 'ta', u'Tamil'),
|
||||
('tat', 'tt', u'Tatar'),
|
||||
('tel', 'te', u'Telugu'),
|
||||
('tem', u'Timne'),
|
||||
('ter', u'Tereno'),
|
||||
('tet', u'Tetum'),
|
||||
('tgk', 'tg', u'Tajik'),
|
||||
('tgl', 'tl', u'Tagalog'),
|
||||
('tha', 'th', u'Thai'),
|
||||
('tib', 'bo', u'Tibetan'),
|
||||
('tig', u'Tigre'),
|
||||
('tir', 'ti', u'Tigrinya'),
|
||||
('tiv', u'Tiv'),
|
||||
('tkl', u'Tokelau'),
|
||||
('tlh', u'Klingon'),
|
||||
('tli', u'Tlingit'),
|
||||
('tmh', u'Tamashek'),
|
||||
('tog', u'Tonga '),
|
||||
('ton', 'to', u'Tonga '),
|
||||
('tpi', u'Tok Pisin'),
|
||||
('tsi', u'Tsimshian'),
|
||||
('tsn', 'tn', u'Tswana'),
|
||||
('tso', 'ts', u'Tsonga'),
|
||||
('tuk', 'tk', u'Turkmen'),
|
||||
('tum', u'Tumbuka'),
|
||||
('tup', u'Tupi languages'),
|
||||
('tur', 'tr', u'Turkish'),
|
||||
('tut', u'Altaic '),
|
||||
('tvl', u'Tuvalu'),
|
||||
('twi', 'tw', u'Twi'),
|
||||
('tyv', u'Tuvinian'),
|
||||
('udm', u'Udmurt'),
|
||||
('uga', u'Ugaritic'),
|
||||
('uig', 'ug', u'Uighur'),
|
||||
('ukr', 'uk', u'Ukrainian'),
|
||||
('umb', u'Umbundu'),
|
||||
('und', u'Undetermined'),
|
||||
('urd', 'ur', u'Urdu'),
|
||||
('uzb', 'uz', u'Uzbek'),
|
||||
('vai', u'Vai'),
|
||||
('ven', 've', u'Venda'),
|
||||
('vie', 'vi', u'Vietnamese'),
|
||||
('vol', 'vo', u'Volap\xfck'),
|
||||
('vot', u'Votic'),
|
||||
('wak', u'Wakashan languages'),
|
||||
('wal', u'Walamo'),
|
||||
('war', u'Waray'),
|
||||
('was', u'Washo'),
|
||||
('wel', 'cy', u'Welsh'),
|
||||
('wen', u'Sorbian languages'),
|
||||
('wln', 'wa', u'Walloon'),
|
||||
('wol', 'wo', u'Wolof'),
|
||||
('xal', u'Kalmyk'),
|
||||
('xho', 'xh', u'Xhosa'),
|
||||
('yao', u'Yao'),
|
||||
('yap', u'Yapese'),
|
||||
('yid', 'yi', u'Yiddish'),
|
||||
('yor', 'yo', u'Yoruba'),
|
||||
('ypk', u'Yupik languages'),
|
||||
('zap', u'Zapotec'),
|
||||
('zbl', u'Blissymbols'),
|
||||
('zen', u'Zenaga'),
|
||||
('zha', 'za', u'Zhuang'),
|
||||
('znd', u'Zande languages'),
|
||||
('zul', 'zu', u'Zulu'),
|
||||
('zun', u'Zuni'),
|
||||
('zxx', u'No linguistic content'),
|
||||
('zza', u'Zaza'),
|
||||
)
|
||||
@@ -0,0 +1,841 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2003-2006 Thomas Schueppel <stain@acm.org>
|
||||
# Copyright (C) 2003-2006 Dirk Meyer <dischi@freevo.org>
|
||||
# Copyright (C) 2003-2006 Jason Tackaberry <tack@urandom.ca>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__all__ = ['Parser']
|
||||
|
||||
from datetime import datetime
|
||||
from exceptions import *
|
||||
from struct import unpack
|
||||
import core
|
||||
import logging
|
||||
import re
|
||||
|
||||
# get logging object
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# Main IDs for the Matroska streams
|
||||
MATROSKA_VIDEO_TRACK = 0x01
|
||||
MATROSKA_AUDIO_TRACK = 0x02
|
||||
MATROSKA_SUBTITLES_TRACK = 0x11
|
||||
|
||||
MATROSKA_HEADER_ID = 0x1A45DFA3
|
||||
MATROSKA_TRACKS_ID = 0x1654AE6B
|
||||
MATROSKA_CUES_ID = 0x1C53BB6B
|
||||
MATROSKA_SEGMENT_ID = 0x18538067
|
||||
MATROSKA_SEGMENT_INFO_ID = 0x1549A966
|
||||
MATROSKA_CLUSTER_ID = 0x1F43B675
|
||||
MATROSKA_VOID_ID = 0xEC
|
||||
MATROSKA_CRC_ID = 0xBF
|
||||
MATROSKA_TIMECODESCALE_ID = 0x2AD7B1
|
||||
MATROSKA_DURATION_ID = 0x4489
|
||||
MATROSKA_CRC32_ID = 0xBF
|
||||
MATROSKA_TIMECODESCALE_ID = 0x2AD7B1
|
||||
MATROSKA_MUXING_APP_ID = 0x4D80
|
||||
MATROSKA_WRITING_APP_ID = 0x5741
|
||||
MATROSKA_CODEC_ID = 0x86
|
||||
MATROSKA_CODEC_PRIVATE_ID = 0x63A2
|
||||
MATROSKA_FRAME_DURATION_ID = 0x23E383
|
||||
MATROSKA_VIDEO_SETTINGS_ID = 0xE0
|
||||
MATROSKA_VIDEO_WIDTH_ID = 0xB0
|
||||
MATROSKA_VIDEO_HEIGHT_ID = 0xBA
|
||||
MATROSKA_VIDEO_INTERLACED_ID = 0x9A
|
||||
MATROSKA_VIDEO_DISPLAY_WIDTH_ID = 0x54B0
|
||||
MATROSKA_VIDEO_DISPLAY_HEIGHT_ID = 0x54BA
|
||||
MATROSKA_AUDIO_SETTINGS_ID = 0xE1
|
||||
MATROSKA_AUDIO_SAMPLERATE_ID = 0xB5
|
||||
MATROSKA_AUDIO_CHANNELS_ID = 0x9F
|
||||
MATROSKA_TRACK_UID_ID = 0x73C5
|
||||
MATROSKA_TRACK_NUMBER_ID = 0xD7
|
||||
MATROSKA_TRACK_TYPE_ID = 0x83
|
||||
MATROSKA_TRACK_LANGUAGE_ID = 0x22B59C
|
||||
MATROSKA_TRACK_OFFSET = 0x537F
|
||||
MATROSKA_TRACK_FLAG_DEFAULT_ID = 0x88
|
||||
MATROSKA_TRACK_FLAG_ENABLED_ID = 0xB9
|
||||
MATROSKA_TITLE_ID = 0x7BA9
|
||||
MATROSKA_DATE_UTC_ID = 0x4461
|
||||
MATROSKA_NAME_ID = 0x536E
|
||||
|
||||
MATROSKA_CHAPTERS_ID = 0x1043A770
|
||||
MATROSKA_CHAPTER_UID_ID = 0x73C4
|
||||
MATROSKA_EDITION_ENTRY_ID = 0x45B9
|
||||
MATROSKA_CHAPTER_ATOM_ID = 0xB6
|
||||
MATROSKA_CHAPTER_TIME_START_ID = 0x91
|
||||
MATROSKA_CHAPTER_TIME_END_ID = 0x92
|
||||
MATROSKA_CHAPTER_FLAG_ENABLED_ID = 0x4598
|
||||
MATROSKA_CHAPTER_DISPLAY_ID = 0x80
|
||||
MATROSKA_CHAPTER_LANGUAGE_ID = 0x437C
|
||||
MATROSKA_CHAPTER_STRING_ID = 0x85
|
||||
|
||||
MATROSKA_ATTACHMENTS_ID = 0x1941A469
|
||||
MATROSKA_ATTACHED_FILE_ID = 0x61A7
|
||||
MATROSKA_FILE_DESC_ID = 0x467E
|
||||
MATROSKA_FILE_NAME_ID = 0x466E
|
||||
MATROSKA_FILE_MIME_TYPE_ID = 0x4660
|
||||
MATROSKA_FILE_DATA_ID = 0x465C
|
||||
|
||||
MATROSKA_SEEKHEAD_ID = 0x114D9B74
|
||||
MATROSKA_SEEK_ID = 0x4DBB
|
||||
MATROSKA_SEEKID_ID = 0x53AB
|
||||
MATROSKA_SEEK_POSITION_ID = 0x53AC
|
||||
|
||||
MATROSKA_TAGS_ID = 0x1254C367
|
||||
MATROSKA_TAG_ID = 0x7373
|
||||
MATROSKA_TARGETS_ID = 0x63C0
|
||||
MATROSKA_TARGET_TYPE_VALUE_ID = 0x68CA
|
||||
MATROSKA_TARGET_TYPE_ID = 0x63CA
|
||||
MATRSOKA_TAGS_TRACK_UID_ID = 0x63C5
|
||||
MATRSOKA_TAGS_EDITION_UID_ID = 0x63C9
|
||||
MATRSOKA_TAGS_CHAPTER_UID_ID = 0x63C4
|
||||
MATRSOKA_TAGS_ATTACHMENT_UID_ID = 0x63C6
|
||||
MATROSKA_SIMPLE_TAG_ID = 0x67C8
|
||||
MATROSKA_TAG_NAME_ID = 0x45A3
|
||||
MATROSKA_TAG_LANGUAGE_ID = 0x447A
|
||||
MATROSKA_TAG_STRING_ID = 0x4487
|
||||
MATROSKA_TAG_BINARY_ID = 0x4485
|
||||
|
||||
|
||||
# See mkv spec for details:
|
||||
# http://www.matroska.org/technical/specs/index.html
|
||||
|
||||
# Map to convert to well known codes
|
||||
# http://haali.cs.msu.ru/mkv/codecs.pdf
|
||||
FOURCCMap = {
|
||||
'V_THEORA': 'THEO',
|
||||
'V_SNOW': 'SNOW',
|
||||
'V_MPEG4/ISO/ASP': 'MP4V',
|
||||
'V_MPEG4/ISO/AVC': 'AVC1',
|
||||
'A_AC3': 0x2000,
|
||||
'A_MPEG/L3': 0x0055,
|
||||
'A_MPEG/L2': 0x0050,
|
||||
'A_MPEG/L1': 0x0050,
|
||||
'A_DTS': 0x2001,
|
||||
'A_PCM/INT/LIT': 0x0001,
|
||||
'A_PCM/FLOAT/IEEE': 0x003,
|
||||
'A_TTA1': 0x77a1,
|
||||
'A_WAVPACK4': 0x5756,
|
||||
'A_VORBIS': 0x6750,
|
||||
'A_FLAC': 0xF1AC,
|
||||
'A_AAC': 0x00ff,
|
||||
'A_AAC/': 0x00ff
|
||||
}
|
||||
|
||||
|
||||
def matroska_date_to_datetime(date):
|
||||
"""
|
||||
Converts a date in Matroska's date format to a python datetime object.
|
||||
Returns the given date string if it could not be converted.
|
||||
"""
|
||||
# From the specs:
|
||||
# The fields with dates should have the following format: YYYY-MM-DD
|
||||
# HH:MM:SS.MSS [...] To store less accuracy, you remove items starting
|
||||
# from the right. To store only the year, you would use, "2004". To store
|
||||
# a specific day such as May 1st, 2003, you would use "2003-05-01".
|
||||
format = re.split(r'([-:. ])', '%Y-%m-%d %H:%M:%S.%f')
|
||||
while format:
|
||||
try:
|
||||
return datetime.strptime(date, ''.join(format))
|
||||
except ValueError:
|
||||
format = format[:-2]
|
||||
return date
|
||||
|
||||
|
||||
def matroska_bps_to_bitrate(bps):
|
||||
"""
|
||||
Tries to convert a free-form bps string into a bitrate (bits per second).
|
||||
"""
|
||||
m = re.search('([\d.]+)\s*(\D.*)', bps)
|
||||
if m:
|
||||
bps, suffix = m.groups()
|
||||
if 'kbit' in suffix:
|
||||
return float(bps) * 1024
|
||||
elif 'kbyte' in suffix:
|
||||
return float(bps) * 1024 * 8
|
||||
elif 'byte' in suffix:
|
||||
return float(bps) * 8
|
||||
elif 'bps' in suffix or 'bit' in suffix:
|
||||
return float(bps)
|
||||
if bps.replace('.', '').isdigit():
|
||||
if float(bps) < 30000:
|
||||
# Assume kilobits and convert to bps
|
||||
return float(bps) * 1024
|
||||
return float(bps)
|
||||
|
||||
|
||||
# Used to convert the official matroska tag names (only lower-cased) to core
|
||||
# attributes. tag name -> attr, filter
|
||||
TAGS_MAP = {
|
||||
# From Media core
|
||||
u'title': ('title', None),
|
||||
u'subtitle': ('caption', None),
|
||||
u'comment': ('comment', None),
|
||||
u'url': ('url', None),
|
||||
u'artist': ('artist', None),
|
||||
u'keywords': ('keywords', lambda s: [word.strip() for word in s.split(',')]),
|
||||
u'composer_nationality': ('country', None),
|
||||
u'date_released': ('datetime', None),
|
||||
u'date_recorded': ('datetime', None),
|
||||
u'date_written': ('datetime', None),
|
||||
|
||||
# From Video core
|
||||
u'encoder': ('encoder', None),
|
||||
u'bps': ('bitrate', matroska_bps_to_bitrate),
|
||||
u'part_number': ('trackno', int),
|
||||
u'total_parts': ('trackof', int),
|
||||
u'copyright': ('copyright', None),
|
||||
u'genre': ('genre', None),
|
||||
u'actor': ('actors', None),
|
||||
u'written_by': ('writer', None),
|
||||
u'producer': ('producer', None),
|
||||
u'production_studio': ('studio', None),
|
||||
u'law_rating': ('rating', None),
|
||||
u'summary': ('summary', None),
|
||||
u'synopsis': ('synopsis', None),
|
||||
}
|
||||
|
||||
|
||||
class EbmlEntity:
|
||||
"""
|
||||
This is class that is responsible to handle one Ebml entity as described in
|
||||
the Matroska/Ebml spec
|
||||
"""
|
||||
def __init__(self, inbuf):
|
||||
# Compute the EBML id
|
||||
# Set the CRC len to zero
|
||||
self.crc_len = 0
|
||||
# Now loop until we find an entity without CRC
|
||||
try:
|
||||
self.build_entity(inbuf)
|
||||
except IndexError:
|
||||
raise ParseError()
|
||||
while self.get_id() == MATROSKA_CRC32_ID:
|
||||
self.crc_len += self.get_total_len()
|
||||
inbuf = inbuf[self.get_total_len():]
|
||||
self.build_entity(inbuf)
|
||||
|
||||
def build_entity(self, inbuf):
|
||||
self.compute_id(inbuf)
|
||||
|
||||
if self.id_len == 0:
|
||||
log.error("EBML entity not found, bad file format")
|
||||
raise ParseError()
|
||||
|
||||
self.entity_len, self.len_size = self.compute_len(inbuf[self.id_len:])
|
||||
self.entity_data = inbuf[self.get_header_len() : self.get_total_len()]
|
||||
self.ebml_length = self.entity_len
|
||||
self.entity_len = min(len(self.entity_data), self.entity_len)
|
||||
|
||||
# if the data size is 8 or less, it could be a numeric value
|
||||
self.value = 0
|
||||
if self.entity_len <= 8:
|
||||
for pos, shift in zip(range(self.entity_len), range((self.entity_len-1)*8, -1, -8)):
|
||||
self.value |= ord(self.entity_data[pos]) << shift
|
||||
|
||||
|
||||
def add_data(self, data):
|
||||
maxlen = self.ebml_length - len(self.entity_data)
|
||||
if maxlen <= 0:
|
||||
return
|
||||
self.entity_data += data[:maxlen]
|
||||
self.entity_len = len(self.entity_data)
|
||||
|
||||
|
||||
def compute_id(self, inbuf):
|
||||
self.id_len = 0
|
||||
if len(inbuf) < 1:
|
||||
return 0
|
||||
first = ord(inbuf[0])
|
||||
if first & 0x80:
|
||||
self.id_len = 1
|
||||
self.entity_id = first
|
||||
elif first & 0x40:
|
||||
if len(inbuf) < 2:
|
||||
return 0
|
||||
self.id_len = 2
|
||||
self.entity_id = ord(inbuf[0])<<8 | ord(inbuf[1])
|
||||
elif first & 0x20:
|
||||
if len(inbuf) < 3:
|
||||
return 0
|
||||
self.id_len = 3
|
||||
self.entity_id = (ord(inbuf[0])<<16) | (ord(inbuf[1])<<8) | \
|
||||
(ord(inbuf[2]))
|
||||
elif first & 0x10:
|
||||
if len(inbuf) < 4:
|
||||
return 0
|
||||
self.id_len = 4
|
||||
self.entity_id = (ord(inbuf[0])<<24) | (ord(inbuf[1])<<16) | \
|
||||
(ord(inbuf[2])<<8) | (ord(inbuf[3]))
|
||||
self.entity_str = inbuf[0:self.id_len]
|
||||
|
||||
|
||||
def compute_len(self, inbuf):
|
||||
if not inbuf:
|
||||
return 0, 0
|
||||
i = num_ffs = 0
|
||||
len_mask = 0x80
|
||||
len = ord(inbuf[0])
|
||||
while not len & len_mask:
|
||||
i += 1
|
||||
len_mask >>= 1
|
||||
if i >= 8:
|
||||
return 0, 0
|
||||
|
||||
len &= len_mask - 1
|
||||
if len == len_mask - 1:
|
||||
num_ffs += 1
|
||||
for p in range(i):
|
||||
len = (len << 8) | ord(inbuf[p + 1])
|
||||
if len & 0xff == 0xff:
|
||||
num_ffs += 1
|
||||
if num_ffs == i + 1:
|
||||
len = 0
|
||||
return len, i + 1
|
||||
|
||||
|
||||
def get_crc_len(self):
|
||||
return self.crc_len
|
||||
|
||||
|
||||
def get_value(self):
|
||||
return self.value
|
||||
|
||||
|
||||
def get_float_value(self):
|
||||
if len(self.entity_data) == 4:
|
||||
return unpack('!f', self.entity_data)[0]
|
||||
elif len(self.entity_data) == 8:
|
||||
return unpack('!d', self.entity_data)[0]
|
||||
return 0.0
|
||||
|
||||
|
||||
def get_data(self):
|
||||
return self.entity_data
|
||||
|
||||
|
||||
def get_utf8(self):
|
||||
return unicode(self.entity_data, 'utf-8', 'replace')
|
||||
|
||||
|
||||
def get_str(self):
|
||||
return unicode(self.entity_data, 'ascii', 'replace')
|
||||
|
||||
|
||||
def get_id(self):
|
||||
return self.entity_id
|
||||
|
||||
|
||||
def get_str_id(self):
|
||||
return self.entity_str
|
||||
|
||||
|
||||
def get_len(self):
|
||||
return self.entity_len
|
||||
|
||||
|
||||
def get_total_len(self):
|
||||
return self.entity_len + self.id_len + self.len_size
|
||||
|
||||
|
||||
def get_header_len(self):
|
||||
return self.id_len + self.len_size
|
||||
|
||||
|
||||
|
||||
class Matroska(core.AVContainer):
|
||||
"""
|
||||
Matroska video and audio parser. If at least one video stream is
|
||||
detected it will set the type to MEDIA_AV.
|
||||
"""
|
||||
def __init__(self, file):
|
||||
core.AVContainer.__init__(self)
|
||||
self.samplerate = 1
|
||||
|
||||
self.file = file
|
||||
# Read enough that we're likely to get the full seekhead (FIXME: kludge)
|
||||
buffer = file.read(2000)
|
||||
if len(buffer) == 0:
|
||||
# Regular File end
|
||||
raise ParseError()
|
||||
|
||||
# Check the Matroska header
|
||||
header = EbmlEntity(buffer)
|
||||
if header.get_id() != MATROSKA_HEADER_ID:
|
||||
raise ParseError()
|
||||
|
||||
log.debug("HEADER ID found %08X" % header.get_id() )
|
||||
self.mime = 'video/x-matroska'
|
||||
self.type = 'Matroska'
|
||||
self.has_idx = False
|
||||
self.objects_by_uid = {}
|
||||
|
||||
# Now get the segment
|
||||
self.segment = segment = EbmlEntity(buffer[header.get_total_len():])
|
||||
# Record file offset of segment data for seekheads
|
||||
self.segment.offset = header.get_total_len() + segment.get_header_len()
|
||||
if segment.get_id() != MATROSKA_SEGMENT_ID:
|
||||
log.debug("SEGMENT ID not found %08X" % segment.get_id())
|
||||
return
|
||||
|
||||
log.debug("SEGMENT ID found %08X" % segment.get_id())
|
||||
try:
|
||||
for elem in self.process_one_level(segment):
|
||||
if elem.get_id() == MATROSKA_SEEKHEAD_ID:
|
||||
self.process_elem(elem)
|
||||
except ParseError:
|
||||
pass
|
||||
|
||||
if not self.has_idx:
|
||||
log.warning('File has no index')
|
||||
self._set('corrupt', True)
|
||||
|
||||
def process_elem(self, elem):
|
||||
elem_id = elem.get_id()
|
||||
log.debug('BEGIN: process element %s' % hex(elem_id))
|
||||
if elem_id == MATROSKA_SEGMENT_INFO_ID:
|
||||
duration = 0
|
||||
scalecode = 1000000.0
|
||||
|
||||
for ielem in self.process_one_level(elem):
|
||||
ielem_id = ielem.get_id()
|
||||
if ielem_id == MATROSKA_TIMECODESCALE_ID:
|
||||
scalecode = ielem.get_value()
|
||||
elif ielem_id == MATROSKA_DURATION_ID:
|
||||
duration = ielem.get_float_value()
|
||||
elif ielem_id == MATROSKA_TITLE_ID:
|
||||
self.title = ielem.get_utf8()
|
||||
elif ielem_id == MATROSKA_DATE_UTC_ID:
|
||||
timestamp = unpack('!q', ielem.get_data())[0] / 10.0**9
|
||||
# Date is offset 2001-01-01 00:00:00 (timestamp 978307200.0)
|
||||
self.timestamp = int(timestamp + 978307200)
|
||||
|
||||
self.length = duration * scalecode / 1000000000.0
|
||||
|
||||
elif elem_id == MATROSKA_TRACKS_ID:
|
||||
self.process_tracks(elem)
|
||||
|
||||
elif elem_id == MATROSKA_CHAPTERS_ID:
|
||||
self.process_chapters(elem)
|
||||
|
||||
elif elem_id == MATROSKA_ATTACHMENTS_ID:
|
||||
self.process_attachments(elem)
|
||||
|
||||
elif elem_id == MATROSKA_SEEKHEAD_ID:
|
||||
self.process_seekhead(elem)
|
||||
|
||||
elif elem_id == MATROSKA_TAGS_ID:
|
||||
self.process_tags(elem)
|
||||
|
||||
elif elem_id == MATROSKA_CUES_ID:
|
||||
self.has_idx = True
|
||||
|
||||
log.debug('END: process element %s' % hex(elem_id))
|
||||
return True
|
||||
|
||||
|
||||
def process_seekhead(self, elem):
|
||||
for seek_elem in self.process_one_level(elem):
|
||||
if seek_elem.get_id() != MATROSKA_SEEK_ID:
|
||||
continue
|
||||
for sub_elem in self.process_one_level(seek_elem):
|
||||
if sub_elem.get_id() == MATROSKA_SEEKID_ID:
|
||||
if sub_elem.get_value() == MATROSKA_CLUSTER_ID:
|
||||
# Not interested in these.
|
||||
return
|
||||
|
||||
elif sub_elem.get_id() == MATROSKA_SEEK_POSITION_ID:
|
||||
self.file.seek(self.segment.offset + sub_elem.get_value())
|
||||
buffer = self.file.read(100)
|
||||
try:
|
||||
elem = EbmlEntity(buffer)
|
||||
except ParseError:
|
||||
continue
|
||||
|
||||
# Fetch all data necessary for this element.
|
||||
elem.add_data(self.file.read(elem.ebml_length))
|
||||
self.process_elem(elem)
|
||||
|
||||
|
||||
def process_tracks(self, tracks):
|
||||
tracksbuf = tracks.get_data()
|
||||
index = 0
|
||||
while index < tracks.get_len():
|
||||
trackelem = EbmlEntity(tracksbuf[index:])
|
||||
log.debug ("ELEMENT %X found" % trackelem.get_id())
|
||||
self.process_track(trackelem)
|
||||
index += trackelem.get_total_len() + trackelem.get_crc_len()
|
||||
|
||||
|
||||
def process_one_level(self, item):
|
||||
buf = item.get_data()
|
||||
index = 0
|
||||
while index < item.get_len():
|
||||
if len(buf[index:]) == 0:
|
||||
break
|
||||
elem = EbmlEntity(buf[index:])
|
||||
yield elem
|
||||
index += elem.get_total_len() + elem.get_crc_len()
|
||||
|
||||
def set_track_defaults(self, track):
|
||||
track.language = 'eng'
|
||||
|
||||
def process_track(self, track):
|
||||
# Collapse generator into a list since we need to iterate over it
|
||||
# twice.
|
||||
elements = [x for x in self.process_one_level(track)]
|
||||
track_type = [x.get_value() for x in elements if x.get_id() == MATROSKA_TRACK_TYPE_ID]
|
||||
if not track_type:
|
||||
log.debug('Bad track: no type id found')
|
||||
return
|
||||
|
||||
track_type = track_type[0]
|
||||
track = None
|
||||
|
||||
if track_type == MATROSKA_VIDEO_TRACK:
|
||||
log.debug("Video track found")
|
||||
track = self.process_video_track(elements)
|
||||
elif track_type == MATROSKA_AUDIO_TRACK:
|
||||
log.debug("Audio track found")
|
||||
track = self.process_audio_track(elements)
|
||||
elif track_type == MATROSKA_SUBTITLES_TRACK:
|
||||
log.debug("Subtitle track found")
|
||||
track = core.Subtitle()
|
||||
self.set_track_defaults(track)
|
||||
track.id = len(self.subtitles)
|
||||
self.subtitles.append(track)
|
||||
for elem in elements:
|
||||
self.process_track_common(elem, track)
|
||||
|
||||
|
||||
def process_track_common(self, elem, track):
|
||||
elem_id = elem.get_id()
|
||||
if elem_id == MATROSKA_TRACK_LANGUAGE_ID:
|
||||
track.language = elem.get_str()
|
||||
log.debug("Track language found: %s" % track.language)
|
||||
elif elem_id == MATROSKA_NAME_ID:
|
||||
track.title = elem.get_utf8()
|
||||
elif elem_id == MATROSKA_TRACK_NUMBER_ID:
|
||||
track.trackno = elem.get_value()
|
||||
elif elem_id == MATROSKA_TRACK_FLAG_ENABLED_ID:
|
||||
track.enabled = bool(elem.get_value())
|
||||
elif elem_id == MATROSKA_TRACK_FLAG_DEFAULT_ID:
|
||||
track.default = bool(elem.get_value())
|
||||
elif elem_id == MATROSKA_CODEC_ID:
|
||||
track.codec = elem.get_str()
|
||||
elif elem_id == MATROSKA_CODEC_PRIVATE_ID:
|
||||
track.codec_private = elem.get_data()
|
||||
elif elem_id == MATROSKA_TRACK_UID_ID:
|
||||
self.objects_by_uid[elem.get_value()] = track
|
||||
|
||||
|
||||
def process_video_track(self, elements):
|
||||
track = core.VideoStream()
|
||||
# Defaults
|
||||
track.codec = u'Unknown'
|
||||
track.fps = 0
|
||||
self.set_track_defaults(track)
|
||||
|
||||
for elem in elements:
|
||||
elem_id = elem.get_id()
|
||||
if elem_id == MATROSKA_CODEC_ID:
|
||||
track.codec = elem.get_str()
|
||||
|
||||
elif elem_id == MATROSKA_FRAME_DURATION_ID:
|
||||
try:
|
||||
track.fps = 1 / (pow(10, -9) * (elem.get_value()))
|
||||
except ZeroDivisionError:
|
||||
pass
|
||||
|
||||
elif elem_id == MATROSKA_VIDEO_SETTINGS_ID:
|
||||
d_width = d_height = None
|
||||
for settings_elem in self.process_one_level(elem):
|
||||
settings_elem_id = settings_elem.get_id()
|
||||
if settings_elem_id == MATROSKA_VIDEO_WIDTH_ID:
|
||||
track.width = settings_elem.get_value()
|
||||
elif settings_elem_id == MATROSKA_VIDEO_HEIGHT_ID:
|
||||
track.height = settings_elem.get_value()
|
||||
elif settings_elem_id == MATROSKA_VIDEO_DISPLAY_WIDTH_ID:
|
||||
d_width = settings_elem.get_value()
|
||||
elif settings_elem_id == MATROSKA_VIDEO_DISPLAY_HEIGHT_ID:
|
||||
d_height = settings_elem.get_value()
|
||||
elif settings_elem_id == MATROSKA_VIDEO_INTERLACED_ID:
|
||||
value = int(settings_elem.get_value())
|
||||
self._set('interlaced', value)
|
||||
|
||||
if None not in [d_width, d_height]:
|
||||
track.aspect = float(d_width) / d_height
|
||||
|
||||
else:
|
||||
self.process_track_common(elem, track)
|
||||
|
||||
# convert codec information
|
||||
# http://haali.cs.msu.ru/mkv/codecs.pdf
|
||||
if track.codec in FOURCCMap:
|
||||
track.codec = FOURCCMap[track.codec]
|
||||
elif '/' in track.codec and track.codec.split('/')[0] + '/' in FOURCCMap:
|
||||
track.codec = FOURCCMap[track.codec.split('/')[0] + '/']
|
||||
elif track.codec.endswith('FOURCC') and len(track.codec_private or '') == 40:
|
||||
track.codec = track.codec_private[16:20]
|
||||
elif track.codec.startswith('V_REAL/'):
|
||||
track.codec = track.codec[7:]
|
||||
elif track.codec.startswith('V_'):
|
||||
# FIXME: add more video codecs here
|
||||
track.codec = track.codec[2:]
|
||||
|
||||
track.id = len(self.video)
|
||||
self.video.append(track)
|
||||
return track
|
||||
|
||||
|
||||
def process_audio_track(self, elements):
|
||||
track = core.AudioStream()
|
||||
track.codec = u'Unknown'
|
||||
self.set_track_defaults(track)
|
||||
|
||||
for elem in elements:
|
||||
elem_id = elem.get_id()
|
||||
if elem_id == MATROSKA_CODEC_ID:
|
||||
track.codec = elem.get_str()
|
||||
elif elem_id == MATROSKA_AUDIO_SETTINGS_ID:
|
||||
for settings_elem in self.process_one_level(elem):
|
||||
settings_elem_id = settings_elem.get_id()
|
||||
if settings_elem_id == MATROSKA_AUDIO_SAMPLERATE_ID:
|
||||
track.samplerate = settings_elem.get_float_value()
|
||||
elif settings_elem_id == MATROSKA_AUDIO_CHANNELS_ID:
|
||||
track.channels = settings_elem.get_value()
|
||||
else:
|
||||
self.process_track_common(elem, track)
|
||||
|
||||
|
||||
if track.codec in FOURCCMap:
|
||||
track.codec = FOURCCMap[track.codec]
|
||||
elif '/' in track.codec and track.codec.split('/')[0] + '/' in FOURCCMap:
|
||||
track.codec = FOURCCMap[track.codec.split('/')[0] + '/']
|
||||
elif track.codec.startswith('A_'):
|
||||
track.codec = track.codec[2:]
|
||||
|
||||
track.id = len(self.audio)
|
||||
self.audio.append(track)
|
||||
return track
|
||||
|
||||
|
||||
def process_chapters(self, chapters):
|
||||
elements = self.process_one_level(chapters)
|
||||
for elem in elements:
|
||||
if elem.get_id() == MATROSKA_EDITION_ENTRY_ID:
|
||||
buf = elem.get_data()
|
||||
index = 0
|
||||
while index < elem.get_len():
|
||||
sub_elem = EbmlEntity(buf[index:])
|
||||
if sub_elem.get_id() == MATROSKA_CHAPTER_ATOM_ID:
|
||||
self.process_chapter_atom(sub_elem)
|
||||
index += sub_elem.get_total_len() + sub_elem.get_crc_len()
|
||||
|
||||
|
||||
def process_chapter_atom(self, atom):
|
||||
elements = self.process_one_level(atom)
|
||||
chap = core.Chapter()
|
||||
|
||||
for elem in elements:
|
||||
elem_id = elem.get_id()
|
||||
if elem_id == MATROSKA_CHAPTER_TIME_START_ID:
|
||||
# Scale timecode to seconds (float)
|
||||
chap.pos = elem.get_value() / 1000000 / 1000.0
|
||||
elif elem_id == MATROSKA_CHAPTER_FLAG_ENABLED_ID:
|
||||
chap.enabled = elem.get_value()
|
||||
elif elem_id == MATROSKA_CHAPTER_DISPLAY_ID:
|
||||
# Matroska supports multiple (chapter name, language) pairs for
|
||||
# each chapter, so chapter names can be internationalized. This
|
||||
# logic will only take the last one in the list.
|
||||
for display_elem in self.process_one_level(elem):
|
||||
if display_elem.get_id() == MATROSKA_CHAPTER_STRING_ID:
|
||||
chap.name = display_elem.get_utf8()
|
||||
elif elem_id == MATROSKA_CHAPTER_UID_ID:
|
||||
self.objects_by_uid[elem.get_value()] = chap
|
||||
|
||||
log.debug('Chapter "%s" found', chap.name)
|
||||
chap.id = len(self.chapters)
|
||||
self.chapters.append(chap)
|
||||
|
||||
|
||||
def process_attachments(self, attachments):
|
||||
buf = attachments.get_data()
|
||||
index = 0
|
||||
while index < attachments.get_len():
|
||||
elem = EbmlEntity(buf[index:])
|
||||
if elem.get_id() == MATROSKA_ATTACHED_FILE_ID:
|
||||
self.process_attachment(elem)
|
||||
index += elem.get_total_len() + elem.get_crc_len()
|
||||
|
||||
|
||||
def process_attachment(self, attachment):
|
||||
elements = self.process_one_level(attachment)
|
||||
name = desc = mimetype = ""
|
||||
data = None
|
||||
|
||||
for elem in elements:
|
||||
elem_id = elem.get_id()
|
||||
if elem_id == MATROSKA_FILE_NAME_ID:
|
||||
name = elem.get_utf8()
|
||||
elif elem_id == MATROSKA_FILE_DESC_ID:
|
||||
desc = elem.get_utf8()
|
||||
elif elem_id == MATROSKA_FILE_MIME_TYPE_ID:
|
||||
mimetype = elem.get_data()
|
||||
elif elem_id == MATROSKA_FILE_DATA_ID:
|
||||
data = elem.get_data()
|
||||
|
||||
# Right now we only support attachments that could be cover images.
|
||||
# Make a guess to see if this attachment is a cover image.
|
||||
if mimetype.startswith("image/") and u"cover" in (name+desc).lower() and data:
|
||||
self.thumbnail = data
|
||||
|
||||
log.debug('Attachment "%s" found' % name)
|
||||
|
||||
|
||||
def process_tags(self, tags):
|
||||
# Tags spec: http://www.matroska.org/technical/specs/tagging/index.html
|
||||
# Iterate over Tags children. Tags element children is a
|
||||
# Tag element (whose children are SimpleTags) and a Targets element
|
||||
# whose children specific what objects the tags apply to.
|
||||
for tag_elem in self.process_one_level(tags):
|
||||
# Start a new dict to hold all SimpleTag elements.
|
||||
tags_dict = core.Tags()
|
||||
# A list of target uids this tags dict applies too. If empty,
|
||||
# tags are global.
|
||||
targets = []
|
||||
for sub_elem in self.process_one_level(tag_elem):
|
||||
if sub_elem.get_id() == MATROSKA_SIMPLE_TAG_ID:
|
||||
self.process_simple_tag(sub_elem, tags_dict)
|
||||
elif sub_elem.get_id() == MATROSKA_TARGETS_ID:
|
||||
# Targets element: if there is no uid child (track uid,
|
||||
# chapter uid, etc.) then the tags dict applies to the
|
||||
# whole file (top-level Media object).
|
||||
for target_elem in self.process_one_level(sub_elem):
|
||||
target_elem_id = target_elem.get_id()
|
||||
if target_elem_id in (MATRSOKA_TAGS_TRACK_UID_ID, MATRSOKA_TAGS_EDITION_UID_ID,
|
||||
MATRSOKA_TAGS_CHAPTER_UID_ID, MATRSOKA_TAGS_ATTACHMENT_UID_ID):
|
||||
targets.append(target_elem.get_value())
|
||||
elif target_elem_id == MATROSKA_TARGET_TYPE_VALUE_ID:
|
||||
# Target types not supported for now. (Unclear how this
|
||||
# would fit with kaa.metadata.)
|
||||
pass
|
||||
if targets:
|
||||
# Assign tags to all listed uids
|
||||
for target in targets:
|
||||
try:
|
||||
self.objects_by_uid[target].tags.update(tags_dict)
|
||||
self.tags_to_attributes(self.objects_by_uid[target], tags_dict)
|
||||
except KeyError:
|
||||
log.warning('Tags assigned to unknown/unsupported target uid %d', target)
|
||||
else:
|
||||
self.tags.update(tags_dict)
|
||||
self.tags_to_attributes(self, tags_dict)
|
||||
|
||||
|
||||
def process_simple_tag(self, simple_tag_elem, tags_dict):
|
||||
"""
|
||||
Returns a dict representing the Tag element.
|
||||
"""
|
||||
name = lang = value = children = None
|
||||
binary = False
|
||||
for elem in self.process_one_level(simple_tag_elem):
|
||||
elem_id = elem.get_id()
|
||||
if elem_id == MATROSKA_TAG_NAME_ID:
|
||||
name = elem.get_utf8().lower()
|
||||
elif elem_id == MATROSKA_TAG_STRING_ID:
|
||||
value = elem.get_utf8()
|
||||
elif elem_id == MATROSKA_TAG_BINARY_ID:
|
||||
value = elem.get_data()
|
||||
binary = True
|
||||
elif elem_id == MATROSKA_TAG_LANGUAGE_ID:
|
||||
lang = elem.get_utf8()
|
||||
elif elem_id == MATROSKA_SIMPLE_TAG_ID:
|
||||
if children is None:
|
||||
children = core.Tags()
|
||||
self.process_simple_tag(elem, children)
|
||||
|
||||
if children:
|
||||
# Convert ourselves to a Tags object.
|
||||
children.value = value
|
||||
children.langcode = lang
|
||||
value = children
|
||||
else:
|
||||
if name.startswith('date_'):
|
||||
# Try to convert date to a datetime object.
|
||||
value = matroska_date_to_datetime(value)
|
||||
value = core.Tag(value, lang, binary)
|
||||
|
||||
if name in tags_dict:
|
||||
# Multiple items of this tag name.
|
||||
if not isinstance(tags_dict[name], list):
|
||||
# Convert to a list
|
||||
tags_dict[name] = [tags_dict[name]]
|
||||
# Append to list
|
||||
tags_dict[name].append(value)
|
||||
else:
|
||||
tags_dict[name] = value
|
||||
|
||||
|
||||
def tags_to_attributes(self, obj, tags):
|
||||
# Convert tags to core attributes.
|
||||
for name, tag in tags.items():
|
||||
if isinstance(tag, dict):
|
||||
# Nested tags dict, recurse.
|
||||
self.tags_to_attributes(obj, tag)
|
||||
continue
|
||||
elif name not in TAGS_MAP:
|
||||
continue
|
||||
|
||||
attr, filter = TAGS_MAP[name]
|
||||
if attr not in obj._keys and attr not in self._keys:
|
||||
# Tag is not in any core attribute for this object or global,
|
||||
# so skip.
|
||||
continue
|
||||
|
||||
# Pull value out of Tag object or list of Tag objects.
|
||||
value = [item.value for item in tag] if isinstance(tag, list) else tag.value
|
||||
if filter:
|
||||
try:
|
||||
value = [filter(item) for item in value] if isinstance(value, list) else filter(value)
|
||||
except Exception, e:
|
||||
log.warning('Failed to convert tag to core attribute: %s', e)
|
||||
# Special handling for tv series recordings. The 'title' tag
|
||||
# can be used for both the series and the episode name. The
|
||||
# same is true for trackno which may refer to the season
|
||||
# and the episode number. Therefore, if we find these
|
||||
# attributes already set we try some guessing.
|
||||
if attr == 'trackno' and getattr(self, attr) is not None:
|
||||
# delete trackno and save season and episode
|
||||
self.season = self.trackno
|
||||
self.episode = value
|
||||
self.trackno = None
|
||||
continue
|
||||
if attr == 'title' and getattr(self, attr) is not None:
|
||||
# store current value of title as series and use current
|
||||
# value of title as title
|
||||
self.series = self.title
|
||||
if attr in obj._keys:
|
||||
setattr(obj, attr, value)
|
||||
else:
|
||||
setattr(self, attr, value)
|
||||
|
||||
|
||||
Parser = Matroska
|
||||
@@ -0,0 +1,476 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2003-2007 Thomas Schueppel <stain@acm.org>
|
||||
# Copyright (C) 2003-2007 Dirk Meyer <dischi@freevo.org>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__all__ = ['Parser']
|
||||
|
||||
import zlib
|
||||
import logging
|
||||
import StringIO
|
||||
import struct
|
||||
from exceptions import *
|
||||
import core
|
||||
|
||||
# get logging object
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# http://developer.apple.com/documentation/QuickTime/QTFF/index.html
|
||||
# http://developer.apple.com/documentation/QuickTime/QTFF/QTFFChap4/\
|
||||
# chapter_5_section_2.html#//apple_ref/doc/uid/TP40000939-CH206-BBCBIICE
|
||||
# Note: May need to define custom log level to work like ATOM_DEBUG did here
|
||||
|
||||
QTUDTA = {
|
||||
'nam': 'title',
|
||||
'aut': 'artist',
|
||||
'cpy': 'copyright'
|
||||
}
|
||||
|
||||
QTLANGUAGES = {
|
||||
0: "en",
|
||||
1: "fr",
|
||||
2: "de",
|
||||
3: "it",
|
||||
4: "nl",
|
||||
5: "sv",
|
||||
6: "es",
|
||||
7: "da",
|
||||
8: "pt",
|
||||
9: "no",
|
||||
10: "he",
|
||||
11: "ja",
|
||||
12: "ar",
|
||||
13: "fi",
|
||||
14: "el",
|
||||
15: "is",
|
||||
16: "mt",
|
||||
17: "tr",
|
||||
18: "hr",
|
||||
19: "Traditional Chinese",
|
||||
20: "ur",
|
||||
21: "hi",
|
||||
22: "th",
|
||||
23: "ko",
|
||||
24: "lt",
|
||||
25: "pl",
|
||||
26: "hu",
|
||||
27: "et",
|
||||
28: "lv",
|
||||
29: "Lappish",
|
||||
30: "fo",
|
||||
31: "Farsi",
|
||||
32: "ru",
|
||||
33: "Simplified Chinese",
|
||||
34: "Flemish",
|
||||
35: "ga",
|
||||
36: "sq",
|
||||
37: "ro",
|
||||
38: "cs",
|
||||
39: "sk",
|
||||
40: "sl",
|
||||
41: "yi",
|
||||
42: "sr",
|
||||
43: "mk",
|
||||
44: "bg",
|
||||
45: "uk",
|
||||
46: "be",
|
||||
47: "uz",
|
||||
48: "kk",
|
||||
49: "az",
|
||||
50: "AzerbaijanAr",
|
||||
51: "hy",
|
||||
52: "ka",
|
||||
53: "mo",
|
||||
54: "ky",
|
||||
55: "tg",
|
||||
56: "tk",
|
||||
57: "mn",
|
||||
58: "MongolianCyr",
|
||||
59: "ps",
|
||||
60: "ku",
|
||||
61: "ks",
|
||||
62: "sd",
|
||||
63: "bo",
|
||||
64: "ne",
|
||||
65: "sa",
|
||||
66: "mr",
|
||||
67: "bn",
|
||||
68: "as",
|
||||
69: "gu",
|
||||
70: "pa",
|
||||
71: "or",
|
||||
72: "ml",
|
||||
73: "kn",
|
||||
74: "ta",
|
||||
75: "te",
|
||||
76: "si",
|
||||
77: "my",
|
||||
78: "Khmer",
|
||||
79: "lo",
|
||||
80: "vi",
|
||||
81: "id",
|
||||
82: "tl",
|
||||
83: "MalayRoman",
|
||||
84: "MalayArabic",
|
||||
85: "am",
|
||||
86: "ti",
|
||||
87: "om",
|
||||
88: "so",
|
||||
89: "sw",
|
||||
90: "Ruanda",
|
||||
91: "Rundi",
|
||||
92: "Chewa",
|
||||
93: "mg",
|
||||
94: "eo",
|
||||
128: "cy",
|
||||
129: "eu",
|
||||
130: "ca",
|
||||
131: "la",
|
||||
132: "qu",
|
||||
133: "gn",
|
||||
134: "ay",
|
||||
135: "tt",
|
||||
136: "ug",
|
||||
137: "Dzongkha",
|
||||
138: "JavaneseRom",
|
||||
}
|
||||
|
||||
class MPEG4(core.AVContainer):
|
||||
"""
|
||||
Parser for the MP4 container format. This format is mostly
|
||||
identical to Apple Quicktime and 3GP files. It maps to mp4, mov,
|
||||
qt and some other extensions.
|
||||
"""
|
||||
table_mapping = {'QTUDTA': QTUDTA}
|
||||
|
||||
def __init__(self, file):
|
||||
core.AVContainer.__init__(self)
|
||||
self._references = []
|
||||
|
||||
self.mime = 'video/quicktime'
|
||||
self.type = 'Quicktime Video'
|
||||
h = file.read(8)
|
||||
try:
|
||||
(size, type) = struct.unpack('>I4s',h)
|
||||
except struct.error:
|
||||
# EOF.
|
||||
raise ParseError()
|
||||
|
||||
if type == 'ftyp':
|
||||
# file type information
|
||||
if size >= 12:
|
||||
# this should always happen
|
||||
if file.read(4) != 'qt ':
|
||||
# not a quicktime movie, it is a mpeg4 container
|
||||
self.mime = 'video/mp4'
|
||||
self.type = 'MPEG-4 Video'
|
||||
size -= 4
|
||||
file.seek(size-8, 1)
|
||||
h = file.read(8)
|
||||
(size, type) = struct.unpack('>I4s',h)
|
||||
|
||||
while type in ['mdat', 'skip']:
|
||||
# movie data at the beginning, skip
|
||||
file.seek(size-8, 1)
|
||||
h = file.read(8)
|
||||
(size, type) = struct.unpack('>I4s',h)
|
||||
|
||||
if not type in ['moov', 'wide', 'free']:
|
||||
log.debug('invalid header: %r' % type)
|
||||
raise ParseError()
|
||||
|
||||
# Extended size
|
||||
if size == 1:
|
||||
size = struct.unpack('>Q', file.read(8))
|
||||
|
||||
# Back over the atom header we just read, since _readatom expects the
|
||||
# file position to be at the start of an atom.
|
||||
file.seek(-8, 1)
|
||||
while self._readatom(file):
|
||||
pass
|
||||
|
||||
if self._references:
|
||||
self._set('references', self._references)
|
||||
|
||||
|
||||
def _readatom(self, file):
|
||||
s = file.read(8)
|
||||
if len(s) < 8:
|
||||
return 0
|
||||
|
||||
atomsize,atomtype = struct.unpack('>I4s', s)
|
||||
if not str(atomtype).decode('latin1').isalnum():
|
||||
# stop at nonsense data
|
||||
return 0
|
||||
|
||||
log.debug('%s [%X]' % (atomtype,atomsize))
|
||||
|
||||
if atomtype == 'udta':
|
||||
# Userdata (Metadata)
|
||||
pos = 0
|
||||
tabl = {}
|
||||
i18ntabl = {}
|
||||
atomdata = file.read(atomsize-8)
|
||||
while pos < atomsize-12:
|
||||
(datasize, datatype) = struct.unpack('>I4s', atomdata[pos:pos+8])
|
||||
if ord(datatype[0]) == 169:
|
||||
# i18n Metadata...
|
||||
mypos = 8+pos
|
||||
while mypos + 4 < datasize+pos:
|
||||
# first 4 Bytes are i18n header
|
||||
(tlen, lang) = struct.unpack('>HH', atomdata[mypos:mypos+4])
|
||||
i18ntabl[lang] = i18ntabl.get(lang, {})
|
||||
l = atomdata[mypos+4:mypos+tlen+4]
|
||||
i18ntabl[lang][datatype[1:]] = l
|
||||
mypos += tlen+4
|
||||
elif datatype == 'WLOC':
|
||||
# Drop Window Location
|
||||
pass
|
||||
else:
|
||||
if ord(atomdata[pos+8:pos+datasize][0]) > 1:
|
||||
tabl[datatype] = atomdata[pos+8:pos+datasize]
|
||||
pos += datasize
|
||||
if len(i18ntabl.keys()) > 0:
|
||||
for k in i18ntabl.keys():
|
||||
if QTLANGUAGES.has_key(k) and QTLANGUAGES[k] == 'en':
|
||||
self._appendtable('QTUDTA', i18ntabl[k])
|
||||
self._appendtable('QTUDTA', tabl)
|
||||
else:
|
||||
log.debug('NO i18')
|
||||
self._appendtable('QTUDTA', tabl)
|
||||
|
||||
elif atomtype == 'trak':
|
||||
atomdata = file.read(atomsize-8)
|
||||
pos = 0
|
||||
trackinfo = {}
|
||||
tracktype = None
|
||||
while pos < atomsize-8:
|
||||
(datasize, datatype) = struct.unpack('>I4s', atomdata[pos:pos+8])
|
||||
|
||||
if datatype == 'tkhd':
|
||||
tkhd = struct.unpack('>6I8x4H36xII', atomdata[pos+8:pos+datasize])
|
||||
trackinfo['width'] = tkhd[10] >> 16
|
||||
trackinfo['height'] = tkhd[11] >> 16
|
||||
trackinfo['id'] = tkhd[3]
|
||||
|
||||
try:
|
||||
# XXX Timestamp of Seconds is since January 1st 1904!
|
||||
# XXX 2082844800 is the difference between Unix and
|
||||
# XXX Apple time. FIXME to work on Apple, too
|
||||
self.timestamp = int(tkhd[1]) - 2082844800
|
||||
except Exception, e:
|
||||
log.exception('There was trouble extracting timestamp')
|
||||
|
||||
elif datatype == 'mdia':
|
||||
pos += 8
|
||||
datasize -= 8
|
||||
log.debug('--> mdia information')
|
||||
|
||||
while datasize:
|
||||
mdia = struct.unpack('>I4s', atomdata[pos:pos+8])
|
||||
if mdia[1] == 'mdhd':
|
||||
# Parse based on version of mdhd header. See
|
||||
# http://wiki.multimedia.cx/index.php?title=QuickTime_container#mdhd
|
||||
ver = ord(atomdata[pos + 8])
|
||||
if ver == 0:
|
||||
mdhd = struct.unpack('>IIIIIhh', atomdata[pos+8:pos+8+24])
|
||||
elif ver == 1:
|
||||
mdhd = struct.unpack('>IQQIQhh', atomdata[pos+8:pos+8+36])
|
||||
else:
|
||||
mdhd = None
|
||||
|
||||
if mdhd:
|
||||
# duration / time scale
|
||||
trackinfo['length'] = mdhd[4] / mdhd[3]
|
||||
if mdhd[5] in QTLANGUAGES:
|
||||
trackinfo['language'] = QTLANGUAGES[mdhd[5]]
|
||||
# mdhd[6] == quality
|
||||
self.length = max(self.length, mdhd[4] / mdhd[3])
|
||||
elif mdia[1] == 'minf':
|
||||
# minf has only atoms inside
|
||||
pos -= (mdia[0] - 8)
|
||||
datasize += (mdia[0] - 8)
|
||||
elif mdia[1] == 'stbl':
|
||||
# stbl has only atoms inside
|
||||
pos -= (mdia[0] - 8)
|
||||
datasize += (mdia[0] - 8)
|
||||
elif mdia[1] == 'hdlr':
|
||||
hdlr = struct.unpack('>I4s4s', atomdata[pos+8:pos+8+12])
|
||||
if hdlr[1] == 'mhlr':
|
||||
if hdlr[2] == 'vide':
|
||||
tracktype = 'video'
|
||||
if hdlr[2] == 'soun':
|
||||
tracktype = 'audio'
|
||||
elif mdia[1] == 'stsd':
|
||||
stsd = struct.unpack('>2I', atomdata[pos+8:pos+8+8])
|
||||
if stsd[1] > 0:
|
||||
codec = atomdata[pos+16:pos+16+8]
|
||||
codec = struct.unpack('>I4s', codec)
|
||||
trackinfo['codec'] = codec[1]
|
||||
if codec[1] == 'jpeg':
|
||||
tracktype = 'image'
|
||||
elif mdia[1] == 'dinf':
|
||||
dref = struct.unpack('>I4s', atomdata[pos+8:pos+8+8])
|
||||
log.debug(' --> %s, %s (useless)' % mdia)
|
||||
if dref[1] == 'dref':
|
||||
num = struct.unpack('>I', atomdata[pos+20:pos+20+4])[0]
|
||||
rpos = pos+20+4
|
||||
for ref in range(num):
|
||||
# FIXME: do somthing if this references
|
||||
ref = struct.unpack('>I3s', atomdata[rpos:rpos+7])
|
||||
data = atomdata[rpos+7:rpos+ref[0]]
|
||||
rpos += ref[0]
|
||||
else:
|
||||
if mdia[1].startswith('st'):
|
||||
log.debug(' --> %s, %s (sample)' % mdia)
|
||||
elif mdia[1] == 'vmhd' and not tracktype:
|
||||
# indicates that this track is video
|
||||
tracktype = 'video'
|
||||
elif mdia[1] in ['vmhd', 'smhd'] and not tracktype:
|
||||
# indicates that this track is audio
|
||||
tracktype = 'audio'
|
||||
else:
|
||||
log.debug(' --> %s, %s (unknown)' % mdia)
|
||||
|
||||
pos += mdia[0]
|
||||
datasize -= mdia[0]
|
||||
|
||||
elif datatype == 'udta':
|
||||
log.debug(struct.unpack('>I4s', atomdata[:8]))
|
||||
else:
|
||||
if datatype == 'edts':
|
||||
log.debug('--> %s [%d] (edit list)' % \
|
||||
(datatype, datasize))
|
||||
else:
|
||||
log.debug('--> %s [%d] (unknown)' % \
|
||||
(datatype, datasize))
|
||||
pos += datasize
|
||||
|
||||
info = None
|
||||
if tracktype == 'video':
|
||||
info = core.VideoStream()
|
||||
self.video.append(info)
|
||||
if tracktype == 'audio':
|
||||
info = core.AudioStream()
|
||||
self.audio.append(info)
|
||||
if info:
|
||||
for key, value in trackinfo.items():
|
||||
setattr(info, key, value)
|
||||
|
||||
elif atomtype == 'mvhd':
|
||||
# movie header
|
||||
mvhd = struct.unpack('>6I2h', file.read(28))
|
||||
self.length = max(self.length, mvhd[4] / mvhd[3])
|
||||
self.volume = mvhd[6]
|
||||
file.seek(atomsize-8-28,1)
|
||||
|
||||
|
||||
elif atomtype == 'cmov':
|
||||
# compressed movie
|
||||
datasize, atomtype = struct.unpack('>I4s', file.read(8))
|
||||
if not atomtype == 'dcom':
|
||||
return atomsize
|
||||
|
||||
method = struct.unpack('>4s', file.read(datasize-8))[0]
|
||||
|
||||
datasize, atomtype = struct.unpack('>I4s', file.read(8))
|
||||
if not atomtype == 'cmvd':
|
||||
return atomsize
|
||||
|
||||
if method == 'zlib':
|
||||
data = file.read(datasize-8)
|
||||
try:
|
||||
decompressed = zlib.decompress(data)
|
||||
except Exception, e:
|
||||
try:
|
||||
decompressed = zlib.decompress(data[4:])
|
||||
except Exception, e:
|
||||
log.exception('There was a proble decompressiong atom')
|
||||
return atomsize
|
||||
|
||||
decompressedIO = StringIO.StringIO(decompressed)
|
||||
while self._readatom(decompressedIO):
|
||||
pass
|
||||
|
||||
else:
|
||||
log.info('unknown compression %s' % method)
|
||||
# unknown compression method
|
||||
file.seek(datasize-8,1)
|
||||
|
||||
elif atomtype == 'moov':
|
||||
# decompressed movie info
|
||||
while self._readatom(file):
|
||||
pass
|
||||
|
||||
elif atomtype == 'mdat':
|
||||
pos = file.tell() + atomsize - 8
|
||||
# maybe there is data inside the mdat
|
||||
log.info('parsing mdat')
|
||||
while self._readatom(file):
|
||||
pass
|
||||
log.info('end of mdat')
|
||||
file.seek(pos, 0)
|
||||
|
||||
|
||||
elif atomtype == 'rmra':
|
||||
# reference list
|
||||
while self._readatom(file):
|
||||
pass
|
||||
|
||||
elif atomtype == 'rmda':
|
||||
# reference
|
||||
atomdata = file.read(atomsize-8)
|
||||
pos = 0
|
||||
url = ''
|
||||
quality = 0
|
||||
datarate = 0
|
||||
while pos < atomsize-8:
|
||||
(datasize, datatype) = struct.unpack('>I4s', atomdata[pos:pos+8])
|
||||
if datatype == 'rdrf':
|
||||
rflags, rtype, rlen = struct.unpack('>I4sI', atomdata[pos+8:pos+20])
|
||||
if rtype == 'url ':
|
||||
url = atomdata[pos+20:pos+20+rlen]
|
||||
if url.find('\0') > 0:
|
||||
url = url[:url.find('\0')]
|
||||
elif datatype == 'rmqu':
|
||||
quality = struct.unpack('>I', atomdata[pos+8:pos+12])[0]
|
||||
|
||||
elif datatype == 'rmdr':
|
||||
datarate = struct.unpack('>I', atomdata[pos+12:pos+16])[0]
|
||||
|
||||
pos += datasize
|
||||
if url:
|
||||
self._references.append((url, quality, datarate))
|
||||
|
||||
else:
|
||||
if not atomtype in ['wide', 'free']:
|
||||
log.info('unhandled base atom %s' % atomtype)
|
||||
|
||||
# Skip unknown atoms
|
||||
try:
|
||||
file.seek(atomsize-8,1)
|
||||
except IOError:
|
||||
return 0
|
||||
|
||||
return atomsize
|
||||
|
||||
|
||||
Parser = MPEG4
|
||||
@@ -0,0 +1,915 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2003-2006 Thomas Schueppel <stain@acm.org>
|
||||
# Copyright (C) 2003-2006 Dirk Meyer <dischi@freevo.org>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__all__ = ['Parser']
|
||||
|
||||
import os
|
||||
import struct
|
||||
import logging
|
||||
import stat
|
||||
from exceptions import *
|
||||
import core
|
||||
|
||||
# get logging object
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
##------------------------------------------------------------------------
|
||||
## START_CODE
|
||||
##
|
||||
## Start Codes, with 'slice' occupying 0x01..0xAF
|
||||
##------------------------------------------------------------------------
|
||||
START_CODE = {
|
||||
0x00 : 'picture_start_code',
|
||||
0xB0 : 'reserved',
|
||||
0xB1 : 'reserved',
|
||||
0xB2 : 'user_data_start_code',
|
||||
0xB3 : 'sequence_header_code',
|
||||
0xB4 : 'sequence_error_code',
|
||||
0xB5 : 'extension_start_code',
|
||||
0xB6 : 'reserved',
|
||||
0xB7 : 'sequence end',
|
||||
0xB8 : 'group of pictures',
|
||||
}
|
||||
for i in range(0x01,0xAF):
|
||||
START_CODE[i] = 'slice_start_code'
|
||||
|
||||
##------------------------------------------------------------------------
|
||||
## START CODES
|
||||
##------------------------------------------------------------------------
|
||||
PICTURE = 0x00
|
||||
USERDATA = 0xB2
|
||||
SEQ_HEAD = 0xB3
|
||||
SEQ_ERR = 0xB4
|
||||
EXT_START = 0xB5
|
||||
SEQ_END = 0xB7
|
||||
GOP = 0xB8
|
||||
|
||||
SEQ_START_CODE = 0xB3
|
||||
PACK_PKT = 0xBA
|
||||
SYS_PKT = 0xBB
|
||||
PADDING_PKT = 0xBE
|
||||
AUDIO_PKT = 0xC0
|
||||
VIDEO_PKT = 0xE0
|
||||
PRIVATE_STREAM1 = 0xBD
|
||||
PRIVATE_STREAM2 = 0xBf
|
||||
|
||||
TS_PACKET_LENGTH = 188
|
||||
TS_SYNC = 0x47
|
||||
|
||||
##------------------------------------------------------------------------
|
||||
## FRAME_RATE
|
||||
##
|
||||
## A lookup table of all the standard frame rates. Some rates adhere to
|
||||
## a particular profile that ensures compatibility with VLSI capabilities
|
||||
## of the early to mid 1990s.
|
||||
##
|
||||
## CPB
|
||||
## Constrained Parameters Bitstreams, an MPEG-1 set of sampling and
|
||||
## bitstream parameters designed to normalize decoder computational
|
||||
## complexity, buffer size, and memory bandwidth while still addressing
|
||||
## the widest possible range of applications.
|
||||
##
|
||||
## Main Level
|
||||
## MPEG-2 Video Main Profile and Main Level is analogous to MPEG-1's
|
||||
## CPB, with sampling limits at CCIR 601 parameters (720x480x30 Hz or
|
||||
## 720x576x24 Hz).
|
||||
##
|
||||
##------------------------------------------------------------------------
|
||||
FRAME_RATE = [
|
||||
0,
|
||||
24000.0/1001, ## 3-2 pulldown NTSC (CPB/Main Level)
|
||||
24, ## Film (CPB/Main Level)
|
||||
25, ## PAL/SECAM or 625/60 video
|
||||
30000.0/1001, ## NTSC (CPB/Main Level)
|
||||
30, ## drop-frame NTSC or component 525/60 (CPB/Main Level)
|
||||
50, ## double-rate PAL
|
||||
60000.0/1001, ## double-rate NTSC
|
||||
60, ## double-rate, drop-frame NTSC/component 525/60 video
|
||||
]
|
||||
|
||||
##------------------------------------------------------------------------
|
||||
## ASPECT_RATIO -- INCOMPLETE?
|
||||
##
|
||||
## This lookup table maps the header aspect ratio index to a float value.
|
||||
## These are just the defined ratios for CPB I believe. As I understand
|
||||
## it, a stream that doesn't adhere to one of these aspect ratios is
|
||||
## technically considered non-compliant.
|
||||
##------------------------------------------------------------------------
|
||||
ASPECT_RATIO = ( None, # Forbidden
|
||||
1.0, # 1/1 (VGA)
|
||||
4.0 / 3, # 4/3 (TV)
|
||||
16.0 / 9, # 16/9 (Widescreen)
|
||||
2.21 # (Cinema)
|
||||
)
|
||||
|
||||
|
||||
class MPEG(core.AVContainer):
|
||||
"""
|
||||
Parser for various MPEG files. This includes MPEG-1 and MPEG-2
|
||||
program streams, elementary streams and transport streams. The
|
||||
reported length differs from the length reported by most video
|
||||
players but the provides length here is correct. An MPEG file has
|
||||
no additional metadata like title, etc; only codecs, length and
|
||||
resolution is reported back.
|
||||
"""
|
||||
def __init__(self,file):
|
||||
core.AVContainer.__init__(self)
|
||||
self.sequence_header_offset = 0
|
||||
self.mpeg_version = 2
|
||||
|
||||
# detect TS (fast scan)
|
||||
if not self.isTS(file):
|
||||
# detect system mpeg (many infos)
|
||||
if not self.isMPEG(file):
|
||||
# detect PES
|
||||
if not self.isPES(file):
|
||||
# Maybe it's MPEG-ES
|
||||
if self.isES(file):
|
||||
# If isES() succeeds, we needn't do anything further.
|
||||
return
|
||||
if file.name.lower().endswith('mpeg') or \
|
||||
file.name.lower().endswith('mpg'):
|
||||
# This has to be an mpeg file. It could be a bad
|
||||
# recording from an ivtv based hardware encoder with
|
||||
# same bytes missing at the beginning.
|
||||
# Do some more digging...
|
||||
if not self.isMPEG(file, force=True) or \
|
||||
not self.video or not self.audio:
|
||||
# does not look like an mpeg at all
|
||||
raise ParseError()
|
||||
else:
|
||||
# no mpeg at all
|
||||
raise ParseError()
|
||||
|
||||
self.mime = 'video/mpeg'
|
||||
if not self.video:
|
||||
self.video.append(core.VideoStream())
|
||||
|
||||
if self.sequence_header_offset <= 0:
|
||||
return
|
||||
|
||||
self.progressive(file)
|
||||
|
||||
for vi in self.video:
|
||||
vi.width, vi.height = self.dxy(file)
|
||||
vi.fps, vi.aspect = self.framerate_aspect(file)
|
||||
vi.bitrate = self.bitrate(file)
|
||||
if self.length:
|
||||
vi.length = self.length
|
||||
|
||||
if not self.type:
|
||||
self.type = 'MPEG Video'
|
||||
|
||||
# set fourcc codec for video and audio
|
||||
vc, ac = 'MP2V', 'MP2A'
|
||||
if self.mpeg_version == 1:
|
||||
vc, ac = 'MPEG', 0x0050
|
||||
for v in self.video:
|
||||
v.codec = vc
|
||||
for a in self.audio:
|
||||
if not a.codec:
|
||||
a.codec = ac
|
||||
|
||||
|
||||
def dxy(self,file):
|
||||
"""
|
||||
get width and height of the video
|
||||
"""
|
||||
file.seek(self.sequence_header_offset+4,0)
|
||||
v = file.read(4)
|
||||
x = struct.unpack('>H',v[:2])[0] >> 4
|
||||
y = struct.unpack('>H',v[1:3])[0] & 0x0FFF
|
||||
return (x,y)
|
||||
|
||||
|
||||
def framerate_aspect(self,file):
|
||||
"""
|
||||
read framerate and aspect ratio
|
||||
"""
|
||||
file.seek(self.sequence_header_offset+7,0)
|
||||
v = struct.unpack( '>B', file.read(1) )[0]
|
||||
try:
|
||||
fps = FRAME_RATE[v&0xf]
|
||||
except IndexError:
|
||||
fps = None
|
||||
if v>>4 < len(ASPECT_RATIO):
|
||||
aspect = ASPECT_RATIO[v>>4]
|
||||
else:
|
||||
aspect = None
|
||||
return (fps, aspect)
|
||||
|
||||
|
||||
def progressive(self, file):
|
||||
"""
|
||||
Try to find out with brute force if the mpeg is interlaced or not.
|
||||
Search for the Sequence_Extension in the extension header (01B5)
|
||||
"""
|
||||
file.seek(0)
|
||||
buffer = ''
|
||||
count = 0
|
||||
while 1:
|
||||
if len(buffer) < 1000:
|
||||
count += 1
|
||||
if count > 1000:
|
||||
break
|
||||
buffer += file.read(1024)
|
||||
if len(buffer) < 1000:
|
||||
break
|
||||
pos = buffer.find('\x00\x00\x01\xb5')
|
||||
if pos == -1 or len(buffer) - pos < 5:
|
||||
buffer = buffer[-10:]
|
||||
continue
|
||||
ext = (ord(buffer[pos+4]) >> 4)
|
||||
if ext == 8:
|
||||
pass
|
||||
elif ext == 1:
|
||||
if (ord(buffer[pos+5]) >> 3) & 1:
|
||||
self._set('progressive', True)
|
||||
else:
|
||||
self._set('interlaced', True)
|
||||
return True
|
||||
else:
|
||||
log.debug('ext', ext)
|
||||
buffer = buffer[pos+4:]
|
||||
return False
|
||||
|
||||
|
||||
##------------------------------------------------------------------------
|
||||
## bitrate()
|
||||
##
|
||||
## From the MPEG-2.2 spec:
|
||||
##
|
||||
## bit_rate -- This is a 30-bit integer. The lower 18 bits of the
|
||||
## integer are in bit_rate_value and the upper 12 bits are in
|
||||
## bit_rate_extension. The 30-bit integer specifies the bitrate of the
|
||||
## bitstream measured in units of 400 bits/second, rounded upwards.
|
||||
## The value zero is forbidden.
|
||||
##
|
||||
## So ignoring all the variable bitrate stuff for now, this 30 bit integer
|
||||
## multiplied times 400 bits/sec should give the rate in bits/sec.
|
||||
##
|
||||
## TODO: Variable bitrates? I need one that implements this.
|
||||
##
|
||||
## Continued from the MPEG-2.2 spec:
|
||||
##
|
||||
## If the bitstream is a constant bitrate stream, the bitrate specified
|
||||
## is the actual rate of operation of the VBV specified in annex C. If
|
||||
## the bitstream is a variable bitrate stream, the STD specifications in
|
||||
## ISO/IEC 13818-1 supersede the VBV, and the bitrate specified here is
|
||||
## used to dimension the transport stream STD (2.4.2 in ITU-T Rec. xxx |
|
||||
## ISO/IEC 13818-1), or the program stream STD (2.4.5 in ITU-T Rec. xxx |
|
||||
## ISO/IEC 13818-1).
|
||||
##
|
||||
## If the bitstream is not a constant rate bitstream the vbv_delay
|
||||
## field shall have the value FFFF in hexadecimal.
|
||||
##
|
||||
## Given the value encoded in the bitrate field, the bitstream shall be
|
||||
## generated so that the video encoding and the worst case multiplex
|
||||
## jitter do not cause STD buffer overflow or underflow.
|
||||
##
|
||||
##
|
||||
##------------------------------------------------------------------------
|
||||
|
||||
|
||||
## Some parts in the code are based on mpgtx (mpgtx.sf.net)
|
||||
|
||||
def bitrate(self,file):
|
||||
"""
|
||||
read the bitrate (most of the time broken)
|
||||
"""
|
||||
file.seek(self.sequence_header_offset+8,0)
|
||||
t,b = struct.unpack( '>HB', file.read(3) )
|
||||
vrate = t << 2 | b >> 6
|
||||
return vrate * 400
|
||||
|
||||
|
||||
def ReadSCRMpeg2(self, buffer):
|
||||
"""
|
||||
read SCR (timestamp) for MPEG2 at the buffer beginning (6 Bytes)
|
||||
"""
|
||||
if len(buffer) < 6:
|
||||
return None
|
||||
|
||||
highbit = (ord(buffer[0])&0x20)>>5
|
||||
|
||||
low4Bytes= ((long(ord(buffer[0])) & 0x18) >> 3) << 30
|
||||
low4Bytes |= (ord(buffer[0]) & 0x03) << 28
|
||||
low4Bytes |= ord(buffer[1]) << 20
|
||||
low4Bytes |= (ord(buffer[2]) & 0xF8) << 12
|
||||
low4Bytes |= (ord(buffer[2]) & 0x03) << 13
|
||||
low4Bytes |= ord(buffer[3]) << 5
|
||||
low4Bytes |= (ord(buffer[4])) >> 3
|
||||
|
||||
sys_clock_ref=(ord(buffer[4]) & 0x3) << 7
|
||||
sys_clock_ref|=(ord(buffer[5]) >> 1)
|
||||
|
||||
return (long(highbit * (1<<16) * (1<<16)) + low4Bytes) / 90000
|
||||
|
||||
|
||||
def ReadSCRMpeg1(self, buffer):
|
||||
"""
|
||||
read SCR (timestamp) for MPEG1 at the buffer beginning (5 Bytes)
|
||||
"""
|
||||
if len(buffer) < 5:
|
||||
return None
|
||||
|
||||
highbit = (ord(buffer[0]) >> 3) & 0x01
|
||||
|
||||
low4Bytes = ((long(ord(buffer[0])) >> 1) & 0x03) << 30
|
||||
low4Bytes |= ord(buffer[1]) << 22;
|
||||
low4Bytes |= (ord(buffer[2]) >> 1) << 15;
|
||||
low4Bytes |= ord(buffer[3]) << 7;
|
||||
low4Bytes |= ord(buffer[4]) >> 1;
|
||||
|
||||
return (long(highbit) * (1<<16) * (1<<16) + low4Bytes) / 90000;
|
||||
|
||||
|
||||
def ReadPTS(self, buffer):
|
||||
"""
|
||||
read PTS (PES timestamp) at the buffer beginning (5 Bytes)
|
||||
"""
|
||||
high = ((ord(buffer[0]) & 0xF) >> 1)
|
||||
med = (ord(buffer[1]) << 7) + (ord(buffer[2]) >> 1)
|
||||
low = (ord(buffer[3]) << 7) + (ord(buffer[4]) >> 1)
|
||||
return ((long(high) << 30 ) + (med << 15) + low) / 90000
|
||||
|
||||
|
||||
def ReadHeader(self, buffer, offset):
|
||||
"""
|
||||
Handle MPEG header in buffer on position offset
|
||||
Return None on error, new offset or 0 if the new offset can't be scanned
|
||||
"""
|
||||
if buffer[offset:offset+3] != '\x00\x00\x01':
|
||||
return None
|
||||
|
||||
id = ord(buffer[offset+3])
|
||||
|
||||
if id == PADDING_PKT:
|
||||
return offset + (ord(buffer[offset+4]) << 8) + \
|
||||
ord(buffer[offset+5]) + 6
|
||||
|
||||
if id == PACK_PKT:
|
||||
if ord(buffer[offset+4]) & 0xF0 == 0x20:
|
||||
self.type = 'MPEG-1 Video'
|
||||
self.get_time = self.ReadSCRMpeg1
|
||||
self.mpeg_version = 1
|
||||
return offset + 12
|
||||
elif (ord(buffer[offset+4]) & 0xC0) == 0x40:
|
||||
self.type = 'MPEG-2 Video'
|
||||
self.get_time = self.ReadSCRMpeg2
|
||||
return offset + (ord(buffer[offset+13]) & 0x07) + 14
|
||||
else:
|
||||
# I have no idea what just happened, but for some DVB
|
||||
# recordings done with mencoder this points to a
|
||||
# PACK_PKT describing something odd. Returning 0 here
|
||||
# (let's hope there are no extensions in the header)
|
||||
# fixes it.
|
||||
return 0
|
||||
|
||||
if 0xC0 <= id <= 0xDF:
|
||||
# code for audio stream
|
||||
for a in self.audio:
|
||||
if a.id == id:
|
||||
break
|
||||
else:
|
||||
self.audio.append(core.AudioStream())
|
||||
self.audio[-1]._set('id', id)
|
||||
return 0
|
||||
|
||||
if 0xE0 <= id <= 0xEF:
|
||||
# code for video stream
|
||||
for v in self.video:
|
||||
if v.id == id:
|
||||
break
|
||||
else:
|
||||
self.video.append(core.VideoStream())
|
||||
self.video[-1]._set('id', id)
|
||||
return 0
|
||||
|
||||
if id == SEQ_HEAD:
|
||||
# sequence header, remember that position for later use
|
||||
self.sequence_header_offset = offset
|
||||
return 0
|
||||
|
||||
if id in [PRIVATE_STREAM1, PRIVATE_STREAM2]:
|
||||
# private stream. we don't know, but maybe we can guess later
|
||||
add = ord(buffer[offset+8])
|
||||
# if (ord(buffer[offset+6]) & 4) or 1:
|
||||
# id = ord(buffer[offset+10+add])
|
||||
if buffer[offset+11+add:offset+15+add].find('\x0b\x77') != -1:
|
||||
# AC3 stream
|
||||
for a in self.audio:
|
||||
if a.id == id:
|
||||
break
|
||||
else:
|
||||
self.audio.append(core.AudioStream())
|
||||
self.audio[-1]._set('id', id)
|
||||
self.audio[-1].codec = 0x2000 # AC3
|
||||
return 0
|
||||
|
||||
if id == SYS_PKT:
|
||||
return 0
|
||||
|
||||
if id == EXT_START:
|
||||
return 0
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
# Normal MPEG (VCD, SVCD) ========================================
|
||||
|
||||
def isMPEG(self, file, force=False):
|
||||
"""
|
||||
This MPEG starts with a sequence of 0x00 followed by a PACK Header
|
||||
http://dvd.sourceforge.net/dvdinfo/packhdr.html
|
||||
"""
|
||||
file.seek(0,0)
|
||||
buffer = file.read(10000)
|
||||
offset = 0
|
||||
|
||||
# seek until the 0 byte stop
|
||||
while offset < len(buffer)-100 and buffer[offset] == '\0':
|
||||
offset += 1
|
||||
offset -= 2
|
||||
|
||||
# test for mpeg header 0x00 0x00 0x01
|
||||
header = '\x00\x00\x01%s' % chr(PACK_PKT)
|
||||
if offset < 0 or not buffer[offset:offset+4] == header:
|
||||
if not force:
|
||||
return 0
|
||||
# brute force and try to find the pack header in the first
|
||||
# 10000 bytes somehow
|
||||
offset = buffer.find(header)
|
||||
if offset < 0:
|
||||
return 0
|
||||
|
||||
# scan the 100000 bytes of data
|
||||
buffer += file.read(100000)
|
||||
|
||||
# scan first header, to get basic info about
|
||||
# how to read a timestamp
|
||||
self.ReadHeader(buffer, offset)
|
||||
|
||||
# store first timestamp
|
||||
self.start = self.get_time(buffer[offset+4:])
|
||||
while len(buffer) > offset + 1000 and \
|
||||
buffer[offset:offset+3] == '\x00\x00\x01':
|
||||
# read the mpeg header
|
||||
new_offset = self.ReadHeader(buffer, offset)
|
||||
|
||||
# header scanning detected error, this is no mpeg
|
||||
if new_offset == None:
|
||||
return 0
|
||||
|
||||
if new_offset:
|
||||
# we have a new offset
|
||||
offset = new_offset
|
||||
|
||||
# skip padding 0 before a new header
|
||||
while len(buffer) > offset + 10 and \
|
||||
not ord(buffer[offset+2]):
|
||||
offset += 1
|
||||
|
||||
else:
|
||||
# seek to new header by brute force
|
||||
offset += buffer[offset+4:].find('\x00\x00\x01') + 4
|
||||
|
||||
# fill in values for support functions:
|
||||
self.__seek_size__ = 1000000
|
||||
self.__sample_size__ = 10000
|
||||
self.__search__ = self._find_timer_
|
||||
self.filename = file.name
|
||||
|
||||
# get length of the file
|
||||
self.length = self.get_length()
|
||||
return 1
|
||||
|
||||
|
||||
def _find_timer_(self, buffer):
|
||||
"""
|
||||
Return position of timer in buffer or None if not found.
|
||||
This function is valid for 'normal' mpeg files
|
||||
"""
|
||||
pos = buffer.find('\x00\x00\x01%s' % chr(PACK_PKT))
|
||||
if pos == -1:
|
||||
return None
|
||||
return pos + 4
|
||||
|
||||
|
||||
|
||||
# PES ============================================================
|
||||
|
||||
|
||||
def ReadPESHeader(self, offset, buffer, id=0):
|
||||
"""
|
||||
Parse a PES header.
|
||||
Since it starts with 0x00 0x00 0x01 like 'normal' mpegs, this
|
||||
function will return (0, None) when it is no PES header or
|
||||
(packet length, timestamp position (maybe None))
|
||||
|
||||
http://dvd.sourceforge.net/dvdinfo/pes-hdr.html
|
||||
"""
|
||||
if not buffer[0:3] == '\x00\x00\x01':
|
||||
return 0, None
|
||||
|
||||
packet_length = (ord(buffer[4]) << 8) + ord(buffer[5]) + 6
|
||||
align = ord(buffer[6]) & 4
|
||||
header_length = ord(buffer[8])
|
||||
|
||||
# PES ID (starting with 001)
|
||||
if ord(buffer[3]) & 0xE0 == 0xC0:
|
||||
id = id or ord(buffer[3]) & 0x1F
|
||||
for a in self.audio:
|
||||
if a.id == id:
|
||||
break
|
||||
else:
|
||||
self.audio.append(core.AudioStream())
|
||||
self.audio[-1]._set('id', id)
|
||||
|
||||
elif ord(buffer[3]) & 0xF0 == 0xE0:
|
||||
id = id or ord(buffer[3]) & 0xF
|
||||
for v in self.video:
|
||||
if v.id == id:
|
||||
break
|
||||
else:
|
||||
self.video.append(core.VideoStream())
|
||||
self.video[-1]._set('id', id)
|
||||
|
||||
# new mpeg starting
|
||||
if buffer[header_length+9:header_length+13] == \
|
||||
'\x00\x00\x01\xB3' and not self.sequence_header_offset:
|
||||
# yes, remember offset for later use
|
||||
self.sequence_header_offset = offset + header_length+9
|
||||
elif ord(buffer[3]) == 189 or ord(buffer[3]) == 191:
|
||||
# private stream. we don't know, but maybe we can guess later
|
||||
id = id or ord(buffer[3]) & 0xF
|
||||
if align and \
|
||||
buffer[header_length+9:header_length+11] == '\x0b\x77':
|
||||
# AC3 stream
|
||||
for a in self.audio:
|
||||
if a.id == id:
|
||||
break
|
||||
else:
|
||||
self.audio.append(core.AudioStream())
|
||||
self.audio[-1]._set('id', id)
|
||||
self.audio[-1].codec = 0x2000 # AC3
|
||||
|
||||
else:
|
||||
# unknown content
|
||||
pass
|
||||
|
||||
ptsdts = ord(buffer[7]) >> 6
|
||||
|
||||
if ptsdts and ptsdts == ord(buffer[9]) >> 4:
|
||||
if ord(buffer[9]) >> 4 != ptsdts:
|
||||
log.warning('WARNING: bad PTS/DTS, please contact us')
|
||||
return packet_length, None
|
||||
|
||||
# timestamp = self.ReadPTS(buffer[9:14])
|
||||
high = ((ord(buffer[9]) & 0xF) >> 1)
|
||||
med = (ord(buffer[10]) << 7) + (ord(buffer[11]) >> 1)
|
||||
low = (ord(buffer[12]) << 7) + (ord(buffer[13]) >> 1)
|
||||
return packet_length, 9
|
||||
|
||||
return packet_length, None
|
||||
|
||||
|
||||
|
||||
def isPES(self, file):
|
||||
log.info('trying mpeg-pes scan')
|
||||
file.seek(0,0)
|
||||
buffer = file.read(3)
|
||||
|
||||
# header (also valid for all mpegs)
|
||||
if not buffer == '\x00\x00\x01':
|
||||
return 0
|
||||
|
||||
self.sequence_header_offset = 0
|
||||
buffer += file.read(10000)
|
||||
|
||||
offset = 0
|
||||
while offset + 1000 < len(buffer):
|
||||
pos, timestamp = self.ReadPESHeader(offset, buffer[offset:])
|
||||
if not pos:
|
||||
return 0
|
||||
if timestamp != None and not hasattr(self, 'start'):
|
||||
self.get_time = self.ReadPTS
|
||||
bpos = buffer[offset+timestamp:offset+timestamp+5]
|
||||
self.start = self.get_time(bpos)
|
||||
if self.sequence_header_offset and hasattr(self, 'start'):
|
||||
# we have all informations we need
|
||||
break
|
||||
|
||||
offset += pos
|
||||
if offset + 1000 < len(buffer) and len(buffer) < 1000000 or 1:
|
||||
# looks like a pes, read more
|
||||
buffer += file.read(10000)
|
||||
|
||||
if not self.video and not self.audio:
|
||||
# no video and no audio?
|
||||
return 0
|
||||
|
||||
self.type = 'MPEG-PES'
|
||||
|
||||
# fill in values for support functions:
|
||||
self.__seek_size__ = 10000000 # 10 MB
|
||||
self.__sample_size__ = 500000 # 500 k scanning
|
||||
self.__search__ = self._find_timer_PES_
|
||||
self.filename = file.name
|
||||
|
||||
# get length of the file
|
||||
self.length = self.get_length()
|
||||
return 1
|
||||
|
||||
|
||||
def _find_timer_PES_(self, buffer):
|
||||
"""
|
||||
Return position of timer in buffer or -1 if not found.
|
||||
This function is valid for PES files
|
||||
"""
|
||||
pos = buffer.find('\x00\x00\x01')
|
||||
offset = 0
|
||||
if pos == -1 or offset + 1000 >= len(buffer):
|
||||
return None
|
||||
|
||||
retpos = -1
|
||||
ackcount = 0
|
||||
while offset + 1000 < len(buffer):
|
||||
pos, timestamp = self.ReadPESHeader(offset, buffer[offset:])
|
||||
if timestamp != None and retpos == -1:
|
||||
retpos = offset + timestamp
|
||||
if pos == 0:
|
||||
# Oops, that was a mpeg header, no PES header
|
||||
offset += buffer[offset:].find('\x00\x00\x01')
|
||||
retpos = -1
|
||||
ackcount = 0
|
||||
else:
|
||||
offset += pos
|
||||
if retpos != -1:
|
||||
ackcount += 1
|
||||
if ackcount > 10:
|
||||
# looks ok to me
|
||||
return retpos
|
||||
return None
|
||||
|
||||
|
||||
# Elementary Stream ===============================================
|
||||
|
||||
def isES(self, file):
|
||||
file.seek(0, 0)
|
||||
try:
|
||||
header = struct.unpack('>LL', file.read(8))
|
||||
except (struct.error, IOError):
|
||||
return False
|
||||
|
||||
if header[0] != 0x1B3:
|
||||
return False
|
||||
|
||||
# Is an mpeg video elementary stream
|
||||
|
||||
self.mime = 'video/mpeg'
|
||||
video = core.VideoStream()
|
||||
video.width = header[1] >> 20
|
||||
video.height = (header[1] >> 8) & 0xfff
|
||||
if header[1] & 0xf < len(FRAME_RATE):
|
||||
video.fps = FRAME_RATE[header[1] & 0xf]
|
||||
if (header[1] >> 4) & 0xf < len(ASPECT_RATIO):
|
||||
# FIXME: Empirically the aspect looks like PAR rather than DAR
|
||||
video.aspect = ASPECT_RATIO[(header[1] >> 4) & 0xf]
|
||||
self.video.append(video)
|
||||
return True
|
||||
|
||||
|
||||
# Transport Stream ===============================================
|
||||
|
||||
def isTS(self, file):
|
||||
file.seek(0,0)
|
||||
|
||||
buffer = file.read(TS_PACKET_LENGTH * 2)
|
||||
c = 0
|
||||
|
||||
while c + TS_PACKET_LENGTH < len(buffer):
|
||||
if ord(buffer[c]) == ord(buffer[c+TS_PACKET_LENGTH]) == TS_SYNC:
|
||||
break
|
||||
c += 1
|
||||
else:
|
||||
return 0
|
||||
|
||||
buffer += file.read(10000)
|
||||
self.type = 'MPEG-TS'
|
||||
|
||||
while c + TS_PACKET_LENGTH < len(buffer):
|
||||
start = ord(buffer[c+1]) & 0x40
|
||||
# maybe load more into the buffer
|
||||
if c + 2 * TS_PACKET_LENGTH > len(buffer) and c < 500000:
|
||||
buffer += file.read(10000)
|
||||
|
||||
# wait until the ts payload contains a payload header
|
||||
if not start:
|
||||
c += TS_PACKET_LENGTH
|
||||
continue
|
||||
|
||||
tsid = ((ord(buffer[c+1]) & 0x3F) << 8) + ord(buffer[c+2])
|
||||
adapt = (ord(buffer[c+3]) & 0x30) >> 4
|
||||
|
||||
offset = 4
|
||||
if adapt & 0x02:
|
||||
# meta info present, skip it for now
|
||||
adapt_len = ord(buffer[c+offset])
|
||||
offset += adapt_len + 1
|
||||
|
||||
if not ord(buffer[c+1]) & 0x40:
|
||||
# no new pes or psi in stream payload starting
|
||||
pass
|
||||
elif adapt & 0x01:
|
||||
# PES
|
||||
timestamp = self.ReadPESHeader(c+offset, buffer[c+offset:],
|
||||
tsid)[1]
|
||||
if timestamp != None:
|
||||
if not hasattr(self, 'start'):
|
||||
self.get_time = self.ReadPTS
|
||||
timestamp = c + offset + timestamp
|
||||
self.start = self.get_time(buffer[timestamp:timestamp+5])
|
||||
elif not hasattr(self, 'audio_ok'):
|
||||
timestamp = c + offset + timestamp
|
||||
start = self.get_time(buffer[timestamp:timestamp+5])
|
||||
if start is not None and self.start is not None and \
|
||||
abs(start - self.start) < 10:
|
||||
# looks ok
|
||||
self.audio_ok = True
|
||||
else:
|
||||
# timestamp broken
|
||||
del self.start
|
||||
log.warning('Timestamp error, correcting')
|
||||
|
||||
if hasattr(self, 'start') and self.start and \
|
||||
self.sequence_header_offset and self.video and self.audio:
|
||||
break
|
||||
|
||||
c += TS_PACKET_LENGTH
|
||||
|
||||
|
||||
if not self.sequence_header_offset:
|
||||
return 0
|
||||
|
||||
# fill in values for support functions:
|
||||
self.__seek_size__ = 10000000 # 10 MB
|
||||
self.__sample_size__ = 100000 # 100 k scanning
|
||||
self.__search__ = self._find_timer_TS_
|
||||
self.filename = file.name
|
||||
|
||||
# get length of the file
|
||||
self.length = self.get_length()
|
||||
return 1
|
||||
|
||||
|
||||
def _find_timer_TS_(self, buffer):
|
||||
c = 0
|
||||
|
||||
while c + TS_PACKET_LENGTH < len(buffer):
|
||||
if ord(buffer[c]) == ord(buffer[c+TS_PACKET_LENGTH]) == TS_SYNC:
|
||||
break
|
||||
c += 1
|
||||
else:
|
||||
return None
|
||||
|
||||
while c + TS_PACKET_LENGTH < len(buffer):
|
||||
start = ord(buffer[c+1]) & 0x40
|
||||
if not start:
|
||||
c += TS_PACKET_LENGTH
|
||||
continue
|
||||
|
||||
tsid = ((ord(buffer[c+1]) & 0x3F) << 8) + ord(buffer[c+2])
|
||||
adapt = (ord(buffer[c+3]) & 0x30) >> 4
|
||||
|
||||
offset = 4
|
||||
if adapt & 0x02:
|
||||
# meta info present, skip it for now
|
||||
offset += ord(buffer[c+offset]) + 1
|
||||
|
||||
if adapt & 0x01:
|
||||
timestamp = self.ReadPESHeader(c+offset, buffer[c+offset:], tsid)[1]
|
||||
if timestamp is None:
|
||||
# this should not happen
|
||||
log.error('bad TS')
|
||||
return None
|
||||
return c + offset + timestamp
|
||||
c += TS_PACKET_LENGTH
|
||||
return None
|
||||
|
||||
|
||||
|
||||
# Support functions ==============================================
|
||||
|
||||
def get_endpos(self):
|
||||
"""
|
||||
get the last timestamp of the mpeg, return -1 if this is not possible
|
||||
"""
|
||||
if not hasattr(self, 'filename') or not hasattr(self, 'start'):
|
||||
return None
|
||||
|
||||
length = os.stat(self.filename)[stat.ST_SIZE]
|
||||
if length < self.__sample_size__:
|
||||
return
|
||||
|
||||
file = open(self.filename)
|
||||
file.seek(length - self.__sample_size__)
|
||||
buffer = file.read(self.__sample_size__)
|
||||
|
||||
end = None
|
||||
while 1:
|
||||
pos = self.__search__(buffer)
|
||||
if pos == None:
|
||||
break
|
||||
end = self.get_time(buffer[pos:]) or end
|
||||
buffer = buffer[pos+100:]
|
||||
|
||||
file.close()
|
||||
return end
|
||||
|
||||
|
||||
def get_length(self):
|
||||
"""
|
||||
get the length in seconds, return -1 if this is not possible
|
||||
"""
|
||||
end = self.get_endpos()
|
||||
if end == None or self.start == None:
|
||||
return None
|
||||
if self.start > end:
|
||||
return int(((long(1) << 33) - 1 ) / 90000) - self.start + end
|
||||
return end - self.start
|
||||
|
||||
|
||||
def seek(self, end_time):
|
||||
"""
|
||||
Return the byte position in the file where the time position
|
||||
is 'pos' seconds. Return 0 if this is not possible
|
||||
"""
|
||||
if not hasattr(self, 'filename') or not hasattr(self, 'start'):
|
||||
return 0
|
||||
|
||||
file = open(self.filename)
|
||||
seek_to = 0
|
||||
|
||||
while 1:
|
||||
file.seek(self.__seek_size__, 1)
|
||||
buffer = file.read(self.__sample_size__)
|
||||
if len(buffer) < 10000:
|
||||
break
|
||||
pos = self.__search__(buffer)
|
||||
if pos != None:
|
||||
# found something
|
||||
nt = self.get_time(buffer[pos:])
|
||||
if nt is not None and nt >= end_time:
|
||||
# too much, break
|
||||
break
|
||||
# that wasn't enough
|
||||
seek_to = file.tell()
|
||||
|
||||
file.close()
|
||||
return seek_to
|
||||
|
||||
|
||||
def __scan__(self):
|
||||
"""
|
||||
scan file for timestamps (may take a long time)
|
||||
"""
|
||||
if not hasattr(self, 'filename') or not hasattr(self, 'start'):
|
||||
return 0
|
||||
|
||||
file = open(self.filename)
|
||||
log.debug('scanning file...')
|
||||
while 1:
|
||||
file.seek(self.__seek_size__ * 10, 1)
|
||||
buffer = file.read(self.__sample_size__)
|
||||
if len(buffer) < 10000:
|
||||
break
|
||||
pos = self.__search__(buffer)
|
||||
if pos == None:
|
||||
continue
|
||||
log.debug('buffer position: %s' % self.get_time(buffer[pos:]))
|
||||
|
||||
file.close()
|
||||
log.debug('done scanning file')
|
||||
|
||||
|
||||
Parser = MPEG
|
||||
@@ -0,0 +1,301 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2003-2006 Thomas Schueppel <stain@acm.org>
|
||||
# Copyright (C) 2003-2006 Dirk Meyer <dischi@freevo.org>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__all__ = ['Parser']
|
||||
|
||||
import struct
|
||||
import re
|
||||
import stat
|
||||
import os
|
||||
import logging
|
||||
from exceptions import *
|
||||
import core
|
||||
|
||||
# get logging object
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
PACKET_TYPE_HEADER = 0x01
|
||||
PACKED_TYPE_METADATA = 0x03
|
||||
PACKED_TYPE_SETUP = 0x05
|
||||
PACKET_TYPE_BITS = 0x07
|
||||
PACKET_IS_SYNCPOINT = 0x08
|
||||
|
||||
#VORBIS_VIDEO_PACKET_INFO = 'video'
|
||||
|
||||
STREAM_HEADER_VIDEO = '<4sIQQIIHII'
|
||||
STREAM_HEADER_AUDIO = '<4sIQQIIHHHI'
|
||||
|
||||
VORBISCOMMENT = { 'TITLE': 'title',
|
||||
'ALBUM': 'album',
|
||||
'ARTIST': 'artist',
|
||||
'COMMENT': 'comment',
|
||||
'ENCODER': 'encoder',
|
||||
'TRACKNUMBER': 'trackno',
|
||||
'LANGUAGE': 'language',
|
||||
'GENRE': 'genre',
|
||||
}
|
||||
|
||||
# FIXME: check VORBISCOMMENT date and convert to timestamp
|
||||
# Deactived tag: 'DATE': 'date',
|
||||
|
||||
MAXITERATIONS = 30
|
||||
|
||||
class Ogm(core.AVContainer):
|
||||
|
||||
table_mapping = { 'VORBISCOMMENT' : VORBISCOMMENT }
|
||||
|
||||
def __init__(self, file):
|
||||
core.AVContainer.__init__(self)
|
||||
self.samplerate = 1
|
||||
self.all_streams = [] # used to add meta data to streams
|
||||
self.all_header = []
|
||||
|
||||
for i in range(MAXITERATIONS):
|
||||
granule, nextlen = self._parseOGGS(file)
|
||||
if granule == None:
|
||||
if i == 0:
|
||||
# oops, bad file
|
||||
raise ParseError()
|
||||
break
|
||||
elif granule > 0:
|
||||
# ok, file started
|
||||
break
|
||||
|
||||
# seek to the end of the stream, to avoid scanning the whole file
|
||||
if (os.stat(file.name)[stat.ST_SIZE] > 50000):
|
||||
file.seek(os.stat(file.name)[stat.ST_SIZE]-49000)
|
||||
|
||||
# read the rest of the file into a buffer
|
||||
h = file.read()
|
||||
|
||||
# find last OggS to get length info
|
||||
if len(h) > 200:
|
||||
idx = h.find('OggS')
|
||||
pos = -49000 + idx
|
||||
if idx:
|
||||
file.seek(os.stat(file.name)[stat.ST_SIZE] + pos)
|
||||
while 1:
|
||||
granule, nextlen = self._parseOGGS(file)
|
||||
if not nextlen:
|
||||
break
|
||||
|
||||
# Copy metadata to the streams
|
||||
if len(self.all_header) == len(self.all_streams):
|
||||
for i in range(len(self.all_header)):
|
||||
|
||||
# get meta info
|
||||
for key in self.all_streams[i].keys():
|
||||
if self.all_header[i].has_key(key):
|
||||
self.all_streams[i][key] = self.all_header[i][key]
|
||||
del self.all_header[i][key]
|
||||
if self.all_header[i].has_key(key.upper()):
|
||||
asi = self.all_header[i][key.upper()]
|
||||
self.all_streams[i][key] = asi
|
||||
del self.all_header[i][key.upper()]
|
||||
|
||||
# Chapter parser
|
||||
if self.all_header[i].has_key('CHAPTER01') and \
|
||||
not self.chapters:
|
||||
while 1:
|
||||
s = 'CHAPTER%02d' % (len(self.chapters) + 1)
|
||||
if self.all_header[i].has_key(s) and \
|
||||
self.all_header[i].has_key(s + 'NAME'):
|
||||
pos = self.all_header[i][s]
|
||||
try:
|
||||
pos = int(pos)
|
||||
except ValueError:
|
||||
new_pos = 0
|
||||
for v in pos.split(':'):
|
||||
new_pos = new_pos * 60 + float(v)
|
||||
pos = int(new_pos)
|
||||
|
||||
c = self.all_header[i][s + 'NAME']
|
||||
c = core.Chapter(c, pos)
|
||||
del self.all_header[i][s + 'NAME']
|
||||
del self.all_header[i][s]
|
||||
self.chapters.append(c)
|
||||
else:
|
||||
break
|
||||
|
||||
# If there are no video streams in this ogg container, it
|
||||
# must be an audio file. Raise an exception to cause the
|
||||
# factory to fall back to audio.ogg.
|
||||
if len(self.video) == 0:
|
||||
raise ParseError
|
||||
|
||||
# Copy Metadata from tables into the main set of attributes
|
||||
for header in self.all_header:
|
||||
self._appendtable('VORBISCOMMENT', header)
|
||||
|
||||
|
||||
def _parseOGGS(self,file):
|
||||
h = file.read(27)
|
||||
if len(h) == 0:
|
||||
# Regular File end
|
||||
return None, None
|
||||
elif len(h) < 27:
|
||||
log.debug("%d Bytes of Garbage found after End." % len(h))
|
||||
return None, None
|
||||
if h[:4] != "OggS":
|
||||
log.debug("Invalid Ogg")
|
||||
raise ParseError()
|
||||
|
||||
version = ord(h[4])
|
||||
if version != 0:
|
||||
log.debug("Unsupported OGG/OGM Version %d." % version)
|
||||
return None, None
|
||||
|
||||
head = struct.unpack('<BQIIIB', h[5:])
|
||||
headertype, granulepos, serial, pageseqno, checksum, \
|
||||
pageSegCount = head
|
||||
|
||||
self.mime = 'application/ogm'
|
||||
self.type = 'OGG Media'
|
||||
tab = file.read(pageSegCount)
|
||||
nextlen = 0
|
||||
for i in range(len(tab)):
|
||||
nextlen += ord(tab[i])
|
||||
else:
|
||||
h = file.read(1)
|
||||
packettype = ord(h[0]) & PACKET_TYPE_BITS
|
||||
if packettype == PACKET_TYPE_HEADER:
|
||||
h += file.read(nextlen-1)
|
||||
self._parseHeader(h, granulepos)
|
||||
elif packettype == PACKED_TYPE_METADATA:
|
||||
h += file.read(nextlen-1)
|
||||
self._parseMeta(h)
|
||||
else:
|
||||
file.seek(nextlen-1,1)
|
||||
if len(self.all_streams) > serial:
|
||||
stream = self.all_streams[serial]
|
||||
if hasattr(stream, 'samplerate') and \
|
||||
stream.samplerate:
|
||||
stream.length = granulepos / stream.samplerate
|
||||
elif hasattr(stream, 'bitrate') and \
|
||||
stream.bitrate:
|
||||
stream.length = granulepos / stream.bitrate
|
||||
|
||||
return granulepos, nextlen + 27 + pageSegCount
|
||||
|
||||
|
||||
def _parseMeta(self,h):
|
||||
flags = ord(h[0])
|
||||
headerlen = len(h)
|
||||
if headerlen >= 7 and h[1:7] == 'vorbis':
|
||||
header = {}
|
||||
nextlen, self.encoder = self._extractHeaderString(h[7:])
|
||||
numItems = struct.unpack('<I',h[7+nextlen:7+nextlen+4])[0]
|
||||
start = 7+4+nextlen
|
||||
for i in range(numItems):
|
||||
(nextlen, s) = self._extractHeaderString(h[start:])
|
||||
start += nextlen
|
||||
if s:
|
||||
a = re.split('=',s)
|
||||
header[(a[0]).upper()]=a[1]
|
||||
# Put Header fields into info fields
|
||||
self.type = 'OGG Vorbis'
|
||||
self.subtype = ''
|
||||
self.all_header.append(header)
|
||||
|
||||
|
||||
def _parseHeader(self,header,granule):
|
||||
headerlen = len(header)
|
||||
flags = ord(header[0])
|
||||
|
||||
if headerlen >= 30 and header[1:7] == 'vorbis':
|
||||
ai = core.AudioStream()
|
||||
ai.version, ai.channels, ai.samplerate, bitrate_max, ai.bitrate, \
|
||||
bitrate_min, blocksize, framing = \
|
||||
struct.unpack('<IBIiiiBB',header[7:7+23])
|
||||
ai.codec = 'Vorbis'
|
||||
#ai.granule = granule
|
||||
#ai.length = granule / ai.samplerate
|
||||
self.audio.append(ai)
|
||||
self.all_streams.append(ai)
|
||||
|
||||
elif headerlen >= 7 and header[1:7] == 'theora':
|
||||
# Theora Header
|
||||
# XXX Finish Me
|
||||
vi = core.VideoStream()
|
||||
vi.codec = 'theora'
|
||||
self.video.append(vi)
|
||||
self.all_streams.append(vi)
|
||||
|
||||
elif headerlen >= 142 and \
|
||||
header[1:36] == 'Direct Show Samples embedded in Ogg':
|
||||
# Old Directshow format
|
||||
# XXX Finish Me
|
||||
vi = core.VideoStream()
|
||||
vi.codec = 'dshow'
|
||||
self.video.append(vi)
|
||||
self.all_streams.append(vi)
|
||||
|
||||
elif flags & PACKET_TYPE_BITS == PACKET_TYPE_HEADER and \
|
||||
headerlen >= struct.calcsize(STREAM_HEADER_VIDEO)+1:
|
||||
# New Directshow Format
|
||||
htype = header[1:9]
|
||||
|
||||
if htype[:5] == 'video':
|
||||
sh = header[9:struct.calcsize(STREAM_HEADER_VIDEO)+9]
|
||||
streamheader = struct.unpack(STREAM_HEADER_VIDEO, sh)
|
||||
vi = core.VideoStream()
|
||||
(type, ssize, timeunit, samplerate, vi.length, buffersize, \
|
||||
vi.bitrate, vi.width, vi.height) = streamheader
|
||||
|
||||
vi.width /= 65536
|
||||
vi.height /= 65536
|
||||
# XXX length, bitrate are very wrong
|
||||
vi.codec = type
|
||||
vi.fps = 10000000 / timeunit
|
||||
self.video.append(vi)
|
||||
self.all_streams.append(vi)
|
||||
|
||||
elif htype[:5] == 'audio':
|
||||
sha = header[9:struct.calcsize(STREAM_HEADER_AUDIO)+9]
|
||||
streamheader = struct.unpack(STREAM_HEADER_AUDIO, sha)
|
||||
ai = core.AudioStream()
|
||||
(type, ssize, timeunit, ai.samplerate, ai.length, buffersize, \
|
||||
ai.bitrate, ai.channels, bloc, ai.bitrate) = streamheader
|
||||
self.samplerate = ai.samplerate
|
||||
log.debug("Samplerate %d" % self.samplerate)
|
||||
self.audio.append(ai)
|
||||
self.all_streams.append(ai)
|
||||
|
||||
elif htype[:4] == 'text':
|
||||
subtitle = core.Subtitle()
|
||||
# FIXME: add more info
|
||||
self.subtitles.append(subtitle)
|
||||
self.all_streams.append(subtitle)
|
||||
|
||||
else:
|
||||
log.debug("Unknown Header")
|
||||
|
||||
|
||||
def _extractHeaderString(self,header):
|
||||
len = struct.unpack('<I', header[:4])[0]
|
||||
try:
|
||||
return (len+4,unicode(header[4:4+len], 'utf-8'))
|
||||
except (KeyError, IndexError, UnicodeDecodeError):
|
||||
return (len+4,None)
|
||||
|
||||
|
||||
Parser = Ogm
|
||||
@@ -0,0 +1,120 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2003-2006 Thomas Schueppel <stain@acm.org>
|
||||
# Copyright (C) 2003-2006 Dirk Meyer <dischi@freevo.org>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__all__ = ['Parser']
|
||||
|
||||
import struct
|
||||
import logging
|
||||
from exceptions import *
|
||||
import core
|
||||
|
||||
# http://www.pcisys.net/~melanson/codecs/rmff.htm
|
||||
# http://www.pcisys.net/~melanson/codecs/
|
||||
|
||||
# get logging object
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
class RealVideo(core.AVContainer):
|
||||
def __init__(self,file):
|
||||
core.AVContainer.__init__(self)
|
||||
self.mime = 'video/real'
|
||||
self.type = 'Real Video'
|
||||
h = file.read(10)
|
||||
try:
|
||||
(object_id,object_size,object_version) = struct.unpack('>4sIH',h)
|
||||
except struct.error:
|
||||
# EOF.
|
||||
raise ParseError()
|
||||
|
||||
if not object_id == '.RMF':
|
||||
raise ParseError()
|
||||
|
||||
file_version, num_headers = struct.unpack('>II', file.read(8))
|
||||
log.debug("size: %d, ver: %d, headers: %d" % \
|
||||
(object_size, file_version,num_headers))
|
||||
for i in range(0,num_headers):
|
||||
try:
|
||||
oi = struct.unpack('>4sIH',file.read(10))
|
||||
except (struct.error, IOError):
|
||||
# Header data we expected wasn't there. File may be
|
||||
# only partially complete.
|
||||
break
|
||||
|
||||
if object_id == 'DATA' and oi[0] != 'INDX':
|
||||
log.debug('INDX chunk expected after DATA but not found -- file corrupt')
|
||||
break
|
||||
|
||||
(object_id,object_size,object_version) = oi
|
||||
if object_id == 'DATA':
|
||||
# Seek over the data chunk rather than reading it in.
|
||||
file.seek(object_size - 10, 1)
|
||||
else:
|
||||
self._read_header(object_id, file.read(object_size-10))
|
||||
log.debug("%s [%d]" % (object_id,object_size-10))
|
||||
# Read all the following headers
|
||||
|
||||
|
||||
def _read_header(self,object_id,s):
|
||||
if object_id == 'PROP':
|
||||
prop = struct.unpack('>9IHH', s)
|
||||
log.debug(prop)
|
||||
if object_id == 'MDPR':
|
||||
mdpr = struct.unpack('>H7I', s[:30])
|
||||
log.debug(mdpr)
|
||||
self.length = mdpr[7]/1000.0
|
||||
(stream_name_size,) = struct.unpack('>B', s[30:31])
|
||||
stream_name = s[31:31+stream_name_size]
|
||||
pos = 31+stream_name_size
|
||||
(mime_type_size,) = struct.unpack('>B', s[pos:pos+1])
|
||||
mime = s[pos+1:pos+1+mime_type_size]
|
||||
pos += mime_type_size+1
|
||||
(type_specific_len,) = struct.unpack('>I', s[pos:pos+4])
|
||||
type_specific = s[pos+4:pos+4+type_specific_len]
|
||||
pos += 4+type_specific_len
|
||||
if mime[:5] == 'audio':
|
||||
ai = core.AudioStream()
|
||||
ai.id = mdpr[0]
|
||||
ai.bitrate = mdpr[2]
|
||||
self.audio.append(ai)
|
||||
elif mime[:5] == 'video':
|
||||
vi = core.VideoStream()
|
||||
vi.id = mdpr[0]
|
||||
vi.bitrate = mdpr[2]
|
||||
self.video.append(vi)
|
||||
else:
|
||||
log.debug("Unknown: %s" % mime)
|
||||
if object_id == 'CONT':
|
||||
pos = 0
|
||||
(title_len,) = struct.unpack('>H', s[pos:pos+2])
|
||||
self.title = s[2:title_len+2]
|
||||
pos += title_len+2
|
||||
(author_len,) = struct.unpack('>H', s[pos:pos+2])
|
||||
self.artist = s[pos+2:pos+author_len+2]
|
||||
pos += author_len+2
|
||||
(copyright_len,) = struct.unpack('>H', s[pos:pos+2])
|
||||
self.copyright = s[pos+2:pos+copyright_len+2]
|
||||
pos += copyright_len+2
|
||||
(comment_len,) = struct.unpack('>H', s[pos:pos+2])
|
||||
self.comment = s[pos+2:pos+comment_len+2]
|
||||
|
||||
|
||||
Parser = RealVideo
|
||||
@@ -0,0 +1,568 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2003-2006 Thomas Schueppel <stain@acm.org>
|
||||
# Copyright (C) 2003-2006 Dirk Meyer <dischi@freevo.org>
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__all__ = ['Parser']
|
||||
|
||||
import os
|
||||
import struct
|
||||
import string
|
||||
import logging
|
||||
import time
|
||||
from exceptions import *
|
||||
import core
|
||||
|
||||
# get logging object
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# List of tags
|
||||
# http://kibus1.narod.ru/frames_eng.htm?sof/abcavi/infotags.htm
|
||||
# http://www.divx-digest.com/software/avitags_dll.html
|
||||
# File Format: google for odmlff2.pdf
|
||||
|
||||
AVIINFO = {
|
||||
'INAM': 'title',
|
||||
'IART': 'artist',
|
||||
'IPRD': 'product',
|
||||
'ISFT': 'software',
|
||||
'ICMT': 'comment',
|
||||
'ILNG': 'language',
|
||||
'IKEY': 'keywords',
|
||||
'IPRT': 'trackno',
|
||||
'IFRM': 'trackof',
|
||||
'IPRO': 'producer',
|
||||
'IWRI': 'writer',
|
||||
'IGNR': 'genre',
|
||||
'ICOP': 'copyright'
|
||||
}
|
||||
|
||||
# Taken from libavcodec/mpeg4data.h (pixel_aspect struct)
|
||||
PIXEL_ASPECT = {
|
||||
1: (1, 1),
|
||||
2: (12, 11),
|
||||
3: (10, 11),
|
||||
4: (16, 11),
|
||||
5: (40, 33)
|
||||
}
|
||||
|
||||
|
||||
class Riff(core.AVContainer):
|
||||
"""
|
||||
AVI parser also parsing metadata like title, languages, etc.
|
||||
"""
|
||||
table_mapping = { 'AVIINFO' : AVIINFO }
|
||||
|
||||
def __init__(self,file):
|
||||
core.AVContainer.__init__(self)
|
||||
# read the header
|
||||
h = file.read(12)
|
||||
if h[:4] != "RIFF" and h[:4] != 'SDSS':
|
||||
raise ParseError()
|
||||
|
||||
self.has_idx = False
|
||||
self.header = {}
|
||||
self.junkStart = None
|
||||
self.infoStart = None
|
||||
self.type = h[8:12]
|
||||
if self.type == 'AVI ':
|
||||
self.mime = 'video/avi'
|
||||
elif self.type == 'WAVE':
|
||||
self.mime = 'audio/wav'
|
||||
try:
|
||||
while self._parseRIFFChunk(file):
|
||||
pass
|
||||
except IOError:
|
||||
log.exception('error in file, stop parsing')
|
||||
|
||||
self._find_subtitles(file.name)
|
||||
|
||||
if not self.has_idx and isinstance(self, core.AVContainer):
|
||||
log.debug('WARNING: avi has no index')
|
||||
self._set('corrupt', True)
|
||||
|
||||
|
||||
def _find_subtitles(self, filename):
|
||||
"""
|
||||
Search for subtitle files. Right now only VobSub is supported
|
||||
"""
|
||||
base = os.path.splitext(filename)[0]
|
||||
if os.path.isfile(base+'.idx') and \
|
||||
(os.path.isfile(base+'.sub') or os.path.isfile(base+'.rar')):
|
||||
file = open(base+'.idx')
|
||||
if file.readline().find('VobSub index file') > 0:
|
||||
for line in file.readlines():
|
||||
if line.find('id') == 0:
|
||||
sub = core.Subtitle()
|
||||
sub.language = line[4:6]
|
||||
sub.trackno = base + '.idx' # Maybe not?
|
||||
self.subtitles.append(sub)
|
||||
file.close()
|
||||
|
||||
|
||||
def _parseAVIH(self,t):
|
||||
retval = {}
|
||||
v = struct.unpack('<IIIIIIIIIIIIII',t[0:56])
|
||||
( retval['dwMicroSecPerFrame'],
|
||||
retval['dwMaxBytesPerSec'],
|
||||
retval['dwPaddingGranularity'],
|
||||
retval['dwFlags'],
|
||||
retval['dwTotalFrames'],
|
||||
retval['dwInitialFrames'],
|
||||
retval['dwStreams'],
|
||||
retval['dwSuggestedBufferSize'],
|
||||
retval['dwWidth'],
|
||||
retval['dwHeight'],
|
||||
retval['dwScale'],
|
||||
retval['dwRate'],
|
||||
retval['dwStart'],
|
||||
retval['dwLength'] ) = v
|
||||
if retval['dwMicroSecPerFrame'] == 0:
|
||||
log.warning("ERROR: Corrupt AVI")
|
||||
raise ParseError()
|
||||
|
||||
return retval
|
||||
|
||||
|
||||
def _parseSTRH(self,t):
|
||||
retval = {}
|
||||
retval['fccType'] = t[0:4]
|
||||
log.debug("_parseSTRH(%s) : %d bytes" % (retval['fccType'], len(t)))
|
||||
if retval['fccType'] != 'auds':
|
||||
retval['fccHandler'] = t[4:8]
|
||||
v = struct.unpack('<IHHIIIIIIIII',t[8:52])
|
||||
( retval['dwFlags'],
|
||||
retval['wPriority'],
|
||||
retval['wLanguage'],
|
||||
retval['dwInitialFrames'],
|
||||
retval['dwScale'],
|
||||
retval['dwRate'],
|
||||
retval['dwStart'],
|
||||
retval['dwLength'],
|
||||
retval['dwSuggestedBufferSize'],
|
||||
retval['dwQuality'],
|
||||
retval['dwSampleSize'],
|
||||
retval['rcFrame'] ) = v
|
||||
else:
|
||||
try:
|
||||
v = struct.unpack('<IHHIIIIIIIII',t[8:52])
|
||||
( retval['dwFlags'],
|
||||
retval['wPriority'],
|
||||
retval['wLanguage'],
|
||||
retval['dwInitialFrames'],
|
||||
retval['dwScale'],
|
||||
retval['dwRate'],
|
||||
retval['dwStart'],
|
||||
retval['dwLength'],
|
||||
retval['dwSuggestedBufferSize'],
|
||||
retval['dwQuality'],
|
||||
retval['dwSampleSize'],
|
||||
retval['rcFrame'] ) = v
|
||||
self.delay = float(retval['dwStart']) / \
|
||||
(float(retval['dwRate']) / retval['dwScale'])
|
||||
except (KeyError, IndexError, ValueError, ZeroDivisionError):
|
||||
pass
|
||||
|
||||
return retval
|
||||
|
||||
|
||||
def _parseSTRF(self,t,strh):
|
||||
fccType = strh['fccType']
|
||||
retval = {}
|
||||
if fccType == 'auds':
|
||||
v = struct.unpack('<HHHHHH',t[0:12])
|
||||
( retval['wFormatTag'],
|
||||
retval['nChannels'],
|
||||
retval['nSamplesPerSec'],
|
||||
retval['nAvgBytesPerSec'],
|
||||
retval['nBlockAlign'],
|
||||
retval['nBitsPerSample'],
|
||||
) = v
|
||||
ai = core.AudioStream()
|
||||
ai.samplerate = retval['nSamplesPerSec']
|
||||
ai.channels = retval['nChannels']
|
||||
# FIXME: Bitrate calculation is completely wrong.
|
||||
#ai.samplebits = retval['nBitsPerSample']
|
||||
#ai.bitrate = retval['nAvgBytesPerSec'] * 8
|
||||
|
||||
# TODO: set code if possible
|
||||
# http://www.stats.uwa.edu.au/Internal/Specs/DXALL/FileSpec/\
|
||||
# Languages
|
||||
# ai.language = strh['wLanguage']
|
||||
ai.codec = retval['wFormatTag']
|
||||
self.audio.append(ai)
|
||||
elif fccType == 'vids':
|
||||
v = struct.unpack('<IIIHH',t[0:16])
|
||||
( retval['biSize'],
|
||||
retval['biWidth'],
|
||||
retval['biHeight'],
|
||||
retval['biPlanes'],
|
||||
retval['biBitCount'] ) = v
|
||||
v = struct.unpack('IIIII',t[20:40])
|
||||
( retval['biSizeImage'],
|
||||
retval['biXPelsPerMeter'],
|
||||
retval['biYPelsPerMeter'],
|
||||
retval['biClrUsed'],
|
||||
retval['biClrImportant'] ) = v
|
||||
vi = core.VideoStream()
|
||||
vi.codec = t[16:20]
|
||||
vi.width = retval['biWidth']
|
||||
vi.height = retval['biHeight']
|
||||
# FIXME: Bitrate calculation is completely wrong.
|
||||
#vi.bitrate = strh['dwRate']
|
||||
vi.fps = float(strh['dwRate']) / strh['dwScale']
|
||||
vi.length = strh['dwLength'] / vi.fps
|
||||
self.video.append(vi)
|
||||
return retval
|
||||
|
||||
|
||||
def _parseSTRL(self,t):
|
||||
retval = {}
|
||||
size = len(t)
|
||||
i = 0
|
||||
|
||||
while i < len(t) - 8:
|
||||
key = t[i:i+4]
|
||||
sz = struct.unpack('<I',t[i+4:i+8])[0]
|
||||
i+=8
|
||||
value = t[i:]
|
||||
|
||||
if key == 'strh':
|
||||
retval[key] = self._parseSTRH(value)
|
||||
elif key == 'strf':
|
||||
retval[key] = self._parseSTRF(value, retval['strh'])
|
||||
else:
|
||||
log.debug("_parseSTRL: unsupported stream tag '%s'", key)
|
||||
|
||||
i += sz
|
||||
|
||||
return retval, i
|
||||
|
||||
|
||||
def _parseODML(self,t):
|
||||
retval = {}
|
||||
size = len(t)
|
||||
i = 0
|
||||
key = t[i:i+4]
|
||||
sz = struct.unpack('<I',t[i+4:i+8])[0]
|
||||
i += 8
|
||||
value = t[i:]
|
||||
if key != 'dmlh':
|
||||
log.debug("_parseODML: Error")
|
||||
|
||||
i += sz - 8
|
||||
return (retval, i)
|
||||
|
||||
|
||||
def _parseVPRP(self,t):
|
||||
retval = {}
|
||||
v = struct.unpack('<IIIIIIIIII',t[:4*10])
|
||||
|
||||
( retval['VideoFormat'],
|
||||
retval['VideoStandard'],
|
||||
retval['RefreshRate'],
|
||||
retval['HTotalIn'],
|
||||
retval['VTotalIn'],
|
||||
retval['FrameAspectRatio'],
|
||||
retval['wPixel'],
|
||||
retval['hPixel'] ) = v[1:-1]
|
||||
|
||||
# I need an avi with more informations
|
||||
# enum {FORMAT_UNKNOWN, FORMAT_PAL_SQUARE, FORMAT_PAL_CCIR_601,
|
||||
# FORMAT_NTSC_SQUARE, FORMAT_NTSC_CCIR_601,...} VIDEO_FORMAT;
|
||||
# enum {STANDARD_UNKNOWN, STANDARD_PAL, STANDARD_NTSC, STANDARD_SECAM}
|
||||
# VIDEO_STANDARD;
|
||||
#
|
||||
r = retval['FrameAspectRatio']
|
||||
r = float(r >> 16) / (r & 0xFFFF)
|
||||
retval['FrameAspectRatio'] = r
|
||||
if self.video:
|
||||
map(lambda v: setattr(v, 'aspect', r), self.video)
|
||||
return (retval, v[0])
|
||||
|
||||
|
||||
def _parseLISTmovi(self, size, file):
|
||||
"""
|
||||
Digs into movi list, looking for a Video Object Layer header in an
|
||||
mpeg4 stream in order to determine aspect ratio.
|
||||
"""
|
||||
i = 0
|
||||
n_dc = 0
|
||||
done = False
|
||||
# If the VOL header doesn't appear within 5MB or 5 video chunks,
|
||||
# give up. The 5MB limit is not likely to apply except in
|
||||
# pathological cases.
|
||||
while i < min(1024*1024*5, size - 8) and n_dc < 5:
|
||||
data = file.read(8)
|
||||
if ord(data[0]) == 0:
|
||||
# Eat leading nulls.
|
||||
data = data[1:] + file.read(1)
|
||||
i += 1
|
||||
|
||||
key, sz = struct.unpack('<4sI', data)
|
||||
if key[2:] != 'dc' or sz > 1024*500:
|
||||
# This chunk is not video or is unusually big (> 500KB);
|
||||
# skip it.
|
||||
file.seek(sz, 1)
|
||||
i += 8 + sz
|
||||
continue
|
||||
|
||||
n_dc += 1
|
||||
# Read video chunk into memory
|
||||
data = file.read(sz)
|
||||
|
||||
#for p in range(0,min(80, sz)):
|
||||
# print "%02x " % ord(data[p]),
|
||||
#print "\n\n"
|
||||
|
||||
# Look through the picture header for VOL startcode. The basic
|
||||
# logic for this is taken from libavcodec, h263.c
|
||||
pos = 0
|
||||
startcode = 0xff
|
||||
def bits(v, o, n):
|
||||
# Returns n bits in v, offset o bits.
|
||||
return (v & 2**n-1 << (64-n-o)) >> 64-n-o
|
||||
|
||||
while pos < sz:
|
||||
startcode = ((startcode << 8) | ord(data[pos])) & 0xffffffff
|
||||
pos += 1
|
||||
if startcode & 0xFFFFFF00 != 0x100:
|
||||
# No startcode found yet
|
||||
continue
|
||||
|
||||
if startcode >= 0x120 and startcode <= 0x12F:
|
||||
# We have the VOL startcode. Pull 64 bits of it and treat
|
||||
# as a bitstream
|
||||
v = struct.unpack(">Q", data[pos : pos+8])[0]
|
||||
offset = 10
|
||||
if bits(v, 9, 1):
|
||||
# is_ol_id, skip over vo_ver_id and vo_priority
|
||||
offset += 7
|
||||
ar_info = bits(v, offset, 4)
|
||||
if ar_info == 15:
|
||||
# Extended aspect
|
||||
num = bits(v, offset + 4, 8)
|
||||
den = bits(v, offset + 12, 8)
|
||||
else:
|
||||
# A standard pixel aspect
|
||||
num, den = PIXEL_ASPECT.get(ar_info, (0, 0))
|
||||
|
||||
# num/den indicates pixel aspect; convert to video aspect,
|
||||
# so we need frame width and height.
|
||||
if 0 not in [num, den]:
|
||||
width, height = self.video[-1].width, self.video[-1].height
|
||||
self.video[-1].aspect = num / float(den) * width / height
|
||||
|
||||
done = True
|
||||
break
|
||||
|
||||
startcode = 0xff
|
||||
|
||||
i += 8 + len(data)
|
||||
|
||||
if done:
|
||||
# We have the aspect, no need to continue parsing the movi
|
||||
# list, so break out of the loop.
|
||||
break
|
||||
|
||||
|
||||
if i < size:
|
||||
# Seek past whatever might be remaining of the movi list.
|
||||
file.seek(size-i,1)
|
||||
|
||||
|
||||
|
||||
def _parseLIST(self,t):
|
||||
retval = {}
|
||||
i = 0
|
||||
size = len(t)
|
||||
|
||||
while i < size-8:
|
||||
# skip zero
|
||||
if ord(t[i]) == 0: i += 1
|
||||
key = t[i:i+4]
|
||||
sz = 0
|
||||
|
||||
if key == 'LIST':
|
||||
sz = struct.unpack('<I',t[i+4:i+8])[0]
|
||||
i+=8
|
||||
key = "LIST:"+t[i:i+4]
|
||||
value = self._parseLIST(t[i:i+sz])
|
||||
if key == 'strl':
|
||||
for k in value.keys():
|
||||
retval[k] = value[k]
|
||||
else:
|
||||
retval[key] = value
|
||||
i+=sz
|
||||
elif key == 'avih':
|
||||
sz = struct.unpack('<I',t[i+4:i+8])[0]
|
||||
i += 8
|
||||
value = self._parseAVIH(t[i:i+sz])
|
||||
i += sz
|
||||
retval[key] = value
|
||||
elif key == 'strl':
|
||||
i += 4
|
||||
(value, sz) = self._parseSTRL(t[i:])
|
||||
key = value['strh']['fccType']
|
||||
i += sz
|
||||
retval[key] = value
|
||||
elif key == 'odml':
|
||||
i += 4
|
||||
(value, sz) = self._parseODML(t[i:])
|
||||
i += sz
|
||||
elif key == 'vprp':
|
||||
i += 4
|
||||
(value, sz) = self._parseVPRP(t[i:])
|
||||
retval[key] = value
|
||||
i += sz
|
||||
elif key == 'JUNK':
|
||||
sz = struct.unpack('<I',t[i+4:i+8])[0]
|
||||
i += sz + 8
|
||||
else:
|
||||
sz = struct.unpack('<I',t[i+4:i+8])[0]
|
||||
i+=8
|
||||
# in most cases this is some info stuff
|
||||
if not key in AVIINFO.keys() and key != 'IDIT':
|
||||
log.debug("Unknown Key: %s, len: %d" % (key,sz))
|
||||
value = t[i:i+sz]
|
||||
if key == 'ISFT':
|
||||
# product information
|
||||
if value.find('\0') > 0:
|
||||
# works for Casio S500 camera videos
|
||||
value = value[:value.find('\0')]
|
||||
value = value.replace('\0', '').lstrip().rstrip()
|
||||
value = value.replace('\0', '').lstrip().rstrip()
|
||||
if value:
|
||||
retval[key] = value
|
||||
if key in ['IDIT', 'ICRD']:
|
||||
# Timestamp the video was created. Spec says it
|
||||
# should be a format like "Wed Jan 02 02:03:55 1990"
|
||||
# Casio S500 uses "2005/12/24/ 14:11", but I've
|
||||
# also seen "December 24, 2005"
|
||||
specs = ('%a %b %d %H:%M:%S %Y', '%Y/%m/%d/ %H:%M', '%B %d, %Y')
|
||||
for tmspec in specs:
|
||||
try:
|
||||
tm = time.strptime(value, tmspec)
|
||||
# save timestamp as int
|
||||
self.timestamp = int(time.mktime(tm))
|
||||
break
|
||||
except ValueError:
|
||||
pass
|
||||
else:
|
||||
log.debug('no support for time format %s', value)
|
||||
i+=sz
|
||||
return retval
|
||||
|
||||
|
||||
def _parseRIFFChunk(self,file):
|
||||
h = file.read(8)
|
||||
if len(h) < 8:
|
||||
return False
|
||||
name = h[:4]
|
||||
size = struct.unpack('<I',h[4:8])[0]
|
||||
|
||||
if name == 'LIST':
|
||||
pos = file.tell() - 8
|
||||
key = file.read(4)
|
||||
if key == 'movi' and self.video and not self.video[-1].aspect and \
|
||||
self.video[-1].width and self.video[-1].height and \
|
||||
self.video[-1].format in ['DIVX', 'XVID', 'FMP4']: # any others?
|
||||
# If we don't have the aspect (i.e. it isn't in odml vprp
|
||||
# header), but we do know the video's dimensions, and
|
||||
# we're dealing with an mpeg4 stream, try to get the aspect
|
||||
# from the VOL header in the mpeg4 stream.
|
||||
self._parseLISTmovi(size-4, file)
|
||||
return True
|
||||
elif size > 80000:
|
||||
log.debug('RIFF LIST "%s" too long to parse: %s bytes' % (key, size))
|
||||
t = file.seek(size-4,1)
|
||||
return True
|
||||
elif size < 5:
|
||||
log.debug('RIFF LIST "%s" too short: %s bytes' % (key, size))
|
||||
return True
|
||||
|
||||
t = file.read(size-4)
|
||||
log.debug('parse RIFF LIST "%s": %d bytes' % (key, size))
|
||||
value = self._parseLIST(t)
|
||||
self.header[key] = value
|
||||
if key == 'INFO':
|
||||
self.infoStart = pos
|
||||
self._appendtable('AVIINFO', value)
|
||||
elif key == 'MID ':
|
||||
self._appendtable('AVIMID', value)
|
||||
elif key == 'hdrl':
|
||||
# no need to add this info to a table
|
||||
pass
|
||||
else:
|
||||
log.debug('Skipping table info %s' % key)
|
||||
|
||||
elif name == 'JUNK':
|
||||
self.junkStart = file.tell() - 8
|
||||
self.junkSize = size
|
||||
file.seek(size, 1)
|
||||
elif name == 'idx1':
|
||||
self.has_idx = True
|
||||
log.debug('idx1: %s bytes' % size)
|
||||
# no need to parse this
|
||||
t = file.seek(size,1)
|
||||
elif name == 'RIFF':
|
||||
log.debug("New RIFF chunk, extended avi [%i]" % size)
|
||||
type = file.read(4)
|
||||
if type != 'AVIX':
|
||||
log.debug("Second RIFF chunk is %s, not AVIX, skipping", type)
|
||||
file.seek(size-4, 1)
|
||||
# that's it, no new informations should be in AVIX
|
||||
return False
|
||||
elif name == 'fmt ' and size <= 50:
|
||||
# This is a wav file.
|
||||
data = file.read(size)
|
||||
fmt = struct.unpack("<HHLLHH", data[:16])
|
||||
self._set('codec', hex(fmt[0]))
|
||||
self._set('samplerate', fmt[2])
|
||||
# fmt[3] is average bytes per second, so we must divide it
|
||||
# by 125 to get kbits per second
|
||||
self._set('bitrate', fmt[3] / 125)
|
||||
# ugly hack: remember original rate in bytes per second
|
||||
# so that the length can be calculated in next elif block
|
||||
self._set('byterate', fmt[3])
|
||||
# Set a dummy fourcc so codec will be resolved in finalize.
|
||||
self._set('fourcc', 'dummy')
|
||||
elif name == 'data':
|
||||
# XXX: this is naive and may not be right. For example if the
|
||||
# stream is something that supports VBR like mp3, the value
|
||||
# will be off. The only way to properly deal with this issue
|
||||
# is to decode part of the stream based on its codec, but
|
||||
# kaa.metadata doesn't have this capability (yet?)
|
||||
# ugly hack: use original rate in bytes per second
|
||||
self._set('length', size / float(self.byterate))
|
||||
file.seek(size, 1)
|
||||
elif not name.strip(string.printable + string.whitespace):
|
||||
# check if name is something usefull at all, maybe it is no
|
||||
# avi or broken
|
||||
t = file.seek(size,1)
|
||||
log.debug("Skipping %s [%i]" % (name,size))
|
||||
else:
|
||||
# bad avi
|
||||
log.debug("Bad or broken avi")
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
Parser = Riff
|
||||
@@ -0,0 +1,80 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# enzyme - Video metadata parser
|
||||
# Copyright (C) 2011 Antoine Bertin <diaoulael@gmail.com>
|
||||
# Copyright (C) 2006-2009 Dirk Meyer <dischi@freevo.org>
|
||||
# Copyright (C) 2006-2009 Jason Tackaberry
|
||||
#
|
||||
# This file is part of enzyme.
|
||||
#
|
||||
# enzyme is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# enzyme is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__all__ = ['ENCODING', 'str_to_unicode', 'unicode_to_str']
|
||||
|
||||
import locale
|
||||
|
||||
# find the correct encoding
|
||||
try:
|
||||
ENCODING = locale.getdefaultlocale()[1]
|
||||
''.encode(ENCODING)
|
||||
except (UnicodeError, TypeError):
|
||||
ENCODING = 'latin-1'
|
||||
|
||||
|
||||
def str_to_unicode(s, encoding=None):
|
||||
"""
|
||||
Attempts to convert a string of unknown character set to a unicode
|
||||
string. First it tries to decode the string based on the locale's
|
||||
preferred encoding, and if that fails, fall back to UTF-8 and then
|
||||
latin-1. If all fails, it will force encoding to the preferred
|
||||
charset, replacing unknown characters. If the given object is no
|
||||
string, this function will return the given object.
|
||||
"""
|
||||
if not type(s) == str:
|
||||
return s
|
||||
|
||||
if not encoding:
|
||||
encoding = ENCODING
|
||||
|
||||
for c in [encoding, "utf-8", "latin-1"]:
|
||||
try:
|
||||
return s.decode(c)
|
||||
except UnicodeDecodeError:
|
||||
pass
|
||||
|
||||
return s.decode(encoding, "replace")
|
||||
|
||||
|
||||
def unicode_to_str(s, encoding=None):
|
||||
"""
|
||||
Attempts to convert a unicode string of unknown character set to a
|
||||
string. First it tries to encode the string based on the locale's
|
||||
preferred encoding, and if that fails, fall back to UTF-8 and then
|
||||
latin-1. If all fails, it will force encoding to the preferred
|
||||
charset, replacing unknown characters. If the given object is no
|
||||
unicode string, this function will return the given object.
|
||||
"""
|
||||
if not type(s) == unicode:
|
||||
return s
|
||||
|
||||
if not encoding:
|
||||
encoding = ENCODING
|
||||
|
||||
for c in [encoding, "utf-8", "latin-1"]:
|
||||
try:
|
||||
return s.encode(c)
|
||||
except UnicodeDecodeError:
|
||||
pass
|
||||
|
||||
return s.encode(encoding, "replace")
|
||||
Reference in New Issue
Block a user