Libraries for Subliminal
This commit is contained in:
@@ -0,0 +1,485 @@
|
||||
aar||aa|Afar|afar
|
||||
abk||ab|Abkhazian|abkhaze
|
||||
ace|||Achinese|aceh
|
||||
ach|||Acoli|acoli
|
||||
ada|||Adangme|adangme
|
||||
ady|||Adyghe; Adygei|adyghé
|
||||
afa|||Afro-Asiatic languages|afro-asiatiques, langues
|
||||
afh|||Afrihili|afrihili
|
||||
afr||af|Afrikaans|afrikaans
|
||||
ain|||Ainu|aïnou
|
||||
aka||ak|Akan|akan
|
||||
akk|||Akkadian|akkadien
|
||||
alb|sqi|sq|Albanian|albanais
|
||||
ale|||Aleut|aléoute
|
||||
alg|||Algonquian languages|algonquines, langues
|
||||
alt|||Southern Altai|altai du Sud
|
||||
amh||am|Amharic|amharique
|
||||
ang|||English, Old (ca.450-1100)|anglo-saxon (ca.450-1100)
|
||||
anp|||Angika|angika
|
||||
apa|||Apache languages|apaches, langues
|
||||
ara||ar|Arabic|arabe
|
||||
arc|||Official Aramaic (700-300 BCE); Imperial Aramaic (700-300 BCE)|araméen d'empire (700-300 BCE)
|
||||
arg||an|Aragonese|aragonais
|
||||
arm|hye|hy|Armenian|arménien
|
||||
arn|||Mapudungun; Mapuche|mapudungun; mapuche; mapuce
|
||||
arp|||Arapaho|arapaho
|
||||
art|||Artificial languages|artificielles, langues
|
||||
arw|||Arawak|arawak
|
||||
asm||as|Assamese|assamais
|
||||
ast|||Asturian; Bable; Leonese; Asturleonese|asturien; bable; léonais; asturoléonais
|
||||
ath|||Athapascan languages|athapascanes, langues
|
||||
aus|||Australian languages|australiennes, langues
|
||||
ava||av|Avaric|avar
|
||||
ave||ae|Avestan|avestique
|
||||
awa|||Awadhi|awadhi
|
||||
aym||ay|Aymara|aymara
|
||||
aze||az|Azerbaijani|azéri
|
||||
bad|||Banda languages|banda, langues
|
||||
bai|||Bamileke languages|bamiléké, langues
|
||||
bak||ba|Bashkir|bachkir
|
||||
bal|||Baluchi|baloutchi
|
||||
bam||bm|Bambara|bambara
|
||||
ban|||Balinese|balinais
|
||||
baq|eus|eu|Basque|basque
|
||||
bas|||Basa|basa
|
||||
bat|||Baltic languages|baltes, langues
|
||||
bej|||Beja; Bedawiyet|bedja
|
||||
bel||be|Belarusian|biélorusse
|
||||
bem|||Bemba|bemba
|
||||
ben||bn|Bengali|bengali
|
||||
ber|||Berber languages|berbères, langues
|
||||
bho|||Bhojpuri|bhojpuri
|
||||
bih||bh|Bihari languages|langues biharis
|
||||
bik|||Bikol|bikol
|
||||
bin|||Bini; Edo|bini; edo
|
||||
bis||bi|Bislama|bichlamar
|
||||
bla|||Siksika|blackfoot
|
||||
bnt|||Bantu (Other)|bantoues, autres langues
|
||||
bos||bs|Bosnian|bosniaque
|
||||
bra|||Braj|braj
|
||||
bre||br|Breton|breton
|
||||
btk|||Batak languages|batak, langues
|
||||
bua|||Buriat|bouriate
|
||||
bug|||Buginese|bugi
|
||||
bul||bg|Bulgarian|bulgare
|
||||
bur|mya|my|Burmese|birman
|
||||
byn|||Blin; Bilin|blin; bilen
|
||||
cad|||Caddo|caddo
|
||||
cai|||Central American Indian languages|amérindiennes de L'Amérique centrale, langues
|
||||
car|||Galibi Carib|karib; galibi; carib
|
||||
cat||ca|Catalan; Valencian|catalan; valencien
|
||||
cau|||Caucasian languages|caucasiennes, langues
|
||||
ceb|||Cebuano|cebuano
|
||||
cel|||Celtic languages|celtiques, langues; celtes, langues
|
||||
cha||ch|Chamorro|chamorro
|
||||
chb|||Chibcha|chibcha
|
||||
che||ce|Chechen|tchétchène
|
||||
chg|||Chagatai|djaghataï
|
||||
chi|zho|zh|Chinese|chinois
|
||||
chk|||Chuukese|chuuk
|
||||
chm|||Mari|mari
|
||||
chn|||Chinook jargon|chinook, jargon
|
||||
cho|||Choctaw|choctaw
|
||||
chp|||Chipewyan; Dene Suline|chipewyan
|
||||
chr|||Cherokee|cherokee
|
||||
chu||cu|Church Slavic; Old Slavonic; Church Slavonic; Old Bulgarian; Old Church Slavonic|slavon d'église; vieux slave; slavon liturgique; vieux bulgare
|
||||
chv||cv|Chuvash|tchouvache
|
||||
chy|||Cheyenne|cheyenne
|
||||
cmc|||Chamic languages|chames, langues
|
||||
cop|||Coptic|copte
|
||||
cor||kw|Cornish|cornique
|
||||
cos||co|Corsican|corse
|
||||
cpe|||Creoles and pidgins, English based|créoles et pidgins basés sur l'anglais
|
||||
cpf|||Creoles and pidgins, French-based |créoles et pidgins basés sur le français
|
||||
cpp|||Creoles and pidgins, Portuguese-based |créoles et pidgins basés sur le portugais
|
||||
cre||cr|Cree|cree
|
||||
crh|||Crimean Tatar; Crimean Turkish|tatar de Crimé
|
||||
crp|||Creoles and pidgins |créoles et pidgins
|
||||
csb|||Kashubian|kachoube
|
||||
cus|||Cushitic languages|couchitiques, langues
|
||||
cze|ces|cs|Czech|tchèque
|
||||
dak|||Dakota|dakota
|
||||
dan||da|Danish|danois
|
||||
dar|||Dargwa|dargwa
|
||||
day|||Land Dayak languages|dayak, langues
|
||||
del|||Delaware|delaware
|
||||
den|||Slave (Athapascan)|esclave (athapascan)
|
||||
dgr|||Dogrib|dogrib
|
||||
din|||Dinka|dinka
|
||||
div||dv|Divehi; Dhivehi; Maldivian|maldivien
|
||||
doi|||Dogri|dogri
|
||||
dra|||Dravidian languages|dravidiennes, langues
|
||||
dsb|||Lower Sorbian|bas-sorabe
|
||||
dua|||Duala|douala
|
||||
dum|||Dutch, Middle (ca.1050-1350)|néerlandais moyen (ca. 1050-1350)
|
||||
dut|nld|nl|Dutch; Flemish|néerlandais; flamand
|
||||
dyu|||Dyula|dioula
|
||||
dzo||dz|Dzongkha|dzongkha
|
||||
efi|||Efik|efik
|
||||
egy|||Egyptian (Ancient)|égyptien
|
||||
eka|||Ekajuk|ekajuk
|
||||
elx|||Elamite|élamite
|
||||
eng||en|English|anglais
|
||||
enm|||English, Middle (1100-1500)|anglais moyen (1100-1500)
|
||||
epo||eo|Esperanto|espéranto
|
||||
est||et|Estonian|estonien
|
||||
ewe||ee|Ewe|éwé
|
||||
ewo|||Ewondo|éwondo
|
||||
fan|||Fang|fang
|
||||
fao||fo|Faroese|féroïen
|
||||
fat|||Fanti|fanti
|
||||
fij||fj|Fijian|fidjien
|
||||
fil|||Filipino; Pilipino|filipino; pilipino
|
||||
fin||fi|Finnish|finnois
|
||||
fiu|||Finno-Ugrian languages|finno-ougriennes, langues
|
||||
fon|||Fon|fon
|
||||
fre|fra|fr|French|français
|
||||
frm|||French, Middle (ca.1400-1600)|français moyen (1400-1600)
|
||||
fro|||French, Old (842-ca.1400)|français ancien (842-ca.1400)
|
||||
frr|||Northern Frisian|frison septentrional
|
||||
frs|||Eastern Frisian|frison oriental
|
||||
fry||fy|Western Frisian|frison occidental
|
||||
ful||ff|Fulah|peul
|
||||
fur|||Friulian|frioulan
|
||||
gaa|||Ga|ga
|
||||
gay|||Gayo|gayo
|
||||
gba|||Gbaya|gbaya
|
||||
gem|||Germanic languages|germaniques, langues
|
||||
geo|kat|ka|Georgian|géorgien
|
||||
ger|deu|de|German|allemand
|
||||
gez|||Geez|guèze
|
||||
gil|||Gilbertese|kiribati
|
||||
gla||gd|Gaelic; Scottish Gaelic|gaélique; gaélique écossais
|
||||
gle||ga|Irish|irlandais
|
||||
glg||gl|Galician|galicien
|
||||
glv||gv|Manx|manx; mannois
|
||||
gmh|||German, Middle High (ca.1050-1500)|allemand, moyen haut (ca. 1050-1500)
|
||||
goh|||German, Old High (ca.750-1050)|allemand, vieux haut (ca. 750-1050)
|
||||
gon|||Gondi|gond
|
||||
gor|||Gorontalo|gorontalo
|
||||
got|||Gothic|gothique
|
||||
grb|||Grebo|grebo
|
||||
grc|||Greek, Ancient (to 1453)|grec ancien (jusqu'à 1453)
|
||||
gre|ell|el|Greek, Modern (1453-)|grec moderne (après 1453)
|
||||
grn||gn|Guarani|guarani
|
||||
gsw|||Swiss German; Alemannic; Alsatian|suisse alémanique; alémanique; alsacien
|
||||
guj||gu|Gujarati|goudjrati
|
||||
gwi|||Gwich'in|gwich'in
|
||||
hai|||Haida|haida
|
||||
hat||ht|Haitian; Haitian Creole|haïtien; créole haïtien
|
||||
hau||ha|Hausa|haoussa
|
||||
haw|||Hawaiian|hawaïen
|
||||
heb||he|Hebrew|hébreu
|
||||
her||hz|Herero|herero
|
||||
hil|||Hiligaynon|hiligaynon
|
||||
him|||Himachali languages; Western Pahari languages|langues himachalis; langues paharis occidentales
|
||||
hin||hi|Hindi|hindi
|
||||
hit|||Hittite|hittite
|
||||
hmn|||Hmong; Mong|hmong
|
||||
hmo||ho|Hiri Motu|hiri motu
|
||||
hrv||hr|Croatian|croate
|
||||
hsb|||Upper Sorbian|haut-sorabe
|
||||
hun||hu|Hungarian|hongrois
|
||||
hup|||Hupa|hupa
|
||||
iba|||Iban|iban
|
||||
ibo||ig|Igbo|igbo
|
||||
ice|isl|is|Icelandic|islandais
|
||||
ido||io|Ido|ido
|
||||
iii||ii|Sichuan Yi; Nuosu|yi de Sichuan
|
||||
ijo|||Ijo languages|ijo, langues
|
||||
iku||iu|Inuktitut|inuktitut
|
||||
ile||ie|Interlingue; Occidental|interlingue
|
||||
ilo|||Iloko|ilocano
|
||||
ina||ia|Interlingua (International Auxiliary Language Association)|interlingua (langue auxiliaire internationale)
|
||||
inc|||Indic languages|indo-aryennes, langues
|
||||
ind||id|Indonesian|indonésien
|
||||
ine|||Indo-European languages|indo-européennes, langues
|
||||
inh|||Ingush|ingouche
|
||||
ipk||ik|Inupiaq|inupiaq
|
||||
ira|||Iranian languages|iraniennes, langues
|
||||
iro|||Iroquoian languages|iroquoises, langues
|
||||
ita||it|Italian|italien
|
||||
jav||jv|Javanese|javanais
|
||||
jbo|||Lojban|lojban
|
||||
jpn||ja|Japanese|japonais
|
||||
jpr|||Judeo-Persian|judéo-persan
|
||||
jrb|||Judeo-Arabic|judéo-arabe
|
||||
kaa|||Kara-Kalpak|karakalpak
|
||||
kab|||Kabyle|kabyle
|
||||
kac|||Kachin; Jingpho|kachin; jingpho
|
||||
kal||kl|Kalaallisut; Greenlandic|groenlandais
|
||||
kam|||Kamba|kamba
|
||||
kan||kn|Kannada|kannada
|
||||
kar|||Karen languages|karen, langues
|
||||
kas||ks|Kashmiri|kashmiri
|
||||
kau||kr|Kanuri|kanouri
|
||||
kaw|||Kawi|kawi
|
||||
kaz||kk|Kazakh|kazakh
|
||||
kbd|||Kabardian|kabardien
|
||||
kha|||Khasi|khasi
|
||||
khi|||Khoisan languages|khoïsan, langues
|
||||
khm||km|Central Khmer|khmer central
|
||||
kho|||Khotanese; Sakan|khotanais; sakan
|
||||
kik||ki|Kikuyu; Gikuyu|kikuyu
|
||||
kin||rw|Kinyarwanda|rwanda
|
||||
kir||ky|Kirghiz; Kyrgyz|kirghiz
|
||||
kmb|||Kimbundu|kimbundu
|
||||
kok|||Konkani|konkani
|
||||
kom||kv|Komi|kom
|
||||
kon||kg|Kongo|kongo
|
||||
kor||ko|Korean|coréen
|
||||
kos|||Kosraean|kosrae
|
||||
kpe|||Kpelle|kpellé
|
||||
krc|||Karachay-Balkar|karatchai balkar
|
||||
krl|||Karelian|carélien
|
||||
kro|||Kru languages|krou, langues
|
||||
kru|||Kurukh|kurukh
|
||||
kua||kj|Kuanyama; Kwanyama|kuanyama; kwanyama
|
||||
kum|||Kumyk|koumyk
|
||||
kur||ku|Kurdish|kurde
|
||||
kut|||Kutenai|kutenai
|
||||
lad|||Ladino|judéo-espagnol
|
||||
lah|||Lahnda|lahnda
|
||||
lam|||Lamba|lamba
|
||||
lao||lo|Lao|lao
|
||||
lat||la|Latin|latin
|
||||
lav||lv|Latvian|letton
|
||||
lez|||Lezghian|lezghien
|
||||
lim||li|Limburgan; Limburger; Limburgish|limbourgeois
|
||||
lin||ln|Lingala|lingala
|
||||
lit||lt|Lithuanian|lituanien
|
||||
lol|||Mongo|mongo
|
||||
loz|||Lozi|lozi
|
||||
ltz||lb|Luxembourgish; Letzeburgesch|luxembourgeois
|
||||
lua|||Luba-Lulua|luba-lulua
|
||||
lub||lu|Luba-Katanga|luba-katanga
|
||||
lug||lg|Ganda|ganda
|
||||
lui|||Luiseno|luiseno
|
||||
lun|||Lunda|lunda
|
||||
luo|||Luo (Kenya and Tanzania)|luo (Kenya et Tanzanie)
|
||||
lus|||Lushai|lushai
|
||||
mac|mkd|mk|Macedonian|macédonien
|
||||
mad|||Madurese|madourais
|
||||
mag|||Magahi|magahi
|
||||
mah||mh|Marshallese|marshall
|
||||
mai|||Maithili|maithili
|
||||
mak|||Makasar|makassar
|
||||
mal||ml|Malayalam|malayalam
|
||||
man|||Mandingo|mandingue
|
||||
mao|mri|mi|Maori|maori
|
||||
map|||Austronesian languages|austronésiennes, langues
|
||||
mar||mr|Marathi|marathe
|
||||
mas|||Masai|massaï
|
||||
may|msa|ms|Malay|malais
|
||||
mdf|||Moksha|moksa
|
||||
mdr|||Mandar|mandar
|
||||
men|||Mende|mendé
|
||||
mga|||Irish, Middle (900-1200)|irlandais moyen (900-1200)
|
||||
mic|||Mi'kmaq; Micmac|mi'kmaq; micmac
|
||||
min|||Minangkabau|minangkabau
|
||||
mis|||Uncoded languages|langues non codées
|
||||
mkh|||Mon-Khmer languages|môn-khmer, langues
|
||||
mlg||mg|Malagasy|malgache
|
||||
mlt||mt|Maltese|maltais
|
||||
mnc|||Manchu|mandchou
|
||||
mni|||Manipuri|manipuri
|
||||
mno|||Manobo languages|manobo, langues
|
||||
moh|||Mohawk|mohawk
|
||||
mon||mn|Mongolian|mongol
|
||||
mos|||Mossi|moré
|
||||
mul|||Multiple languages|multilingue
|
||||
mun|||Munda languages|mounda, langues
|
||||
mus|||Creek|muskogee
|
||||
mwl|||Mirandese|mirandais
|
||||
mwr|||Marwari|marvari
|
||||
myn|||Mayan languages|maya, langues
|
||||
myv|||Erzya|erza
|
||||
nah|||Nahuatl languages|nahuatl, langues
|
||||
nai|||North American Indian languages|nord-amérindiennes, langues
|
||||
nap|||Neapolitan|napolitain
|
||||
nau||na|Nauru|nauruan
|
||||
nav||nv|Navajo; Navaho|navaho
|
||||
nbl||nr|Ndebele, South; South Ndebele|ndébélé du Sud
|
||||
nde||nd|Ndebele, North; North Ndebele|ndébélé du Nord
|
||||
ndo||ng|Ndonga|ndonga
|
||||
nds|||Low German; Low Saxon; German, Low; Saxon, Low|bas allemand; bas saxon; allemand, bas; saxon, bas
|
||||
nep||ne|Nepali|népalais
|
||||
new|||Nepal Bhasa; Newari|nepal bhasa; newari
|
||||
nia|||Nias|nias
|
||||
nic|||Niger-Kordofanian languages|nigéro-kordofaniennes, langues
|
||||
niu|||Niuean|niué
|
||||
nno||nn|Norwegian Nynorsk; Nynorsk, Norwegian|norvégien nynorsk; nynorsk, norvégien
|
||||
nob||nb|Bokmål, Norwegian; Norwegian Bokmål|norvégien bokmål
|
||||
nog|||Nogai|nogaï; nogay
|
||||
non|||Norse, Old|norrois, vieux
|
||||
nor||no|Norwegian|norvégien
|
||||
nqo|||N'Ko|n'ko
|
||||
nso|||Pedi; Sepedi; Northern Sotho|pedi; sepedi; sotho du Nord
|
||||
nub|||Nubian languages|nubiennes, langues
|
||||
nwc|||Classical Newari; Old Newari; Classical Nepal Bhasa|newari classique
|
||||
nya||ny|Chichewa; Chewa; Nyanja|chichewa; chewa; nyanja
|
||||
nym|||Nyamwezi|nyamwezi
|
||||
nyn|||Nyankole|nyankolé
|
||||
nyo|||Nyoro|nyoro
|
||||
nzi|||Nzima|nzema
|
||||
oci||oc|Occitan (post 1500); Provençal|occitan (après 1500); provençal
|
||||
oji||oj|Ojibwa|ojibwa
|
||||
ori||or|Oriya|oriya
|
||||
orm||om|Oromo|galla
|
||||
osa|||Osage|osage
|
||||
oss||os|Ossetian; Ossetic|ossète
|
||||
ota|||Turkish, Ottoman (1500-1928)|turc ottoman (1500-1928)
|
||||
oto|||Otomian languages|otomi, langues
|
||||
paa|||Papuan languages|papoues, langues
|
||||
pag|||Pangasinan|pangasinan
|
||||
pal|||Pahlavi|pahlavi
|
||||
pam|||Pampanga; Kapampangan|pampangan
|
||||
pan||pa|Panjabi; Punjabi|pendjabi
|
||||
pap|||Papiamento|papiamento
|
||||
pau|||Palauan|palau
|
||||
peo|||Persian, Old (ca.600-400 B.C.)|perse, vieux (ca. 600-400 av. J.-C.)
|
||||
per|fas|fa|Persian|persan
|
||||
phi|||Philippine languages|philippines, langues
|
||||
phn|||Phoenician|phénicien
|
||||
pli||pi|Pali|pali
|
||||
pol||pl|Polish|polonais
|
||||
pon|||Pohnpeian|pohnpei
|
||||
por||pt|Portuguese|portugais
|
||||
pra|||Prakrit languages|prâkrit, langues
|
||||
pro|||Provençal, Old (to 1500)|provençal ancien (jusqu'à 1500)
|
||||
pus||ps|Pushto; Pashto|pachto
|
||||
qaa-qtz|||Reserved for local use|réservée à l'usage local
|
||||
que||qu|Quechua|quechua
|
||||
raj|||Rajasthani|rajasthani
|
||||
rap|||Rapanui|rapanui
|
||||
rar|||Rarotongan; Cook Islands Maori|rarotonga; maori des îles Cook
|
||||
roa|||Romance languages|romanes, langues
|
||||
roh||rm|Romansh|romanche
|
||||
rom|||Romany|tsigane
|
||||
rum|ron|ro|Romanian; Moldavian; Moldovan|roumain; moldave
|
||||
run||rn|Rundi|rundi
|
||||
rup|||Aromanian; Arumanian; Macedo-Romanian|aroumain; macédo-roumain
|
||||
rus||ru|Russian|russe
|
||||
sad|||Sandawe|sandawe
|
||||
sag||sg|Sango|sango
|
||||
sah|||Yakut|iakoute
|
||||
sai|||South American Indian (Other)|indiennes d'Amérique du Sud, autres langues
|
||||
sal|||Salishan languages|salishennes, langues
|
||||
sam|||Samaritan Aramaic|samaritain
|
||||
san||sa|Sanskrit|sanskrit
|
||||
sas|||Sasak|sasak
|
||||
sat|||Santali|santal
|
||||
scn|||Sicilian|sicilien
|
||||
sco|||Scots|écossais
|
||||
sel|||Selkup|selkoupe
|
||||
sem|||Semitic languages|sémitiques, langues
|
||||
sga|||Irish, Old (to 900)|irlandais ancien (jusqu'à 900)
|
||||
sgn|||Sign Languages|langues des signes
|
||||
shn|||Shan|chan
|
||||
sid|||Sidamo|sidamo
|
||||
sin||si|Sinhala; Sinhalese|singhalais
|
||||
sio|||Siouan languages|sioux, langues
|
||||
sit|||Sino-Tibetan languages|sino-tibétaines, langues
|
||||
sla|||Slavic languages|slaves, langues
|
||||
slo|slk|sk|Slovak|slovaque
|
||||
slv||sl|Slovenian|slovène
|
||||
sma|||Southern Sami|sami du Sud
|
||||
sme||se|Northern Sami|sami du Nord
|
||||
smi|||Sami languages|sames, langues
|
||||
smj|||Lule Sami|sami de Lule
|
||||
smn|||Inari Sami|sami d'Inari
|
||||
smo||sm|Samoan|samoan
|
||||
sms|||Skolt Sami|sami skolt
|
||||
sna||sn|Shona|shona
|
||||
snd||sd|Sindhi|sindhi
|
||||
snk|||Soninke|soninké
|
||||
sog|||Sogdian|sogdien
|
||||
som||so|Somali|somali
|
||||
son|||Songhai languages|songhai, langues
|
||||
sot||st|Sotho, Southern|sotho du Sud
|
||||
spa||es|Spanish; Castilian|espagnol; castillan
|
||||
srd||sc|Sardinian|sarde
|
||||
srn|||Sranan Tongo|sranan tongo
|
||||
srp||sr|Serbian|serbe
|
||||
srr|||Serer|sérère
|
||||
ssa|||Nilo-Saharan languages|nilo-sahariennes, langues
|
||||
ssw||ss|Swati|swati
|
||||
suk|||Sukuma|sukuma
|
||||
sun||su|Sundanese|soundanais
|
||||
sus|||Susu|soussou
|
||||
sux|||Sumerian|sumérien
|
||||
swa||sw|Swahili|swahili
|
||||
swe||sv|Swedish|suédois
|
||||
syc|||Classical Syriac|syriaque classique
|
||||
syr|||Syriac|syriaque
|
||||
tah||ty|Tahitian|tahitien
|
||||
tai|||Tai languages|tai, langues
|
||||
tam||ta|Tamil|tamoul
|
||||
tat||tt|Tatar|tatar
|
||||
tel||te|Telugu|télougou
|
||||
tem|||Timne|temne
|
||||
ter|||Tereno|tereno
|
||||
tet|||Tetum|tetum
|
||||
tgk||tg|Tajik|tadjik
|
||||
tgl||tl|Tagalog|tagalog
|
||||
tha||th|Thai|thaï
|
||||
tib|bod|bo|Tibetan|tibétain
|
||||
tig|||Tigre|tigré
|
||||
tir||ti|Tigrinya|tigrigna
|
||||
tiv|||Tiv|tiv
|
||||
tkl|||Tokelau|tokelau
|
||||
tlh|||Klingon; tlhIngan-Hol|klingon
|
||||
tli|||Tlingit|tlingit
|
||||
tmh|||Tamashek|tamacheq
|
||||
tog|||Tonga (Nyasa)|tonga (Nyasa)
|
||||
ton||to|Tonga (Tonga Islands)|tongan (Îles Tonga)
|
||||
tpi|||Tok Pisin|tok pisin
|
||||
tsi|||Tsimshian|tsimshian
|
||||
tsn||tn|Tswana|tswana
|
||||
tso||ts|Tsonga|tsonga
|
||||
tuk||tk|Turkmen|turkmène
|
||||
tum|||Tumbuka|tumbuka
|
||||
tup|||Tupi languages|tupi, langues
|
||||
tur||tr|Turkish|turc
|
||||
tut|||Altaic languages|altaïques, langues
|
||||
tvl|||Tuvalu|tuvalu
|
||||
twi||tw|Twi|twi
|
||||
tyv|||Tuvinian|touva
|
||||
udm|||Udmurt|oudmourte
|
||||
uga|||Ugaritic|ougaritique
|
||||
uig||ug|Uighur; Uyghur|ouïgour
|
||||
ukr||uk|Ukrainian|ukrainien
|
||||
umb|||Umbundu|umbundu
|
||||
und|||Undetermined|indéterminée
|
||||
urd||ur|Urdu|ourdou
|
||||
uzb||uz|Uzbek|ouszbek
|
||||
vai|||Vai|vaï
|
||||
ven||ve|Venda|venda
|
||||
vie||vi|Vietnamese|vietnamien
|
||||
vol||vo|Volapük|volapük
|
||||
vot|||Votic|vote
|
||||
wak|||Wakashan languages|wakashanes, langues
|
||||
wal|||Walamo|walamo
|
||||
war|||Waray|waray
|
||||
was|||Washo|washo
|
||||
wel|cym|cy|Welsh|gallois
|
||||
wen|||Sorbian languages|sorabes, langues
|
||||
wln||wa|Walloon|wallon
|
||||
wol||wo|Wolof|wolof
|
||||
xal|||Kalmyk; Oirat|kalmouk; oïrat
|
||||
xho||xh|Xhosa|xhosa
|
||||
yao|||Yao|yao
|
||||
yap|||Yapese|yapois
|
||||
yid||yi|Yiddish|yiddish
|
||||
yor||yo|Yoruba|yoruba
|
||||
ypk|||Yupik languages|yupik, langues
|
||||
zap|||Zapotec|zapotèque
|
||||
zbl|||Blissymbols; Blissymbolics; Bliss|symboles Bliss; Bliss
|
||||
zen|||Zenaga|zenaga
|
||||
zha||za|Zhuang; Chuang|zhuang; chuang
|
||||
znd|||Zande languages|zandé, langues
|
||||
zul||zu|Zulu|zoulou
|
||||
zun|||Zuni|zuni
|
||||
zxx|||No linguistic content; Not applicable|pas de contenu linguistique; non applicable
|
||||
zza|||Zaza; Dimili; Dimli; Kirdki; Kirmanjki; Zazaki|zaza; dimili; dimli; kirdki; kirmanjki; zazaki
|
||||
@@ -0,0 +1,130 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__version__ = '0.2'
|
||||
__all__ = [ 'Guess', 'Language',
|
||||
'guess_file_info', 'guess_video_info',
|
||||
'guess_movie_info', 'guess_episode_info' ]
|
||||
|
||||
|
||||
from guessit.guess import Guess, merge_all
|
||||
from guessit.language import Language
|
||||
from guessit.matcher import IterativeMatcher
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit")
|
||||
|
||||
class NullHandler(logging.Handler):
|
||||
def emit(self, record):
|
||||
pass
|
||||
|
||||
# let's be a nicely behaving library
|
||||
h = NullHandler()
|
||||
log.addHandler(h)
|
||||
|
||||
|
||||
|
||||
|
||||
def guess_file_info(filename, filetype, info = [ 'filename' ]):
|
||||
"""info can contain the names of the various plugins, such as 'filename' to
|
||||
detect filename info, or 'hash_md5' to get the md5 hash of the file.
|
||||
|
||||
>>> guess_file_info('test/dummy.srt', 'autodetect', info = ['hash_md5', 'hash_sha1'])
|
||||
{'hash_md5': 'e781de9b94ba2753a8e2945b2c0a123d', 'hash_sha1': 'bfd18e2f4e5d59775c2bc14d80f56971891ed620'}
|
||||
"""
|
||||
result = []
|
||||
hashers = []
|
||||
|
||||
for infotype in info:
|
||||
if infotype == 'filename':
|
||||
m = IterativeMatcher(filename, filetype = filetype)
|
||||
result.append(m.matched())
|
||||
|
||||
elif infotype == 'hash_mpc':
|
||||
import hash_mpc
|
||||
try:
|
||||
result.append(Guess({ 'hash_mpc': hash_mpc.hash_file(filename) },
|
||||
confidence = 1.0))
|
||||
except Exception, e:
|
||||
log.warning('Could not compute MPC-style hash because: %s' % e)
|
||||
|
||||
elif infotype == 'hash_ed2k':
|
||||
import hash_ed2k
|
||||
try:
|
||||
result.append(Guess({ 'hash_ed2k': hash_ed2k.hash_file(filename) },
|
||||
confidence = 1.0))
|
||||
except Exception, e:
|
||||
log.warning('Could not compute ed2k hash because: %s' % e)
|
||||
|
||||
elif infotype.startswith('hash_'):
|
||||
import hashlib
|
||||
hashname = infotype[5:]
|
||||
try:
|
||||
hasher = getattr(hashlib, hashname)()
|
||||
hashers.append((infotype, hasher))
|
||||
except AttributeError:
|
||||
log.warning('Could not compute %s hash because it is not available from python\'s hashlib module' % hashname)
|
||||
|
||||
else:
|
||||
log.warning('Invalid infotype: %s' % infotype)
|
||||
|
||||
|
||||
"""For plugins which depend on some optional library, import them like that:
|
||||
|
||||
if infotype == 'plugin_name':
|
||||
try:
|
||||
import optional_lib
|
||||
except ImportError:
|
||||
raise Exception, 'The plugin module cannot be loaded because the optional_lib lib is missing'
|
||||
|
||||
# do some stuff
|
||||
"""
|
||||
|
||||
# do all the hashes now, but on a single pass
|
||||
if hashers:
|
||||
try:
|
||||
blocksize = 8192
|
||||
hasherobjs = dict(hashers).values()
|
||||
|
||||
with open(filename, 'rb') as f:
|
||||
for chunk in iter(lambda: f.read(blocksize), ''):
|
||||
for hasher in hasherobjs:
|
||||
hasher.update(chunk)
|
||||
|
||||
for infotype, hasher in hashers:
|
||||
result.append(Guess({ infotype: hasher.hexdigest() },
|
||||
confidence = 1.0))
|
||||
except Exception, e:
|
||||
log.warning('Could not compute hash because: %s' % e)
|
||||
|
||||
|
||||
return merge_all(result)
|
||||
|
||||
|
||||
def guess_video_info(filename, info = [ 'filename' ]):
|
||||
return guess_file_info(filename, 'autodetect', info)
|
||||
|
||||
def guess_movie_info(filename, info = [ 'filename' ]):
|
||||
return guess_file_info(filename, 'movie', info)
|
||||
|
||||
def guess_episode_info(filename, info = [ 'filename' ]):
|
||||
return guess_file_info(filename, 'episode', info)
|
||||
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
#from guessit import movie, episode
|
||||
import os, os.path
|
||||
import logging
|
||||
|
||||
log = logging.getLogger('guessit.autodetect')
|
||||
|
||||
def within(x, nrange):
|
||||
"""Return whether a number is inside a given range, specified as a list or tuple
|
||||
of the lower and upper bounds."""
|
||||
low, high = nrange
|
||||
return low <= x <= high
|
||||
|
||||
def guess_filename_info(filename):
|
||||
log.debug('Trying to guess info for file: ' + filename)
|
||||
|
||||
# try to guess info as if it were an episode
|
||||
episode_info = episode.guess_episode_filename(filename)
|
||||
|
||||
# 1- if we found either season/episodeNumber, then we're pretty sure it must
|
||||
# be an episode
|
||||
if 'season' in episode_info or 'episodeNumber' in episode_info:
|
||||
log.debug('Likely an episode as it contains season and/or episodeNumber: ' + filename)
|
||||
episode_info.update({ 'type': 'episode' }, confidence = 0.9)
|
||||
return episode_info
|
||||
|
||||
# try to guess info as if it were a movie
|
||||
movie_info = movie.guess_movie_filename(filename)
|
||||
|
||||
# 2- if the file exists, try to guess its type using its size
|
||||
if os.path.exists(filename):
|
||||
size = os.stat(filename).st_size / (1024 * 1024)
|
||||
|
||||
# if size <= 1/2 of 1CD -> episode (very unlikely a movie so small)
|
||||
if size < 400:
|
||||
log.debug('Likely an episode due to its small size (%dMB): %s' % (size, filename))
|
||||
episode_info.update({ 'type': 'episode' }, confidence = 0.8)
|
||||
return episode_info
|
||||
|
||||
# if size > 2G -> movie (even fullHD eps aren't that big yet)
|
||||
if size > 2048:
|
||||
log.debug('Likely a movie due to its big size (%dMB): %s' % (size, filename))
|
||||
movie_info.update({ 'type': 'movie' }, confidence = 0.8)
|
||||
return movie_info
|
||||
|
||||
# if size == 1CD or 2CDs -> movie
|
||||
if within(size, [690, 710]) or within(size, [1380, 1420]):
|
||||
log.debug('Likely a movie due to its size close to a CD size (%dMB): %s' % (size, filename))
|
||||
movie_info.update({ 'type': 'movie' }, confidence = 0.8)
|
||||
return movie_info
|
||||
|
||||
|
||||
# 3- if all else fails, assume it's a movie
|
||||
log.debug('Couldn\'t make an informed guess... Assuming file is a movie: %s' % filename)
|
||||
movie_info.update({ 'type': 'movie' }, confidence = 0.5)
|
||||
return movie_info
|
||||
@@ -0,0 +1,127 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
import datetime
|
||||
import re
|
||||
|
||||
def search_year(string):
|
||||
"""Looks for year patterns, and if found return the year and group span.
|
||||
Assumes there are sentinels at the beginning and end of the string that
|
||||
always allow matching a non-digit delimiting the date.
|
||||
|
||||
Note this only looks for valid production years, that is between 1920
|
||||
and now + 5 years, so for instance 2000 would be returned as a valid
|
||||
year but 1492 would not.
|
||||
|
||||
>>> search_year('in the year 2000...')
|
||||
(2000, (12, 16))
|
||||
|
||||
>>> search_year('they arrived in 1492.')
|
||||
(None, None)
|
||||
"""
|
||||
match = re.search(r'[^0-9]([0-9]{4})[^0-9]', string)
|
||||
if match:
|
||||
year = int(match.group(1))
|
||||
if 1920 < year < datetime.date.today().year + 5:
|
||||
return (year, match.span(1))
|
||||
|
||||
return (None, None)
|
||||
|
||||
|
||||
def search_date(string):
|
||||
"""Looks for date patterns, and if found return the date and group span.
|
||||
Assumes there are sentinels at the beginning and end of the string that
|
||||
always allow matching a non-digit delimiting the date.
|
||||
|
||||
>>> search_date('This happened on 2002-04-22.')
|
||||
(datetime.date(2002, 4, 22), (17, 27))
|
||||
|
||||
>>> search_date('And this on 17-06-1998.')
|
||||
(datetime.date(1998, 6, 17), (12, 22))
|
||||
|
||||
>>> search_date('no date in here')
|
||||
(None, None)
|
||||
"""
|
||||
|
||||
dsep = r'[-/ \.]'
|
||||
|
||||
date_rexps = [ # 20010823
|
||||
r'[^0-9]' +
|
||||
r'(?P<year>[0-9]{4})' +
|
||||
r'(?P<month>[0-9]{2})' +
|
||||
r'(?P<day>[0-9]{2})' +
|
||||
r'[^0-9]',
|
||||
|
||||
# 2001-08-23
|
||||
r'[^0-9]' +
|
||||
r'(?P<year>[0-9]{4})' + dsep +
|
||||
r'(?P<month>[0-9]{2})' + dsep +
|
||||
r'(?P<day>[0-9]{2})' +
|
||||
r'[^0-9]',
|
||||
|
||||
# 23-08-2001
|
||||
r'[^0-9]' +
|
||||
r'(?P<day>[0-9]{2})' + dsep +
|
||||
r'(?P<month>[0-9]{2})' + dsep +
|
||||
r'(?P<year>[0-9]{4})' +
|
||||
r'[^0-9]',
|
||||
|
||||
# 23-08-01
|
||||
r'[^0-9]' +
|
||||
r'(?P<day>[0-9]{2})' + dsep +
|
||||
r'(?P<month>[0-9]{2})' + dsep +
|
||||
r'(?P<year>[0-9]{2})' +
|
||||
r'[^0-9]',
|
||||
]
|
||||
|
||||
for drexp in date_rexps:
|
||||
match = re.search(drexp, string)
|
||||
if match:
|
||||
d = match.groupdict()
|
||||
year, month, day = int(d['year']), int(d['month']), int(d['day'])
|
||||
# years specified as 2 digits should be adjusted here
|
||||
if year < 100:
|
||||
if year > (datetime.date.today().year % 100)+ 5:
|
||||
year = 1900 + year
|
||||
else:
|
||||
year = 2000 + year
|
||||
|
||||
date = None
|
||||
try:
|
||||
date = datetime.date(year, month, day)
|
||||
except ValueError:
|
||||
try:
|
||||
date = datetime.date(year, day, month)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
if date is None:
|
||||
continue
|
||||
|
||||
# check date plausibility
|
||||
if not 1900 < date.year < datetime.date.today().year + 5:
|
||||
continue
|
||||
|
||||
# looks like we have a valid date
|
||||
# note: span is [+1,-1] because we don't want to include the non-digit char
|
||||
start, end = match.span()
|
||||
return (date, (start+1, end-1))
|
||||
|
||||
return None, None
|
||||
@@ -0,0 +1,84 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
import ntpath
|
||||
import os.path
|
||||
|
||||
|
||||
def split_path(path):
|
||||
r"""Splits the given path into the list of folders and the filename (or the
|
||||
last folder if you gave it a folder path.
|
||||
|
||||
If the given path was an absolute path, the first element will always be:
|
||||
- the '/' root folder on Unix systems
|
||||
- the drive letter on Windows systems (eg: r'C:\')
|
||||
|
||||
>>> split_path('/usr/bin/smewt')
|
||||
['/', 'usr', 'bin', 'smewt']
|
||||
|
||||
>>> split_path('relative_path/to/my_folder/')
|
||||
['relative_path', 'to', 'my_folder']
|
||||
|
||||
>>> split_path(r'C:\Program Files\Smewt\smewt.exe')
|
||||
['C:\\', 'Program Files', 'Smewt', 'smewt.exe']
|
||||
|
||||
>>> split_path(r'Documents and Settings\User\config\\')
|
||||
['Documents and Settings', 'User', 'config']
|
||||
|
||||
"""
|
||||
result = []
|
||||
while True:
|
||||
head, tail = ntpath.split(path)
|
||||
|
||||
# on Unix systems, the root folder is '/'
|
||||
if head == '/' and tail == '':
|
||||
return [ '/' ] + result
|
||||
|
||||
# on Windows, the root folder is a drive letter (eg: 'C:\')
|
||||
if len(head) == 3 and head[1:] == ':\\' and tail == '':
|
||||
return [ head ] + result
|
||||
|
||||
if head == '' and tail == '':
|
||||
return result
|
||||
|
||||
# we just split a directory ending with '/', so tail is empty
|
||||
if not tail:
|
||||
path = head
|
||||
continue
|
||||
|
||||
result = [ tail ] + result
|
||||
path = head
|
||||
|
||||
|
||||
def split_path_components(filename):
|
||||
"""Returns the filename split into [ dir*, basename, ext ]."""
|
||||
result = split_path(filename)
|
||||
basename = result.pop(-1)
|
||||
return result + list(os.path.splitext(basename))
|
||||
|
||||
|
||||
def file_in_same_dir(ref_file, desired_file):
|
||||
"""Return the path for a file in the same dir as a given reference file.
|
||||
|
||||
>>> file_in_same_dir('~/smewt/smewt.db', 'smewt.settings')
|
||||
'~/smewt/smewt.settings'
|
||||
|
||||
"""
|
||||
return os.path.join(*(split_path(ref_file)[:-1] + [ desired_file ]))
|
||||
@@ -0,0 +1,317 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
import json
|
||||
import datetime
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.guess")
|
||||
|
||||
|
||||
class Guess(dict):
|
||||
"""A Guess is a dictionary which has an associated confidence for each of its values.
|
||||
|
||||
As it is a subclass of dict, you can use it everywhere you expect a simple dict"""
|
||||
def __init__(self, *args, **kwargs):
|
||||
try:
|
||||
confidence = kwargs.pop('confidence')
|
||||
except KeyError:
|
||||
confidence = 0
|
||||
|
||||
dict.__init__(self, *args, **kwargs)
|
||||
|
||||
self._confidence = {}
|
||||
for prop in self:
|
||||
self._confidence[prop] = confidence
|
||||
|
||||
def to_utf8_dict(self):
|
||||
from guessit.language import Language
|
||||
data = dict(self)
|
||||
for prop, value in data.items():
|
||||
if isinstance(value, datetime.date):
|
||||
data[prop] = value.isoformat()
|
||||
elif isinstance(value, Language):
|
||||
data[prop] = str(value)
|
||||
elif isinstance(value, unicode):
|
||||
data[prop] = value.encode('utf-8')
|
||||
elif isinstance(value, list):
|
||||
data[prop] = [ str(x) for x in value ]
|
||||
|
||||
return data
|
||||
|
||||
def nice_string(self):
|
||||
data = self.to_utf8_dict()
|
||||
|
||||
parts = json.dumps(data, indent = 4).split('\n')
|
||||
for i, p in enumerate(parts):
|
||||
if p[:5] != ' "':
|
||||
continue
|
||||
|
||||
prop = p.split('"')[1]
|
||||
parts[i] = (' [%.2f] "' % (self._confidence.get(prop) or -1)) + p[5:]
|
||||
|
||||
return '\n'.join(parts)
|
||||
|
||||
def __str__(self):
|
||||
return str(self.to_utf8_dict())
|
||||
|
||||
def confidence(self, prop):
|
||||
return self._confidence[prop]
|
||||
|
||||
def set(self, prop, value, confidence = None):
|
||||
self[prop] = value
|
||||
if confidence is not None:
|
||||
self._confidence[prop] = confidence
|
||||
|
||||
def set_confidence(self, prop, value):
|
||||
self._confidence[prop] = value
|
||||
|
||||
def update(self, other, confidence = None):
|
||||
dict.update(self, other)
|
||||
if isinstance(other, Guess):
|
||||
for prop in other:
|
||||
self._confidence[prop] = other.confidence(prop)
|
||||
|
||||
if confidence is not None:
|
||||
for prop in other:
|
||||
self._confidence[prop] = confidence
|
||||
|
||||
def update_highest_confidence(self, other):
|
||||
"""Update this guess with the values from the given one. In case there is
|
||||
property present in both, only the one with the highest one is kept."""
|
||||
if not isinstance(other, Guess):
|
||||
raise ValueError, 'Can only call this function on Guess instances'
|
||||
|
||||
for prop in other:
|
||||
if prop in self and self._confidence[prop] >= other._confidence[prop]:
|
||||
continue
|
||||
self[prop] = other[prop]
|
||||
self._confidence[prop] = other._confidence[prop]
|
||||
|
||||
|
||||
|
||||
|
||||
def choose_int(g1, g2):
|
||||
"""Function used by merge_similar_guesses to choose between 2 possible properties
|
||||
when they are integers."""
|
||||
v1, c1 = g1 # value, confidence
|
||||
v2, c2 = g2
|
||||
if (v1 == v2):
|
||||
return (v1, 1 - (1-c1)*(1-c2))
|
||||
else:
|
||||
if c1 > c2:
|
||||
return (v1, c1 - c2)
|
||||
else:
|
||||
return (v2, c2 - c1)
|
||||
|
||||
def choose_string(g1, g2):
|
||||
"""Function used by merge_similar_guesses to choose between 2 possible properties
|
||||
when they are strings.
|
||||
|
||||
If the 2 strings are similar, or one is contained in the other, the latter is returned
|
||||
with an increased confidence.
|
||||
|
||||
If the 2 strings are dissimilar, the one with the higher confidence is returned, with
|
||||
a weaker confidence.
|
||||
|
||||
Note that here, 'similar' means that 2 strings are either equal, or that they
|
||||
differ very little, such as one string being the other one with the 'the' word
|
||||
prepended to it.
|
||||
|
||||
>>> choose_string(('Hello', 0.75), ('World', 0.5))
|
||||
('Hello', 0.25)
|
||||
|
||||
>>> choose_string(('Hello', 0.5), ('hello', 0.5))
|
||||
('Hello', 0.75)
|
||||
|
||||
>>> choose_string(('Hello', 0.4), ('Hello World', 0.4))
|
||||
('Hello', 0.64000000000000001)
|
||||
|
||||
>>> choose_string(('simpsons', 0.5), ('The Simpsons', 0.5))
|
||||
('The Simpsons', 0.75)
|
||||
|
||||
"""
|
||||
v1, c1 = g1 # value, confidence
|
||||
v2, c2 = g2
|
||||
|
||||
if not v1:
|
||||
return g2
|
||||
elif not v2:
|
||||
return g1
|
||||
|
||||
v1, v2 = v1.strip(), v2.strip()
|
||||
v1l, v2l = v1.lower(), v2.lower()
|
||||
|
||||
combined_prob = 1 - (1-c1)*(1-c2)
|
||||
|
||||
if v1l == v2l:
|
||||
return (v1, combined_prob)
|
||||
|
||||
# check for common patterns
|
||||
elif v1l == 'the ' + v2l:
|
||||
return (v1, combined_prob)
|
||||
elif v2l == 'the ' + v1l:
|
||||
return (v2, combined_prob)
|
||||
|
||||
# if one string is contained in the other, return the shortest one
|
||||
elif v2l in v1l:
|
||||
return (v2, combined_prob)
|
||||
elif v1l in v2l:
|
||||
return (v1, combined_prob)
|
||||
|
||||
# in case of conflict, return the one with highest priority
|
||||
else:
|
||||
if c1 > c2:
|
||||
return (v1, c1 - c2)
|
||||
else:
|
||||
return (v2, c2 - c1)
|
||||
|
||||
|
||||
def _merge_similar_guesses_nocheck(guesses, prop, choose):
|
||||
"""Take a list of guesses and merge those which have the same properties,
|
||||
increasing or decreasing the confidence depending on whether their values
|
||||
are similar.
|
||||
|
||||
This function assumes there are at least 2 valid guesses."""
|
||||
|
||||
similar = [ guess for guess in guesses if prop in guess ]
|
||||
|
||||
g1, g2 = similar[0], similar[1]
|
||||
|
||||
other_props = set(g1) & set(g2) - set([prop])
|
||||
if other_props:
|
||||
for prop in other_props:
|
||||
if g1[prop] != g2[prop]:
|
||||
log.warning('both guesses to be merged have more than one different property in common, bailing out...')
|
||||
return
|
||||
|
||||
# merge all props of s2 into s1, updating the confidence for the considered property
|
||||
v1, v2 = g1[prop], g2[prop]
|
||||
c1, c2 = g1.confidence(prop), g2.confidence(prop)
|
||||
|
||||
new_value, new_confidence = choose((v1, c1), (v2, c2))
|
||||
if new_confidence >= c1:
|
||||
log.debug("Updating matching property '%s' with confidence %.2f" % (prop, new_confidence))
|
||||
else:
|
||||
log.debug("Updating non-matching property '%s' with confidence %.2f" % (prop, new_confidence))
|
||||
|
||||
g2[prop] = new_value
|
||||
g2.set_confidence(prop, new_confidence)
|
||||
|
||||
g1.update(g2)
|
||||
guesses.remove(g2)
|
||||
|
||||
def merge_similar_guesses(guesses, prop, choose):
|
||||
"""Take a list of guesses and merge those which have the same properties,
|
||||
increasing or decreasing the confidence depending on whether their values
|
||||
are similar."""
|
||||
|
||||
similar = [ guess for guess in guesses if prop in guess ]
|
||||
if len(similar) < 2:
|
||||
# nothing to merge
|
||||
return
|
||||
|
||||
if len(similar) == 2:
|
||||
_merge_similar_guesses_nocheck(guesses, prop, choose)
|
||||
|
||||
if len(similar) > 2:
|
||||
log.debug('complex merge, trying our best...')
|
||||
_merge_similar_guesses_nocheck(guesses, prop, choose)
|
||||
merge_similar_guesses(guesses, prop, choose)
|
||||
return
|
||||
|
||||
|
||||
def merge_append_guesses(guesses, prop):
|
||||
"""Take a list of guesses and merge those which have the same properties by
|
||||
appending them in a list.
|
||||
|
||||
DEPRECATED, remove with old guessers
|
||||
|
||||
"""
|
||||
|
||||
|
||||
similar = [ guess for guess in guesses if prop in guess ]
|
||||
if not similar:
|
||||
return
|
||||
|
||||
merged = similar[0]
|
||||
merged[prop] = [ merged[prop] ]
|
||||
# TODO: what to do with global confidence? mean of them all?
|
||||
|
||||
for m in similar[1:]:
|
||||
for prop2 in m:
|
||||
if prop == prop2:
|
||||
merged[prop].append(m[prop])
|
||||
else:
|
||||
if prop2 in m:
|
||||
log.warning('overwriting property "%s" with value ' % (prop2, m[prop2]))
|
||||
merged[prop2] = m[prop2]
|
||||
# TODO: confidence also
|
||||
|
||||
guesses.remove(m)
|
||||
|
||||
|
||||
def merge_all(guesses, append = []):
|
||||
"""Merges all the guesses in a single result, removes very unlikely values, and returns it.
|
||||
You can specify a list of properties that should be appended into a list instead of being
|
||||
merged.
|
||||
|
||||
>>> merge_all([ Guess({ 'season': 2 }, confidence = 0.6),
|
||||
... Guess({ 'episodeNumber': 13 }, confidence = 0.8) ])
|
||||
{'season': 2, 'episodeNumber': 13}
|
||||
|
||||
>>> merge_all([ Guess({ 'episodeNumber': 27 }, confidence = 0.02),
|
||||
... Guess({ 'season': 1 }, confidence = 0.2) ])
|
||||
{'season': 1}
|
||||
|
||||
"""
|
||||
if not guesses:
|
||||
return Guess()
|
||||
|
||||
result = guesses[0]
|
||||
|
||||
for g in guesses[1:]:
|
||||
# first append our appendable properties
|
||||
for prop in append:
|
||||
if prop in g:
|
||||
result.set(prop, result.get(prop, []) + [ g[prop] ],
|
||||
# TODO: what to do with confidence here? maybe an arithmetic mean...
|
||||
confidence = g.confidence(prop))
|
||||
|
||||
del g[prop]
|
||||
|
||||
# then merge the remaining ones
|
||||
if set(result) & set(g):
|
||||
log.warning('duplicate properties %s in merged result...' % (set(result) & set(g)))
|
||||
|
||||
result.update_highest_confidence(g)
|
||||
|
||||
# delete very unlikely values
|
||||
for p in result.keys():
|
||||
if result.confidence(p) < 0.05:
|
||||
del result[p]
|
||||
|
||||
# make sure our appendable properties contain unique values
|
||||
for prop in append:
|
||||
if prop in result:
|
||||
result[prop] = list(set(result[prop]))
|
||||
|
||||
return result
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import Guess
|
||||
import hashlib, os.path
|
||||
|
||||
def hash_file(filename):
|
||||
"""Returns the ed2k hash of a given file.
|
||||
|
||||
>>> hash_file('test/dummy.srt')
|
||||
'ed2k://|file|dummy.srt|44|1CA0B9DED3473B926AA93A0A546138BB|/'
|
||||
"""
|
||||
return 'ed2k://|file|%s|%d|%s|/' % (os.path.basename(filename),
|
||||
os.path.getsize(filename),
|
||||
hash_filehash(filename).upper())
|
||||
|
||||
def hash_filehash(filename):
|
||||
"""Returns the ed2k hash of a given file.
|
||||
|
||||
This function is taken from:
|
||||
http://www.radicand.org/blog/orz/2010/2/21/edonkey2000-hash-in-python/
|
||||
"""
|
||||
md4 = hashlib.new('md4').copy
|
||||
|
||||
def gen(f):
|
||||
while True:
|
||||
x = f.read(9728000)
|
||||
if x: yield x
|
||||
else: return
|
||||
|
||||
def md4_hash(data):
|
||||
m = md4()
|
||||
m.update(data)
|
||||
return m
|
||||
|
||||
with open(filename, 'rb') as f:
|
||||
a = gen(f)
|
||||
hashes = [md4_hash(data).digest() for data in a]
|
||||
if len(hashes) == 1:
|
||||
return hashes[0].encode("hex")
|
||||
else: return md4_hash(reduce(lambda a,d: a + d, hashes, "")).hexd
|
||||
@@ -0,0 +1,56 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import Guess
|
||||
import struct, os
|
||||
|
||||
def hash_file(filename):
|
||||
"""This function is taken from:
|
||||
http://trac.opensubtitles.org/projects/opensubtitles/wiki/HashSourceCodes
|
||||
and is licensed under the GPL."""
|
||||
|
||||
longlongformat = 'q' # long long
|
||||
bytesize = struct.calcsize(longlongformat)
|
||||
|
||||
f = open(filename, "rb")
|
||||
|
||||
filesize = os.path.getsize(filename)
|
||||
hash = filesize
|
||||
|
||||
if filesize < 65536 * 2:
|
||||
raise Exception, "SizeError: size is %d, should be > 132K..." % filesize
|
||||
|
||||
for x in range(65536/bytesize):
|
||||
buffer = f.read(bytesize)
|
||||
(l_value,)= struct.unpack(longlongformat, buffer)
|
||||
hash += l_value
|
||||
hash = hash & 0xFFFFFFFFFFFFFFFF #to remain as 64bit number
|
||||
|
||||
|
||||
f.seek(max(0,filesize-65536),0)
|
||||
for x in range(65536/bytesize):
|
||||
buffer = f.read(bytesize)
|
||||
(l_value,)= struct.unpack(longlongformat, buffer)
|
||||
hash += l_value
|
||||
hash = hash & 0xFFFFFFFFFFFFFFFF
|
||||
|
||||
f.close()
|
||||
returnedhash = "%016x" % hash
|
||||
return returnedhash
|
||||
@@ -0,0 +1,209 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import fileutils
|
||||
import os.path
|
||||
import re
|
||||
import logging
|
||||
|
||||
log = logging.getLogger('guessit.language')
|
||||
|
||||
|
||||
|
||||
# downloaded from http://www.loc.gov/standards/iso639-2/ISO-639-2_utf-8.txt
|
||||
#
|
||||
# Description of the fields:
|
||||
# "An alpha-3 (bibliographic) code, an alpha-3 (terminologic) code (when given),
|
||||
# an alpha-2 code (when given), an English name, and a French name of a language
|
||||
# are all separated by pipe (|) characters."
|
||||
language_matrix = [ l.strip().decode('utf-8').split('|') for l in open(fileutils.file_in_same_dir(__file__, 'ISO-639-2_utf-8.txt')) ]
|
||||
|
||||
lng3 = frozenset(filter(bool, (l[0] for l in language_matrix)))
|
||||
lng3term = frozenset(filter(bool, (l[1] for l in language_matrix)))
|
||||
lng2 = frozenset(filter(bool, (l[2] for l in language_matrix)))
|
||||
lng_en_name = frozenset(filter(bool, (lng for l in language_matrix for lng in l[3].lower().split('; '))))
|
||||
lng_fr_name = frozenset(filter(bool, (lng for l in language_matrix for lng in l[4].lower().split('; '))))
|
||||
lng_all_names = lng3 | lng3term | lng2 | lng_en_name | lng_fr_name
|
||||
|
||||
lng3_to_lng3term = dict((l[0], l[1]) for l in language_matrix if l[1])
|
||||
lng3term_to_lng3 = dict((l[1], l[0]) for l in language_matrix if l[1])
|
||||
|
||||
lng3_to_lng2 = dict((l[0], l[2]) for l in language_matrix if l[2])
|
||||
lng2_to_lng3 = dict((l[2], l[0]) for l in language_matrix if l[2])
|
||||
|
||||
# we only return the first given english name, hoping it is the most used one
|
||||
lng3_to_lng_en_name = dict((l[0], l[3].split('; ')[0]) for l in language_matrix if l[3])
|
||||
lng_en_name_to_lng3 = dict((en_name.lower(), l[0]) for l in language_matrix if l[3] for en_name in l[3].split('; '))
|
||||
|
||||
# we only return the first given french name, hoping it is the most used one
|
||||
lng3_to_lng_fr_name = dict((l[0], l[4].split('; ')[0]) for l in language_matrix if l[4])
|
||||
lng_fr_name_to_lng3 = dict((fr_name.lower(), l[0]) for l in language_matrix if l[4] for fr_name in l[4].split('; '))
|
||||
|
||||
|
||||
def is_language(language):
|
||||
return language.lower() in lng_all_names
|
||||
|
||||
class Language(object):
|
||||
"""This class represents a human language.
|
||||
|
||||
You can initialize it with pretty much everything, as it knows conversion from
|
||||
ISO-639 2-letter and 3-letter codes, English and French names.
|
||||
|
||||
>>> Language('fr')
|
||||
Language(French)
|
||||
|
||||
>>> Language('eng').french_name()
|
||||
u'anglais'
|
||||
"""
|
||||
def __init__(self, language):
|
||||
lang = None
|
||||
language = language.lower()
|
||||
if len(language) == 2:
|
||||
lang = lng2_to_lng3.get(language)
|
||||
elif len(language) == 3:
|
||||
lang = language if language in lng3 else lng3term_to_lng3.get(language)
|
||||
else:
|
||||
lang = lng_en_name_to_lng3.get(language) or lng_fr_name_to_lng3.get(language)
|
||||
|
||||
if lang is None:
|
||||
raise ValueError, 'The given string "%s" could not be identified as a language' % language
|
||||
|
||||
self.lang = lang
|
||||
|
||||
def lng2(self):
|
||||
return lng3_to_lng2[self.lang]
|
||||
|
||||
def lng3(self):
|
||||
return self.lang
|
||||
|
||||
def lng3term(self):
|
||||
return lng3_to_lng3term[self.lang]
|
||||
|
||||
def english_name(self):
|
||||
return lng3_to_lng_en_name[self.lang]
|
||||
|
||||
def french_name(self):
|
||||
return lng3_to_lng_fr_name[self.lang]
|
||||
|
||||
|
||||
def __hash__(self):
|
||||
return hash(self.lang)
|
||||
|
||||
def __eq__(self, other):
|
||||
if isinstance(other, Language):
|
||||
return self.lang == other.lang
|
||||
|
||||
if isinstance(other, basestring):
|
||||
try:
|
||||
return self == Language(other)
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
return False
|
||||
|
||||
def __ne__(self, other):
|
||||
return not self == other
|
||||
|
||||
def __unicode__(self):
|
||||
return lng3_to_lng_en_name[self.lang]
|
||||
|
||||
def __str__(self):
|
||||
return unicode(self).encode('utf-8')
|
||||
|
||||
def __repr__(self):
|
||||
return 'Language(%s)' % self
|
||||
|
||||
|
||||
|
||||
def search_language(string, lang_filter = None):
|
||||
"""Looks for language patterns, and if found return the language object,
|
||||
its group span and an associated confidence.
|
||||
|
||||
you can specify a list of allowed languages using the lang_filter argument,
|
||||
as in lang_filter = [ 'fr', 'eng', 'spanish' ]
|
||||
|
||||
Assumes there are sentinels at the beginning and end of the string that
|
||||
always allow matching a non-letter delimiting the language.
|
||||
|
||||
>>> search_language('movie [en].avi')
|
||||
(Language(English), (7, 9), 0.80000000000000004)
|
||||
|
||||
>>> search_language('the zen fat cat and the gay mad men got a new fan', lang_filter = ['en', 'fr', 'es'])
|
||||
(None, None, None)
|
||||
"""
|
||||
|
||||
# list of common words which could be interpreted as languages, but which
|
||||
# are far too common to be able to say they represent a language in the
|
||||
# middle of a string (where they most likely carry their commmon meaning)
|
||||
lng_common_words = frozenset([ # english words
|
||||
'is', 'it', 'am', 'mad', 'men', 'man', 'run', 'sin', 'st', 'to',
|
||||
'no', 'non', 'war', 'min', 'new', 'car', 'day', 'bad', 'bat', 'fan',
|
||||
'fry', 'cop', 'zen', 'gay', 'fat', 'cherokee', 'got', 'an', 'as',
|
||||
'cat', 'her', 'be', 'hat', 'sun', 'may', 'my', 'mr',
|
||||
# french words
|
||||
'bas', 'de', 'le', 'son', 'vo', 'vf', 'ne', 'ca', 'ce', 'et', 'que',
|
||||
'mal', 'est', 'vol', 'or', 'mon', 'se',
|
||||
# spanish words
|
||||
'la', 'el', 'del', 'por', 'mar',
|
||||
# other
|
||||
'ind', 'arw', 'ts', 'ii', 'bin', 'chan', 'ss', 'san'
|
||||
])
|
||||
sep = r'[](){} \._-+'
|
||||
|
||||
if lang_filter:
|
||||
lang_filter = set(Language(l) for l in lang_filter)
|
||||
|
||||
slow = string.lower()
|
||||
confidence = 1.0 # for all of them
|
||||
for lang in lng_all_names:
|
||||
|
||||
if lang in lng_common_words:
|
||||
continue
|
||||
|
||||
pos = slow.find(lang)
|
||||
|
||||
if pos != -1:
|
||||
end = pos + len(lang)
|
||||
# make sure our word is always surrounded by separators
|
||||
if slow[pos-1] not in sep or slow[end] not in sep:
|
||||
continue
|
||||
|
||||
language = Language(slow[pos:end])
|
||||
if lang_filter and language not in lang_filter:
|
||||
continue
|
||||
|
||||
# only allow those languages that have a 2-letter code, those who
|
||||
# don't are too esoteric and probably false matches
|
||||
if language.lang not in lng3_to_lng2:
|
||||
continue
|
||||
|
||||
# confidence depends on lng2, lng3, english name, ...
|
||||
if len(lang) == 2:
|
||||
confidence = 0.8
|
||||
elif len(lang) == 3:
|
||||
confidence = 0.9
|
||||
else:
|
||||
# Note: we could either be really confident that we found a language
|
||||
# or assume that full language names are too common words
|
||||
confidence = 0.3 # going with the low-confidence route here
|
||||
|
||||
return language, (pos, end), confidence
|
||||
|
||||
return None, None, None
|
||||
@@ -0,0 +1,621 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import fileutils, textutils
|
||||
from guessit.guess import Guess, merge_similar_guesses, merge_all, choose_int, choose_string
|
||||
from guessit.date import search_date, search_year
|
||||
from guessit.language import search_language
|
||||
from guessit.patterns import video_exts, subtitle_exts, sep, deleted, video_rexps, websites, episode_rexps, weak_episode_rexps, non_episode_title, properties, canonical_form
|
||||
from guessit.matchtree import get_group, find_group, leftover_valid_groups, tree_to_string
|
||||
from guessit.textutils import find_first_level_groups, split_on_groups, blank_region, clean_string, to_utf8
|
||||
from guessit.fileutils import split_path_components
|
||||
import datetime
|
||||
import os.path
|
||||
import re
|
||||
import copy
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.matcher")
|
||||
|
||||
|
||||
|
||||
def split_explicit_groups(string):
|
||||
"""return the string split into explicit groups, that is, those either
|
||||
between parenthese, square brackets or curly braces, and those separated
|
||||
by a dash."""
|
||||
result = find_first_level_groups(string, '()')
|
||||
result = reduce(lambda l, x: l + find_first_level_groups(x, '[]'), result, [])
|
||||
result = reduce(lambda l, x: l + find_first_level_groups(x, '{}'), result, [])
|
||||
# do not do this at this moment, it is not strong enough and can break other
|
||||
# patterns, such as dates, etc...
|
||||
#result = reduce(lambda l, x: l + x.split('-'), result, [])
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def format_guess(guess):
|
||||
"""Format all the found values to their natural type.
|
||||
For instance, a year would be stored as an int value, etc...
|
||||
|
||||
Note that this modifies the dictionary given as input.
|
||||
"""
|
||||
for prop, value in guess.items():
|
||||
if prop in ('season', 'episodeNumber', 'year', 'cdNumber', 'cdNumberTotal'):
|
||||
guess[prop] = int(guess[prop])
|
||||
elif isinstance(value, basestring):
|
||||
if prop in ('edition',):
|
||||
value = clean_string(value)
|
||||
guess[prop] = canonical_form(value)
|
||||
|
||||
return guess
|
||||
|
||||
|
||||
def guess_groups(string, result, filetype):
|
||||
# add sentinels so we can match a separator char at either end of
|
||||
# our groups, even when they are at the beginning or end of the string
|
||||
# we will adjust the span accordingly later
|
||||
#
|
||||
# filetype can either be movie, moviesubtitle, episode, episodesubtitle
|
||||
current = ' ' + string + ' '
|
||||
|
||||
regions = [] # list of (start, end) of matched regions
|
||||
|
||||
def guessed(match_dict, confidence):
|
||||
guess = format_guess(Guess(match_dict, confidence = confidence))
|
||||
result.append(guess)
|
||||
log.debug('Found with confidence %.2f: %s' % (confidence, guess))
|
||||
return guess
|
||||
|
||||
def update_found(string, guess, span, span_adjust = (0,0)):
|
||||
span = (span[0] + span_adjust[0],
|
||||
span[1] + span_adjust[1])
|
||||
regions.append((span, guess))
|
||||
return blank_region(string, span)
|
||||
|
||||
# try to find dates first, as they are very specific
|
||||
date, span = search_date(current)
|
||||
if date:
|
||||
guess = guessed({ 'date': date }, confidence = 1.0)
|
||||
current = update_found(current, guess, span)
|
||||
|
||||
# for non episodes only, look for year information
|
||||
if filetype not in ('episode', 'episodesubtitle'):
|
||||
year, span = search_year(current)
|
||||
if year:
|
||||
guess = guessed({ 'year': year }, confidence = 1.0)
|
||||
current = update_found(current, guess, span)
|
||||
|
||||
# specific regexps (ie: cd number, season X episode, ...)
|
||||
for rexp, confidence, span_adjust in video_rexps:
|
||||
match = re.search(rexp, current, re.IGNORECASE)
|
||||
if match:
|
||||
metadata = match.groupdict()
|
||||
# is this the better place to put it? (maybe, as it is at least the soonest that we can catch it)
|
||||
if 'cdNumberTotal' in metadata and metadata['cdNumberTotal'] is None:
|
||||
del metadata['cdNumberTotal']
|
||||
|
||||
guess = guessed(metadata, confidence = confidence)
|
||||
current = update_found(current, guess, match.span(), span_adjust)
|
||||
|
||||
if filetype in ('episode', 'episodesubtitle'):
|
||||
for rexp, confidence, span_adjust in episode_rexps:
|
||||
match = re.search(rexp, current, re.IGNORECASE)
|
||||
if match:
|
||||
metadata = match.groupdict()
|
||||
guess = guessed(metadata, confidence = confidence)
|
||||
current = update_found(current, guess, match.span(), span_adjust)
|
||||
|
||||
|
||||
# Now websites, but as exact string instead of regexps
|
||||
clow = current.lower()
|
||||
for site in websites:
|
||||
pos = clow.find(site.lower())
|
||||
if pos != -1:
|
||||
guess = guessed({ 'website': site }, confidence = confidence)
|
||||
current = update_found(current, guess, (pos, pos+len(site)))
|
||||
clow = current.lower()
|
||||
|
||||
|
||||
# release groups have certain constraints, cannot be included in the previous general regexps
|
||||
group_names = [ r'\.(Xvid)-(?P<releaseGroup>.*?)[ \.]',
|
||||
r'\.(DivX)-(?P<releaseGroup>.*?)[\. ]',
|
||||
r'\.(DVDivX)-(?P<releaseGroup>.*?)[\. ]',
|
||||
]
|
||||
for rexp in group_names:
|
||||
match = re.search(rexp, current, re.IGNORECASE)
|
||||
if match:
|
||||
metadata = match.groupdict()
|
||||
metadata.update({ 'videoCodec': match.group(1) })
|
||||
guess = guessed(metadata, confidence = 0.8)
|
||||
current = update_found(current, guess, match.span(), span_adjust = (1, -1))
|
||||
|
||||
|
||||
# common well-defined words and regexps
|
||||
clow = current.lower()
|
||||
confidence = 1.0 # for all of them
|
||||
for prop, values in properties.items():
|
||||
for value in values:
|
||||
pos = clow.find(value.lower())
|
||||
if pos != -1:
|
||||
end = pos + len(value)
|
||||
# make sure our word is always surrounded by separators
|
||||
if clow[pos-1] not in sep or clow[end] not in sep:
|
||||
# note: sep is a regexp, but in this case using it as
|
||||
# a sequence achieves the same goal
|
||||
continue
|
||||
|
||||
guess = guessed({ prop: value }, confidence = confidence)
|
||||
current = update_found(current, guess, (pos, end))
|
||||
clow = current.lower()
|
||||
|
||||
# weak guesses for episode number, only run it if we don't have an estimate already
|
||||
if filetype in ('episode', 'episodesubtitle'):
|
||||
if not any('episodeNumber' in match for match in result):
|
||||
for rexp, _, span_adjust in weak_episode_rexps:
|
||||
match = re.search(rexp, current, re.IGNORECASE)
|
||||
if match:
|
||||
metadata = match.groupdict()
|
||||
epnum = int(metadata['episodeNumber'])
|
||||
if epnum > 100:
|
||||
guess = guessed({ 'season': epnum // 100,
|
||||
'episodeNumber': epnum % 100 }, confidence = 0.6)
|
||||
else:
|
||||
guess = guessed(metadata, confidence = 0.3)
|
||||
current = update_found(current, guess, match.span(), span_adjust)
|
||||
|
||||
# try to find languages now
|
||||
language, span, confidence = search_language(current)
|
||||
while language:
|
||||
# is it a subtitle language?
|
||||
if 'sub' in clean_string(current[:span[0]]).lower().split(' '):
|
||||
guess = guessed({ 'subtitleLanguage': language }, confidence = confidence)
|
||||
else:
|
||||
guess = guessed({ 'language': language }, confidence = confidence)
|
||||
current = update_found(current, guess, span)
|
||||
|
||||
language, span, confidence = search_language(current)
|
||||
|
||||
|
||||
# remove our sentinels now and ajust spans accordingly
|
||||
assert(current[0] == ' ' and current[-1] == ' ')
|
||||
current = current[1:-1]
|
||||
regions = [ ((start-1, end-1), guess) for (start, end), guess in regions ]
|
||||
|
||||
# split into '-' separated subgroups (with required separator chars
|
||||
# around the dash)
|
||||
didx = current.find('-')
|
||||
while didx > 0:
|
||||
regions.append(((didx, didx), None))
|
||||
didx = current.find('-', didx+1)
|
||||
|
||||
# cut our final groups, and rematch the guesses to the group that created
|
||||
# id, None if it is a leftover group
|
||||
region_spans = [ span for span, guess in regions ]
|
||||
string_groups = split_on_groups(string, region_spans)
|
||||
remaining_groups = split_on_groups(current, region_spans)
|
||||
guesses = []
|
||||
|
||||
pos = 0
|
||||
for group in string_groups:
|
||||
found = False
|
||||
for span, guess in regions:
|
||||
if span[0] == pos:
|
||||
guesses.append(guess)
|
||||
found = True
|
||||
if not found:
|
||||
guesses.append(None)
|
||||
|
||||
pos += len(group)
|
||||
|
||||
return zip(string_groups,
|
||||
remaining_groups,
|
||||
guesses)
|
||||
|
||||
|
||||
def match_from_epnum_position(match_tree, epnum_pos, guessed, update_found):
|
||||
"""guessed is a callback function to call with the guessed group
|
||||
update_found is a callback to update the match group and returns leftover groups."""
|
||||
pidx, eidx, gidx = epnum_pos
|
||||
|
||||
# a few helper functions to be able to filter using high-level semantics
|
||||
def same_pgroup_before(group):
|
||||
_, (ppidx, eeidx, ggidx) = group
|
||||
return ppidx == pidx and (eeidx, ggidx) < (eidx, gidx)
|
||||
|
||||
def same_pgroup_after(group):
|
||||
_, (ppidx, eeidx, ggidx) = group
|
||||
return ppidx == pidx and (eeidx, ggidx) > (eidx, gidx)
|
||||
|
||||
def same_egroup_before(group):
|
||||
_, (ppidx, eeidx, ggidx) = group
|
||||
return ppidx == pidx and eeidx == eidx and ggidx < gidx
|
||||
|
||||
def same_egroup_after(group):
|
||||
_, (ppidx, eeidx, ggidx) = group
|
||||
return ppidx == pidx and eeidx == eidx and ggidx > gidx
|
||||
|
||||
leftover = leftover_valid_groups(match_tree)
|
||||
|
||||
# if we have at least 1 valid group before the episodeNumber, then it's probably
|
||||
# the series name
|
||||
series_candidates = filter(same_pgroup_before, leftover)
|
||||
if len(series_candidates) >= 1:
|
||||
guess = guessed({ 'series': series_candidates[0][0] }, confidence = 0.7)
|
||||
leftover = update_found(leftover, series_candidates[0][1], guess)
|
||||
|
||||
# only 1 group after (in the same path group) and it's probably the episode title
|
||||
title_candidates = filter(lambda g:g[0].lower() not in non_episode_title,
|
||||
filter(same_pgroup_after, leftover))
|
||||
if len(title_candidates) == 1:
|
||||
guess = guessed({ 'title': title_candidates[0][0] }, confidence = 0.5)
|
||||
leftover = update_found(leftover, title_candidates[0][1], guess)
|
||||
else:
|
||||
# try in the same explicit group, with lower confidence
|
||||
title_candidates = filter(lambda g:g[0].lower() not in non_episode_title,
|
||||
filter(same_egroup_after, leftover))
|
||||
if len(title_candidates) == 1:
|
||||
guess = guessed({ 'title': title_candidates[0][0] }, confidence = 0.4)
|
||||
leftover = update_found(leftover, title_candidates[0][1], guess)
|
||||
|
||||
# epnumber is the first group and there are only 2 after it in same path group
|
||||
# -> season title - episode title
|
||||
already_has_title = (find_group(match_tree, 'title') != [])
|
||||
|
||||
title_candidates = filter(lambda g:g[0].lower() not in non_episode_title,
|
||||
filter(same_pgroup_after, leftover))
|
||||
if (not already_has_title and # no title
|
||||
not filter(same_pgroup_before, leftover) and # no groups before
|
||||
len(title_candidates) == 2): # only 2 groups after
|
||||
|
||||
guess = guessed({ 'series': title_candidates[0][0] }, confidence = 0.4)
|
||||
leftover = update_found(leftover, title_candidates[0][1], guess)
|
||||
guess = guessed({ 'title': title_candidates[1][0] }, confidence = 0.4)
|
||||
leftover = update_found(leftover, title_candidates[1][1], guess)
|
||||
|
||||
|
||||
# if we only have 1 remaining valid group in the pathpart before the filename,
|
||||
# then it's likely that it is the series name
|
||||
series_candidates = [ group for group in leftover if group[1][0] == pidx-1 ]
|
||||
if len(series_candidates) == 1:
|
||||
guess = guessed({ 'series': series_candidates[0][0] }, confidence = 0.5)
|
||||
leftover = update_found(leftover, series_candidates[0][1], guess)
|
||||
|
||||
return match_tree
|
||||
|
||||
|
||||
|
||||
class IterativeMatcher(object):
|
||||
def __init__(self, filename, filetype = 'autodetect'):
|
||||
"""An iterative matcher tries to match different patterns that appear
|
||||
in the filename.
|
||||
|
||||
The 'filetype' argument indicates which type of file you want to match.
|
||||
If it is 'autodetect', the matcher will try to see whether it can guess
|
||||
that the file corresponds to an episode, or otherwise will assume it is
|
||||
a movie.
|
||||
|
||||
The recognized 'filetype' values are:
|
||||
[ autodetect, subtitle, movie, moviesubtitle, episode, episodesubtitle ]
|
||||
|
||||
|
||||
The IterativeMatcher works mainly in 2 steps:
|
||||
|
||||
First, it splits the filename into a match_tree, which is a tree of groups
|
||||
which have a semantic meaning, such as episode number, movie title,
|
||||
etc...
|
||||
|
||||
The match_tree created looks like the following:
|
||||
|
||||
0000000000000000000000000000000000000000000000000000000000000000000000000000000000 111
|
||||
0000011111111111112222222222222233333333444444444444444455555555666777777778888888 000
|
||||
0000000000000000000000000000000001111112011112222333333401123334000011233340000000 000
|
||||
__________________(The.Prestige).______.[____.HP.______.{__-___}.St{__-___}.Chaps].___
|
||||
xxxxxttttttttttttt ffffff vvvv xxxxxx ll lll xx xxx ccc
|
||||
[XCT].Le.Prestige.(The.Prestige).DVDRip.[x264.HP.He-Aac.{Fr-Eng}.St{Fr-Eng}.Chaps].mkv
|
||||
|
||||
The first 3 lines indicates the group index in which a char in the
|
||||
filename is located. So for instance, x264 is the group (0, 4, 1), and
|
||||
it corresponds to a video codec, denoted by the letter'v' in the 4th line.
|
||||
(for more info, see guess.matchtree.tree_to_string)
|
||||
|
||||
|
||||
Second, it tries to merge all this information into a single object
|
||||
containing all the found properties, and does some (basic) conflict
|
||||
resolution when they arise.
|
||||
"""
|
||||
|
||||
if filetype not in ('autodetect', 'subtitle', 'movie', 'moviesubtitle',
|
||||
'episode', 'episodesubtitle'):
|
||||
raise ValueError, "filetype needs to be one of ('autodetect', 'subtitle', 'movie', 'moviesubtitle', 'episode', 'episodesubtitle')"
|
||||
if not isinstance(filename, unicode):
|
||||
log.debug('WARNING: given filename to matcher is not unicode...')
|
||||
|
||||
match_tree = []
|
||||
result = [] # list of found metadata
|
||||
|
||||
def guessed(match_dict, confidence):
|
||||
guess = format_guess(Guess(match_dict, confidence = confidence))
|
||||
result.append(guess)
|
||||
log.debug('Found with confidence %.2f: %s' % (confidence, guess))
|
||||
return guess
|
||||
|
||||
def update_found(leftover, group_pos, guess):
|
||||
pidx, eidx, gidx = group_pos
|
||||
group = match_tree[pidx][eidx][gidx]
|
||||
match_tree[pidx][eidx][gidx] = (group[0],
|
||||
deleted * len(group[0]),
|
||||
guess)
|
||||
return [ g for g in leftover if g[1] != group_pos ]
|
||||
|
||||
|
||||
# 1- first split our path into dirs + basename + ext
|
||||
match_tree = split_path_components(filename)
|
||||
|
||||
fileext = match_tree.pop(-1)[1:].lower()
|
||||
if fileext in subtitle_exts:
|
||||
if 'movie' in filetype:
|
||||
filetype = 'moviesubtitle'
|
||||
elif 'episode' in filetype:
|
||||
filetype = 'episodesubtitle'
|
||||
else:
|
||||
filetype = 'subtitle'
|
||||
extguess = guessed({ 'container': fileext }, confidence = 1.0)
|
||||
elif fileext in video_exts:
|
||||
extguess = guessed({ 'container': fileext }, confidence = 1.0)
|
||||
else:
|
||||
extguess = guessed({ 'extension': fileext}, confidence = 1.0)
|
||||
|
||||
# TODO: depending on the extension, we could already grab some info and maybe specialized
|
||||
# guessers, eg: a lang parser for idx files, an automatic detection of the language
|
||||
# for srt files, a video metadata extractor for avi, mkv, ...
|
||||
|
||||
# if we are on autodetect, try to do it now so we can tell the
|
||||
# guess_groups function what type of info it should be looking for
|
||||
if filetype in ('autodetect', 'subtitle'):
|
||||
for rexp, confidence, span_adjust in episode_rexps:
|
||||
match = re.search(rexp, filename, re.IGNORECASE)
|
||||
if match:
|
||||
if filetype == 'autodetect':
|
||||
filetype = 'episode'
|
||||
elif filetype == 'subtitle':
|
||||
filetype = 'episodesubtitle'
|
||||
break
|
||||
|
||||
# if no episode info found, assume it's a movie
|
||||
if filetype == 'autodetect':
|
||||
filetype = 'movie'
|
||||
elif filetype == 'subtitle':
|
||||
filetype = 'moviesubtitle'
|
||||
|
||||
guessed({ 'type': filetype }, confidence = 1.0)
|
||||
|
||||
|
||||
# 2- split each of those into explicit groups, if any
|
||||
# note: be careful, as this might split some regexps with more confidence such as
|
||||
# Alfleni-Team, or [XCT] or split a date such as (14-01-2008)
|
||||
match_tree = [ split_explicit_groups(part) for part in match_tree ]
|
||||
|
||||
|
||||
# 3- try to match information in decreasing order of confidence and
|
||||
# blank the matching group in the string if we found something
|
||||
for pathpart in match_tree:
|
||||
for gidx, explicit_group in enumerate(pathpart):
|
||||
pathpart[gidx] = guess_groups(explicit_group, result, filetype = filetype)
|
||||
|
||||
# 4- try to identify the remaining unknown groups by looking at their position
|
||||
# relative to other known elements
|
||||
|
||||
if filetype in ('episode', 'episodesubtitle'):
|
||||
eps = find_group(match_tree, 'episodeNumber')
|
||||
if eps:
|
||||
match_tree = match_from_epnum_position(match_tree, eps[0], guessed, update_found)
|
||||
|
||||
leftover = leftover_valid_groups(match_tree)
|
||||
|
||||
if not eps:
|
||||
# if we don't have the episode number, but at least 2 groups in the
|
||||
# last path group, then it's probably series - eptitle
|
||||
title_candidates = filter(lambda g:g[0].lower() not in non_episode_title,
|
||||
filter(lambda g: g[1][0] == len(match_tree)-1,
|
||||
leftover_valid_groups(match_tree)))
|
||||
if len(title_candidates) >= 2:
|
||||
guess = guessed({ 'series': title_candidates[0][0] }, confidence = 0.4)
|
||||
leftover = update_found(leftover, title_candidates[0][1], guess)
|
||||
guess = guessed({ 'title': title_candidates[1][0] }, confidence = 0.4)
|
||||
leftover = update_found(leftover, title_candidates[1][1], guess)
|
||||
|
||||
|
||||
# if there's a path group that only contains the season info, then the previous one
|
||||
# is most likely the series title (ie: .../series/season X/...)
|
||||
eps = [ gpos for gpos in find_group(match_tree, 'season')
|
||||
if 'episodeNumber' not in get_group(match_tree, gpos)[2] ]
|
||||
|
||||
if eps:
|
||||
pidx, eidx, gidx = eps[0]
|
||||
previous = [ group for group in leftover if group[1][0] == pidx - 1 ]
|
||||
if len(previous) == 1:
|
||||
guess = guessed({ 'series': previous[0][0] }, confidence = 0.5)
|
||||
leftover = update_found(leftover, previous[0][1], guess)
|
||||
|
||||
|
||||
elif filetype in ('movie', 'moviesubtitle'):
|
||||
leftover_all = leftover_valid_groups(match_tree)
|
||||
|
||||
# specific cases:
|
||||
# - movies/tttttt (yyyy)/tttttt.ccc
|
||||
try:
|
||||
if match_tree[-3][0][0][0].lower() == 'movies':
|
||||
# Note:too generic, might solve all the unittests as they all contain 'movies'
|
||||
# in their path
|
||||
#
|
||||
#if len(match_tree[-2][0]) == 1:
|
||||
# title = match_tree[-2][0][0]
|
||||
# guess = guessed({ 'title': clean_string(title[0]) }, confidence = 0.7)
|
||||
# update_found(leftover_all, title, guess)
|
||||
|
||||
year_group = filter(lambda gpos: gpos[0] == len(match_tree)-2,
|
||||
find_group(match_tree, 'year'))[0]
|
||||
leftover = leftover_valid_groups(match_tree,
|
||||
valid = lambda g: ((g[0] and g[0][0] not in sep) and
|
||||
g[1][0] == len(match_tree) - 2))
|
||||
if len(match_tree[-2]) == 2 and year_group[1] == 1:
|
||||
title = leftover[0]
|
||||
guess = guessed({ 'title': clean_string(title[0]) },
|
||||
confidence = 0.8)
|
||||
update_found(leftover_all, title[1], guess)
|
||||
raise Exception # to exit the try catch now
|
||||
|
||||
leftover = [ g for g in leftover_all if (g[1][0] == year_group[0] and
|
||||
g[1][1] < year_group[1] and
|
||||
g[1][2] < year_group[2]) ]
|
||||
leftover = sorted(leftover, key = lambda x:x[1])
|
||||
title = leftover[0]
|
||||
guess = guessed({ 'title': title[0] }, confidence = 0.8)
|
||||
leftover = update_found(leftover, title[1], guess)
|
||||
except:
|
||||
pass
|
||||
|
||||
# if we have either format or videoCodec in the folder containing the file
|
||||
# or one of its parents, then we should probably look for the title in
|
||||
# there rather than in the basename
|
||||
props = filter(lambda g: g[0] <= len(match_tree) - 2,
|
||||
find_group(match_tree, 'videoCodec') +
|
||||
find_group(match_tree, 'format') +
|
||||
find_group(match_tree, 'language'))
|
||||
leftover = None
|
||||
if props and all(g[0] == props[0][0] for g in props):
|
||||
leftover = [ g for g in leftover_all if g[1][0] == props[0][0] ]
|
||||
|
||||
if props and leftover:
|
||||
guess = guessed({ 'title': leftover[0][0] }, confidence = 0.7)
|
||||
leftover = update_found(leftover, leftover[0][1], guess)
|
||||
|
||||
else:
|
||||
# first leftover group in the last path part sounds like a good candidate for title,
|
||||
# except if it's only one word and that the first group before has at least 3 words in it
|
||||
# (case where the filename contains an 8 chars short name and the movie title is
|
||||
# actually in the parent directory name)
|
||||
leftover = [ g for g in leftover_all if g[1][0] == len(match_tree)-1 ]
|
||||
if leftover:
|
||||
title, (pidx, eidx, gidx) = leftover[0]
|
||||
previous_pgroup_leftover = filter(lambda g: g[1][0] == pidx-1, leftover_all)
|
||||
|
||||
if (title.count(' ') == 0 and
|
||||
previous_pgroup_leftover and
|
||||
previous_pgroup_leftover[0][0].count(' ') >= 2):
|
||||
|
||||
guess = guessed({ 'title': previous_pgroup_leftover[0][0] }, confidence = 0.6)
|
||||
leftover = update_found(leftover, previous_pgroup_leftover[0][1], guess)
|
||||
|
||||
else:
|
||||
guess = guessed({ 'title': title }, confidence = 0.6)
|
||||
leftover = update_found(leftover, leftover[0][1], guess)
|
||||
else:
|
||||
# if there were no leftover groups in the last path part, look in the one before that
|
||||
previous_pgroup_leftover = filter(lambda g: g[1][0] == len(match_tree)-2, leftover_all)
|
||||
if previous_pgroup_leftover:
|
||||
guess = guessed({ 'title': previous_pgroup_leftover[0][0] }, confidence = 0.6)
|
||||
leftover = update_found(leftover, previous_pgroup_leftover[0][1], guess)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
# 5- perform some post-processing steps
|
||||
|
||||
# 5.1- try to promote language to subtitle language where it makes sense
|
||||
for pidx, eidx, gidx in find_group(match_tree, 'language'):
|
||||
string, remaining, guess = get_group(match_tree, (pidx, eidx, gidx))
|
||||
|
||||
def promote_subtitle():
|
||||
guess.set('subtitleLanguage', guess['language'], confidence = guess.confidence('language'))
|
||||
del guess['language']
|
||||
|
||||
# - if we matched a language in a file with a sub extension and that the group
|
||||
# is the last group of the filename, it is probably the language of the subtitle
|
||||
# (eg: 'xxx.english.srt')
|
||||
if (fileext in subtitle_exts and
|
||||
pidx == len(match_tree) - 1 and
|
||||
eidx == len(match_tree[pidx]) - 1):
|
||||
promote_subtitle()
|
||||
|
||||
# - if a language is in an explicit group just preceded by "st", it is a subtitle
|
||||
# language (eg: '...st[fr-eng]...')
|
||||
if eidx > 0:
|
||||
previous = get_group(match_tree, (pidx, eidx-1, -1))
|
||||
if previous[0][-2:].lower() == 'st':
|
||||
promote_subtitle()
|
||||
|
||||
|
||||
|
||||
# re-append the extension now
|
||||
match_tree.append([[(fileext, deleted*len(fileext), extguess)]])
|
||||
|
||||
self.parts = result
|
||||
self.match_tree = match_tree
|
||||
|
||||
if filename.startswith('/'):
|
||||
filename = ' ' + filename
|
||||
|
||||
log.debug('Found match tree:\n%s\n%s' % (to_utf8(tree_to_string(match_tree)),
|
||||
to_utf8(filename)))
|
||||
|
||||
|
||||
def matched(self):
|
||||
# we need to make a copy here, as the merge functions work in place and
|
||||
# calling them on the match tree would modify it
|
||||
parts = copy.deepcopy(self.parts)
|
||||
|
||||
# 1- start by doing some common preprocessing tasks
|
||||
|
||||
# 1.1- ", the" at the end of a series title should be prepended to it
|
||||
for part in parts:
|
||||
if 'series' not in part:
|
||||
continue
|
||||
|
||||
series = part['series']
|
||||
lseries = series.lower()
|
||||
|
||||
if lseries[-4:] == ',the':
|
||||
part['series'] = 'The ' + series[:-4]
|
||||
|
||||
if lseries[-5:] == ', the':
|
||||
part['series'] = 'The ' + series[:-5]
|
||||
|
||||
|
||||
# 2- try to merge similar information together and give it a higher confidence
|
||||
for int_part in ('year', 'season', 'episodeNumber'):
|
||||
merge_similar_guesses(parts, int_part, choose_int)
|
||||
|
||||
for string_part in ('title', 'series', 'container', 'format', 'releaseGroup', 'website',
|
||||
'audioCodec', 'videoCodec', 'screenSize', 'episodeFormat'):
|
||||
merge_similar_guesses(parts, string_part, choose_string)
|
||||
|
||||
result = merge_all(parts, append = ['language', 'subtitleLanguage', 'other'])
|
||||
|
||||
# 3- some last minute post-processing
|
||||
if (result['type'] == 'episode' and
|
||||
'season' not in result and
|
||||
result.get('episodeFormat', '') == 'Minisode'):
|
||||
result['season'] = 0
|
||||
|
||||
log.debug('Final result: ' + result.nice_string())
|
||||
return result
|
||||
@@ -0,0 +1,153 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.patterns import deleted
|
||||
from guessit.textutils import clean_string
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.matchtree")
|
||||
|
||||
|
||||
|
||||
def tree_to_string(tree):
|
||||
"""Return a string representation for the given tree.
|
||||
|
||||
The lines convey the following information:
|
||||
- line 1: path idx
|
||||
- line 2: explicit group idx
|
||||
- line 3: group index
|
||||
- line 4: remaining info
|
||||
- line 5: meaning conveyed
|
||||
|
||||
Meaning is a letter indicating what type of info was matched by this group,
|
||||
for instance 't' = title, 'f' = format, 'l' = language, etc...
|
||||
|
||||
An example is the following:
|
||||
|
||||
0000000000000000000000000000000000000000000000000000000000000000000000000000000000 111
|
||||
0000011111111111112222222222222233333333444444444444444455555555666777777778888888 000
|
||||
0000000000000000000000000000000001111112011112222333333401123334000011233340000000 000
|
||||
__________________(The.Prestige).______.[____.HP.______.{__-___}.St{__-___}.Chaps].___
|
||||
xxxxxttttttttttttt ffffff vvvv xxxxxx ll lll xx xxx ccc
|
||||
[XCT].Le.Prestige.(The.Prestige).DVDRip.[x264.HP.He-Aac.{Fr-Eng}.St{Fr-Eng}.Chaps].mkv
|
||||
|
||||
(note: the last line representing the filename is not pat of the tree representation)
|
||||
"""
|
||||
m_tree = [ '', # path level index
|
||||
'', # explicit group index
|
||||
'', # matched regexp and dash-separated
|
||||
'', # groups leftover that couldn't be matched
|
||||
'', # meaning conveyed: E = episodenumber, S = season, ...
|
||||
]
|
||||
|
||||
def add_char(pidx, eidx, gidx, remaining, meaning = None):
|
||||
nr = len(remaining)
|
||||
def to_hex(x):
|
||||
if isinstance(x, int):
|
||||
return str(x) if x < 10 else chr(55+x)
|
||||
return x
|
||||
m_tree[0] = m_tree[0] + to_hex(pidx) * nr
|
||||
m_tree[1] = m_tree[1] + to_hex(eidx) * nr
|
||||
m_tree[2] = m_tree[2] + to_hex(gidx) * nr
|
||||
m_tree[3] = m_tree[3] + remaining
|
||||
m_tree[4] = m_tree[4] + str(meaning or ' ') * nr
|
||||
|
||||
def meaning(result):
|
||||
mmap = { 'episodeNumber': 'E',
|
||||
'season': 'S',
|
||||
'extension': 'e',
|
||||
'format': 'f',
|
||||
'language': 'l',
|
||||
'videoCodec': 'v',
|
||||
'audioCodec': 'a',
|
||||
'website': 'w',
|
||||
'container': 'c',
|
||||
'series': 'T',
|
||||
'title': 't',
|
||||
'date': 'd',
|
||||
'year': 'y',
|
||||
'releaseGroup': 'r',
|
||||
'screenSize': 's'
|
||||
}
|
||||
|
||||
if result is None:
|
||||
return ' '
|
||||
|
||||
for prop, l in mmap.items():
|
||||
if prop in result:
|
||||
return l
|
||||
|
||||
return 'x'
|
||||
|
||||
for pidx, pathpart in enumerate(tree):
|
||||
for eidx, explicit_group in enumerate(pathpart):
|
||||
for gidx, (group, remaining, result) in enumerate(explicit_group):
|
||||
add_char(pidx, eidx, gidx, remaining, meaning(result))
|
||||
|
||||
# special conditions for the path separator
|
||||
if pidx < len(tree) - 2:
|
||||
add_char(' ', ' ', ' ', '/')
|
||||
elif pidx == len(tree) - 2:
|
||||
add_char(' ', ' ', ' ', '.')
|
||||
|
||||
return '\n'.join(m_tree)
|
||||
|
||||
|
||||
|
||||
def iterate_groups(match_tree):
|
||||
"""Iterate over all the groups in a match_tree and return them as pairs
|
||||
of (group_pos, group) where:
|
||||
- group_pos = (pidx, eidx, gidx)
|
||||
- group = (string, remaining, guess)
|
||||
"""
|
||||
for pidx, pathpart in enumerate(match_tree):
|
||||
for eidx, explicit_group in enumerate(pathpart):
|
||||
for gidx, group in enumerate(explicit_group):
|
||||
yield (pidx, eidx, gidx), group
|
||||
|
||||
|
||||
def find_group(match_tree, prop):
|
||||
"""Find the list of groups that resulted in a guess that contains the
|
||||
asked property."""
|
||||
result = []
|
||||
for gpos, (string, remaining, guess) in iterate_groups(match_tree):
|
||||
if guess and prop in guess:
|
||||
result.append(gpos)
|
||||
return result
|
||||
|
||||
def get_group(match_tree, gpos):
|
||||
pidx, eidx, gidx = gpos
|
||||
return match_tree[pidx][eidx][gidx]
|
||||
|
||||
|
||||
def leftover_valid_groups(match_tree, valid = lambda s: len(s[0]) > 3):
|
||||
"""Return the list of valid string groups (eg: len(s) > 3) that could not be
|
||||
matched to anything as a list of pairs (cleaned_str, group_pos)."""
|
||||
leftover = []
|
||||
for gpos, (group, remaining, guess) in iterate_groups(match_tree):
|
||||
if not guess:
|
||||
clean_str = clean_string(remaining)
|
||||
if valid((clean_str, gpos)):
|
||||
leftover.append((clean_str, gpos))
|
||||
|
||||
return leftover
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
|
||||
subtitle_exts = [ 'srt', 'idx', 'sub' ]
|
||||
|
||||
video_exts = [ 'avi', 'mkv', 'mpg', 'mp4', 'mov', 'ogg', 'ogm', 'ogv', 'wmv' ]
|
||||
|
||||
# separator character regexp
|
||||
sep = r'[][)(}{+ \._-]' # regexp art, hehe :D
|
||||
|
||||
# character used to represent a deleted char (when matching groups)
|
||||
deleted = '_'
|
||||
|
||||
# format: [ (regexp, confidence, span_adjust) ]
|
||||
episode_rexps = [ # ... Season 2 ...
|
||||
(r'season (?P<season>[0-9]+)', 1.0, (0, 0)),
|
||||
(r'saison (?P<season>[0-9]+)', 1.0, (0, 0)),
|
||||
|
||||
# ... s02e13 ...
|
||||
(r'[Ss](?P<season>[0-9]{1,2}).{,3}[EeXx](?P<episodeNumber>[0-9]{1,2})[^0-9]', 1.0, (0, -1)),
|
||||
|
||||
# ... 2x13 ...
|
||||
(r'[^0-9](?P<season>[0-9]{1,2})[x\.](?P<episodeNumber>[0-9]{2})[^0-9]', 0.8, (1, -1)),
|
||||
|
||||
# ... s02 ...
|
||||
(sep + r's(?P<season>[0-9]{1,2})' + sep + '?', 0.6, (0, 0)),
|
||||
|
||||
# v2 or v3 for some mangas which have multiples rips
|
||||
(sep + r'(?P<episodeNumber>[0-9]{1,3})v[23]' + sep, 0.6, (0, 0)),
|
||||
]
|
||||
|
||||
|
||||
weak_episode_rexps = [ # ... 213 or 0106 ...
|
||||
(sep + r'(?P<episodeNumber>[0-9]{1,4})' + sep, 0.3, (1, -1)),
|
||||
]
|
||||
|
||||
non_episode_title = [ 'extras' ]
|
||||
|
||||
|
||||
video_rexps = [ # cd number
|
||||
(r'cd ?(?P<cdNumber>[0-9])( ?of ?(?P<cdNumberTotal>[0-9]))?', 1.0, (0, 0)),
|
||||
(r'(?P<cdNumberTotal>[1-9]) cds?', 0.9, (0, 0)),
|
||||
|
||||
# special editions
|
||||
(r'edition' + sep + r'(?P<edition>collector)', 1.0, (0, 0)),
|
||||
(r'(?P<edition>collector)' + sep + 'edition', 1.0, (0, 0)),
|
||||
(r'(?P<edition>special)' + sep + 'edition', 1.0, (0, 0)),
|
||||
(r'(?P<edition>criterion)' + sep + 'edition', 1.0, (0, 0)),
|
||||
|
||||
# director's cut
|
||||
(r"(?P<edition>director'?s?" + sep + "cut)", 1.0, (0, 0)),
|
||||
|
||||
# video size
|
||||
(r'(?P<width>[0-9]{3,4})x(?P<height>[0-9]{3,4})', 0.9, (0, 0)),
|
||||
|
||||
# website
|
||||
(r'(?P<website>www(\.[a-zA-Z0-9]+){2,3})', 0.8, (0, 0))
|
||||
]
|
||||
|
||||
websites = [ 'tvu.org.ru', 'emule-island.com', 'UsaBit.com', 'www.divx-overnet.com', 'sharethefiles.com' ]
|
||||
|
||||
properties = { 'format': [ 'DVDRip', 'HD-DVD', 'HDDVD', 'HDDVDRip', 'BluRay', 'Blu-ray', 'BDRip', 'BRRip',
|
||||
'HDRip', 'DVD', 'DVDivX', 'HDTV', 'DVB', 'WEBRip', 'DVDSCR', 'Screener', 'VHS',
|
||||
'VIDEO_TS' ],
|
||||
|
||||
'container': [ 'avi', 'mkv', 'ogv', 'ogm', 'wmv', 'mp4', 'mov' ],
|
||||
|
||||
'screenSize': [ '720p', '720' ],
|
||||
|
||||
'videoCodec': [ 'XviD', 'DivX', 'x264', 'h264', 'Rv10' ],
|
||||
|
||||
'audioCodec': [ 'AC3', 'DTS', 'He-AAC', 'AAC-He', 'AAC' ],
|
||||
|
||||
'audioChannels': [ '5.1' ],
|
||||
|
||||
'releaseGroup': [ 'ESiR', 'WAF', 'SEPTiC', '[XCT]', 'iNT', 'PUKKA',
|
||||
'CHD', 'ViTE', 'TLF', 'DEiTY', 'FLAiTE',
|
||||
'MDX', 'GM4F', 'DVL', 'SVD', 'iLUMiNADOS', ' FiNaLe',
|
||||
'UnSeeN', 'aXXo', 'KLAXXON', 'NoTV', 'ZeaL', 'LOL' ],
|
||||
|
||||
'episodeFormat': [ 'Minisode', 'Minisodes' ],
|
||||
|
||||
'other': [ '5ch', 'PROPER', 'REPACK', 'LIMITED', 'DualAudio', 'iNTERNAL', 'Audiofixed', 'R5',
|
||||
'complete', 'classic', # not so sure about these ones, could appear in a title
|
||||
'ws', # widescreen
|
||||
#'SE', # special edition
|
||||
# TODO: director's cut
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
property_synonyms = { 'DVD': [ 'DVDRip', 'VIDEO_TS' ],
|
||||
'HD-DVD': [ 'HDDVD', 'HDDVDRip' ],
|
||||
'BluRay': [ 'BDRip', 'BRRip', 'Blu-ray' ],
|
||||
'Screener': [ 'DVDSCR' ],
|
||||
'DivX': [ 'DVDivX' ],
|
||||
'h264': [ 'x264' ],
|
||||
'720p': [ '720' ],
|
||||
'AAC': [ 'He-AAC', 'AAC-He' ],
|
||||
'Special Edition': [ 'Special' ],
|
||||
'Collector Edition': [ 'Collector' ],
|
||||
'Criterion Edition': [ 'Criterion' ],
|
||||
'Minisode': [ 'Minisodes' ]
|
||||
}
|
||||
|
||||
|
||||
reverse_synonyms = {}
|
||||
for canonical, synonyms in property_synonyms.items():
|
||||
for synonym in synonyms:
|
||||
reverse_synonyms[synonym.lower()] = canonical
|
||||
|
||||
def canonical_form(string):
|
||||
return reverse_synonyms.get(string.lower(), string)
|
||||
@@ -0,0 +1,60 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Smewt - A smart collection manager
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# Smewt is free software; you can redistribute it and/or modify
|
||||
# it under the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# Smewt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
import logging
|
||||
import sys
|
||||
|
||||
GREEN_FONT = "\x1B[0;32m"
|
||||
YELLOW_FONT = "\x1B[0;33m"
|
||||
BLUE_FONT = "\x1B[0;34m"
|
||||
RED_FONT = "\x1B[0;31m"
|
||||
RESET_FONT = "\x1B[0m"
|
||||
|
||||
|
||||
def setupLogging(colored = True):
|
||||
"""Sets up a nice colored logger as the main application logger (not only smewt itself)."""
|
||||
|
||||
class SimpleFormatter(logging.Formatter):
|
||||
def __init__(self):
|
||||
self.fmt = '%(levelname)-8s %(module)s:%(funcName)s -- %(message)s'
|
||||
logging.Formatter.__init__(self, self.fmt)
|
||||
|
||||
class ColoredFormatter(logging.Formatter):
|
||||
def __init__(self):
|
||||
self.fmt = '%(levelname)-8s ' + BLUE_FONT + '%(module)s:%(funcName)s' + RESET_FONT + ' -- %(message)s'
|
||||
logging.Formatter.__init__(self, self.fmt)
|
||||
|
||||
def format(self, record):
|
||||
result = logging.Formatter.format(self, record)
|
||||
if record.levelno in (logging.DEBUG, logging.INFO):
|
||||
return GREEN_FONT + result
|
||||
elif record.levelno == logging.WARNING:
|
||||
return YELLOW_FONT + result
|
||||
else:
|
||||
return RED_FONT + result
|
||||
|
||||
|
||||
ch = logging.StreamHandler()
|
||||
if colored and sys.platform != 'win32':
|
||||
ch.setFormatter(ColoredFormatter())
|
||||
else:
|
||||
ch.setFormatter(SimpleFormatter())
|
||||
logging.getLogger().addHandler(ch)
|
||||
|
||||
@@ -0,0 +1,210 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Smewt - A smart collection manager
|
||||
# Copyright (c) 2008 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# Smewt is free software; you can redistribute it and/or modify
|
||||
# it under the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# Smewt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.patterns import sep, deleted
|
||||
import copy
|
||||
|
||||
# string-related functions
|
||||
|
||||
def strip_brackets(s):
|
||||
if not s:
|
||||
return s
|
||||
if s[0] == '[' and s[-1] == ']': return s[1:-1]
|
||||
if s[0] == '(' and s[-1] == ')': return s[1:-1]
|
||||
if s[0] == '{' and s[-1] == '}': return s[1:-1]
|
||||
return s
|
||||
|
||||
|
||||
def clean_string(s):
|
||||
for c in sep:
|
||||
s = s.replace(c, ' ')
|
||||
parts = s.split()
|
||||
return ' '.join(p for p in parts if p != '')
|
||||
|
||||
|
||||
def str_replace(string, pos, c):
|
||||
return string[:pos] + c + string[pos+1:]
|
||||
|
||||
def blank_region(string, region, blank_sep = deleted):
|
||||
start, end = region
|
||||
return string[:start] + blank_sep * (end - start) + string[end:]
|
||||
|
||||
|
||||
def between(s, left, right):
|
||||
return s.split(left)[1].split(right)[0]
|
||||
|
||||
def to_utf8(o):
|
||||
'''converts all unicode strings found in the given object to utf-8 strings'''
|
||||
|
||||
if isinstance(o, unicode):
|
||||
return o.encode('utf-8')
|
||||
elif isinstance(o, list):
|
||||
return [ to_utf8(i) for i in o ]
|
||||
elif isinstance(o, dict):
|
||||
result = copy.deepcopy(o) # need to do it like that to handle Guess instances correctly
|
||||
for key, value in o.items():
|
||||
result[to_utf8(key)] = to_utf8(value)
|
||||
return result
|
||||
|
||||
else:
|
||||
return o
|
||||
|
||||
|
||||
def levenshtein(a, b):
|
||||
if not a: return len(b)
|
||||
if not b: return len(a)
|
||||
|
||||
m = len(a)
|
||||
n = len(b)
|
||||
d = []
|
||||
for i in range(m+1):
|
||||
d.append([0] * (n+1))
|
||||
|
||||
for i in range(m+1):
|
||||
d[i][0] = i
|
||||
|
||||
for j in range(n+1):
|
||||
d[0][j] = j
|
||||
|
||||
for i in range(1, m+1):
|
||||
for j in range(1, n+1):
|
||||
if a[i-1] == b[j-1]:
|
||||
cost = 0
|
||||
else:
|
||||
cost = 1
|
||||
|
||||
d[i][j] = min(d[i-1][j] + 1, # deletion
|
||||
d[i][j-1] + 1, # insertion
|
||||
d[i-1][j-1] + cost # substitution
|
||||
)
|
||||
|
||||
return d[m][n]
|
||||
|
||||
|
||||
# group-related functions
|
||||
|
||||
def find_first_level_groups_span(string, enclosing):
|
||||
"""Return a list of pairs (start, end) for the groups delimited by the given
|
||||
enclosing characters.
|
||||
This does not return nested groups, ie: '(ab(c)(d))' will return a single group
|
||||
containing the whole string.
|
||||
|
||||
>>> find_first_level_groups_span('abcd', '()')
|
||||
[]
|
||||
|
||||
>>> find_first_level_groups_span('abc(de)fgh', '()')
|
||||
[(3, 7)]
|
||||
|
||||
>>> find_first_level_groups_span('(ab(c)(d))', '()')
|
||||
[(0, 10)]
|
||||
|
||||
>>> find_first_level_groups_span('ab[c]de[f]gh(i)', '[]')
|
||||
[(2, 5), (7, 10)]
|
||||
"""
|
||||
opening, closing = enclosing
|
||||
depth = [] # depth is a stack of indices where we opened a group
|
||||
result = []
|
||||
for i, c, in enumerate(string):
|
||||
if c == opening:
|
||||
depth.append(i)
|
||||
elif c == closing:
|
||||
try:
|
||||
start = depth.pop()
|
||||
end = i
|
||||
if not depth:
|
||||
# we emptied our stack, so we have a 1st level group
|
||||
result.append((start, end+1))
|
||||
except IndexError:
|
||||
# we closed a group which was not opened before
|
||||
pass
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def split_on_groups(string, groups):
|
||||
"""Split the given string using the different known groups for boundaries.
|
||||
|
||||
>>> split_on_groups('0123456789', [ (2, 4) ])
|
||||
['01', '23', '456789']
|
||||
|
||||
>>> split_on_groups('0123456789', [ (2, 4), (4, 6) ])
|
||||
['01', '23', '45', '6789']
|
||||
|
||||
>>> split_on_groups('0123456789', [ (5, 7), (2, 4) ])
|
||||
['01', '23', '4', '56', '789']
|
||||
|
||||
"""
|
||||
if not groups:
|
||||
return [ string ]
|
||||
|
||||
boundaries = sorted(set(reduce(lambda l, x: l + list(x), groups, [])))
|
||||
if boundaries[0] != 0:
|
||||
boundaries.insert(0, 0)
|
||||
if boundaries[-1] != len(string):
|
||||
boundaries.append(len(string))
|
||||
|
||||
groups = [ string[start:end] for start, end in zip(boundaries[:-1], boundaries[1:]) ]
|
||||
|
||||
return filter(bool, groups) # return only non-empty groups
|
||||
|
||||
|
||||
|
||||
|
||||
def find_first_level_groups(string, enclosing, blank_sep = None):
|
||||
"""Return a list of groups that could be split because of explicit grouping.
|
||||
The groups are delimited by the given enclosing characters.
|
||||
|
||||
You can also specify if you want to blank the separator chars in the returned
|
||||
list of groups by specifying a character for it. None means it won't be replaced.
|
||||
|
||||
This does not return nested groups, ie: '(ab(c)(d))' will return a single group
|
||||
containing the whole string.
|
||||
|
||||
>>> find_first_level_groups('', '()')
|
||||
['']
|
||||
|
||||
>>> find_first_level_groups('abcd', '()')
|
||||
['abcd']
|
||||
|
||||
>>> find_first_level_groups('abc(de)fgh', '()')
|
||||
['abc', '(de)', 'fgh']
|
||||
|
||||
>>> find_first_level_groups('(ab(c)(d))', '()', blank_sep = '_')
|
||||
['_ab(c)(d)_']
|
||||
|
||||
>>> find_first_level_groups('ab[c]de[f]gh(i)', '[]')
|
||||
['ab', '[c]', 'de', '[f]', 'gh(i)']
|
||||
|
||||
>>> find_first_level_groups('()[]()', '()', blank_sep = '-')
|
||||
['--', '[]', '--']
|
||||
|
||||
"""
|
||||
groups = find_first_level_groups_span(string, enclosing)
|
||||
if blank_sep:
|
||||
for start, end in groups:
|
||||
string = str_replace(string, start, blank_sep)
|
||||
string = str_replace(string, end-1, blank_sep)
|
||||
|
||||
return split_on_groups(string, groups)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user