diff --git a/.gitignore b/.gitignore index 3707b770..0f1eb04b 100644 --- a/.gitignore +++ b/.gitignore @@ -103,3 +103,25 @@ ENV/ # IDE settings .vscode/ + +# training artifacts (large, regenerable or archival) +corpus/ +corpus.shards/ +corpus_extra/ +corpus_dropped/ +raw_downloads/ +model_out/ +archive/ +benchmarks/ + +# uv +uv.lock + +# AI assistants +.claude/ +CLAUDE.local.md +.cursor/ +.aider* +.codex/ +.gemini/ +.github/copilot-instructions.md diff --git a/FEATURES b/FEATURES deleted file mode 100644 index 3ce98db3..00000000 --- a/FEATURES +++ /dev/null @@ -1,7480 +0,0 @@ -'\n\xc7' -'\n\xce' -'\n\xd7' -'\n\xd8' -'\n\xd9' -'\n\xda' -'\n\xe1' -'\n\xec' -' (K' -' Ak' -' Ben' -' Bil' -' Dan' -' De ' -' Den' -' Der' -' Det' -' Deu' -' Die' -' ES' -' Ei' -' Ein' -' El' -' El ' -' Er' -' Es' -' Est' -' Et' -' Ges' -' Id' -' Il' -' Il ' -' It ' -' Ita' -' Ku' -' K\xc3' -' Le' -' Le ' -' Les' -' Los' -' Mag' -' Mit' -' M\xc3' -' Nac' -' Ny' -' O ' -' Om' -' Pen' -' Sel' -' Sm' -' The' -' Thi' -' T\xc3' -' USA' -' Ut' -' Ver' -' Vo' -' Vor' -' War' -' Zu' -' a ' -' a C' -' a a' -' a e' -' a k' -' a l' -' a u' -' aa' -' abr' -' ace' -' af' -' aft' -' ah' -' al ' -' ala' -' ale' -' alg' -' am ' -' amb' -' an' -' an ' -' anc' -' and' -' ang' -' ans' -' any' -' ao' -' aq' -' are' -' as ' -' ase' -' asi' -' at ' -' au' -' au ' -' auf' -' aus' -' aux' -' av' -' av ' -' ava' -' ave' -' ay' -' az' -' a\xc3' -' bal' -' bat' -' be ' -' bec' -' bee' -' bei' -' bek' -' ber' -' bes' -' bie' -' bis' -' ble' -' bli' -' bru' -' bud' -' but' -' by ' -' cad' -' ce ' -' cet' -' che' -' cho' -' cua' -' c\xc3' -" d'" -' d.' -' da ' -' dal' -' dan' -' dar' -' das' -' de' -' de ' -' deb' -' deg' -' dei' -' del' -' dem' -' den' -' der' -' des' -' det' -' di ' -' dic' -' die' -' dim' -' din' -' dit' -' do ' -' dol' -' dop' -' dos' -' dou' -' dre' -' du' -' du ' -' d\xc3' -' d\xc3\xa9' -' e ' -' e a' -' e d' -' e i' -' ea' -' ee' -' eg' -' ei' -' ein' -' el ' -' ell' -' em' -' em ' -' en ' -' er' -' er ' -' era' -' ers' -' es' -' es ' -' ese' -' esp' -' ess' -' est' -' et' -' et ' -' ett' -' fai' -' fel' -' fer' -' fie' -' fis' -' fle' -' foi' -' fro' -' fr\xc3' -' fue' -' f\xc3' -' f\xc3\xbc' -' gal' -' gan' -' ge' -' geb' -' gel' -' ger' -' gew' -' g\xc3' -' ha ' -' hac' -' hal' -' har' -' has' -' hat' -' hav' -' hie' -' hin' -' hor' -' hum' -' hy' -' h\xc3' -' h\xc3\xa4' -' i ' -' i d' -' i f' -' ie' -' ih' -' ik' -' il' -' il ' -' im ' -' in ' -' ing' -' inn' -' ir' -' is' -' is ' -' ist' -' it ' -' its' -' iz' -' ja ' -' je' -' je ' -' jed' -' ji' -' jo' -' joi' -' jou' -' k ' -' kin' -' kt' -' kur' -' k\xc3\xb6' -" l'" -' la ' -' lai' -' las' -' lau' -' le ' -' lea' -' lei' -' les' -' li ' -' lie' -' ll' -' lo ' -' lor' -' los' -' l\xc3\xa4' -' mal' -' mee' -' mel' -' mie' -' mis' -' mit' -' m\xc3\xa1' -' m\xc3\xa9' -' nac' -' nad' -' nar' -' nas' -' ne ' -' nel' -' nem' -' neu' -' nic' -' nie' -' nin' -' no ' -' nou' -' nue' -' nur' -' ny' -' n\xc3\xa4' -' o ' -' o c' -' o m' -' obt' -' oc' -' och' -' od' -' ode' -' of' -' of ' -' og' -' oh' -' ol' -' om ' -' on ' -' one' -' ont' -' opi' -' opp' -' or ' -' oth' -' otr' -' ou ' -' pal' -' pan' -' pap' -' pas' -' pel' -' pen' -' pie' -' plu' -' pod' -' por' -' pou' -' pow' -' pue' -' p\xc3\xa5' -' q' -' qu' -' que' -' qui' -' rei' -' ren' -' ric' -' ris' -' roz' -' r\xc3\xa9' -' s ' -' sa ' -' sai' -' sal' -' sau' -' sav' -' sca' -' se ' -' seg' -' sei' -' sem' -' seu' -' si ' -' sic' -' sk' -' ska' -' sob' -' som' -' son' -' sou' -' ste' -' st\xc3' -' su ' -' sul' -' sur' -' sus' -' sz' -' s\xc3\xa4' -' tai' -' tal' -' tam' -' tan' -' tas' -' te ' -' tea' -' tei' -' th' -' tha' -' the' -' thi' -' tho' -' thr' -' tid' -' tie' -' til' -' tin' -' to ' -' tom' -' tor' -' tot' -' tou' -' tro' -' tr\xc3' -' tut' -' tv' -' t\xc3\xa9' -' u ' -' ud' -' ult' -' um' -' um ' -' un ' -' una' -' und' -' une' -' uno' -' unt' -' up ' -' ust' -' uv' -' v ' -' van' -' vas' -' ved' -' vel' -' vil' -' vo' -' voi' -' vol' -' von' -' vor' -' vr' -' vu' -' vy' -' v\xc3\xa4' -' v\xc4' -' wa' -' war' -' was' -' wer' -' wh' -' whe' -' whi' -' who' -' wie' -' wil' -' wir' -' wit' -' wo' -' wor' -' wu' -' wur' -' y ' -' y a' -' y c' -' y d' -' y e' -' y l' -' ya' -' ye' -' yea' -' z' -' z ' -' za' -' ze' -' zi' -' zm' -' zo' -' zu' -' zu ' -' zum' -' zur' -' zw' -' z\xc3' -' \xc3\x9c' -' \xc3\xa0' -' \xc3\xa0 ' -' \xc3\xa1' -' \xc3\xa4' -' \xc3\xa5' -' \xc3\xa9' -' \xc3\xa9t' -' \xc3\xb6' -' \xc3\xba' -' \xc3\xbc' -' \xc3\xbcb' -' \xc4\x8d' -' \xc5\xa0' -' \xc5\xbe' -' \xce' -' \xcf' -' \xd0\x9d' -' \xd0\x9d\xd0' -' \xd0\x9f' -' \xd0\xa1' -' \xd0\xb0' -' \xd0\xb0\xd0' -' \xd0\xb1' -' \xd0\xb1\xd1' -' \xd0\xb2' -' \xd0\xb2 ' -' \xd0\xb2\xd0' -' \xd0\xb2\xd1' -' \xd0\xb3' -' \xd0\xb3\xd0' -' \xd0\xb4' -' \xd0\xb4\xd0' -' \xd0\xb5' -' \xd0\xb7' -' \xd0\xb7\xd0' -' \xd0\xb8' -' \xd0\xb8 ' -' \xd0\xb8\xd0' -' \xd0\xba' -' \xd0\xbb' -' \xd0\xbc' -' \xd0\xbc\xd0' -' \xd0\xbd' -' \xd0\xbd\xd0' -' \xd0\xbe' -' \xd0\xbe\xd0' -' \xd0\xbe\xd1' -' \xd0\xbf' -' \xd0\xbf\xd0' -' \xd0\xbf\xd1' -' \xd1' -' \xd1\x80' -' \xd1\x80\xd0' -' \xd1\x81' -' \xd1\x81\xd0' -' \xd1\x81\xd1' -' \xd1\x82' -' \xd1\x82\xd0' -' \xd1\x86' -' \xd1\x87' -' \xd1\x87\xd0' -' \xd8' -' \xd9' -' \xda' -' \xdb' -' \xdd' -' \xde' -' \xdf' -' \xe1' -' \xe3' -' \xe4' -' \xea' -' \xeb' -' \xec' -' \xed' -'"de"' -'"es' -'"es"' -'"fr"' -'"\xd0' -'"\xd1' -'"\xd7' -'"\xe8' -'%U' -'%y' -"'s " -"'\xc4" -"'\xce" -"'\xd1" -"'\xd7" -'(K' -'(\xc4' -'(\xce' -'(\xcf' -'(\xd7' -'(\xd8' -'(\xd9' -', ab' -', ac' -', an' -', at' -', cu' -', el' -', er' -', et' -', ho' -', il' -', kt' -', le' -', mi' -', o ' -', th' -', w' -', wh' -', \xd1' -',\xe5' -'- Le' -'-1 ' -'-K' -'-Pr' -'-Re' -'-ko' -'-op' -'-\xc3' -'-\xc4' -'-\xc5' -'-\xce' -'-\xd0' -'-\xd1' -'-\xd7' -'. Di' -'. Il' -'. Le' -'. Th' -'.(' -'.cn' -'.\xd0' -'/\xce' -'1 t' -'1\xce' -'="es' -'>\xb5' -'>\xb7' -'>\xb8' -'>\xb9' -'>\xd2' -'>\xd6' -'>\xd7' -'>\xd8' -'>\xda' -'>\xe1' -'>\xec' -'A l' -'A t' -'AJ' -'AND' -'AST' -'Aq' -'Auf' -'Az' -'A\xc3' -'B)' -'Ben' -'Bil' -'B\xe1' -'Cz' -'C\xc3' -'C\xe1' -'Dan' -'Das' -'Das ' -'De ' -'Den' -'Der' -'Der ' -'Det' -'Deu' -'Deut' -'Dez' -'Die' -'Die ' -'Din' -'Dz' -'ET ' -'EW' -'Ee' -'Eg' -'Ei' -'Ein' -'El ' -'En ' -'Es' -'Es ' -'Est' -'Esta' -'Ew' -'Ez' -'E\xc3' -'E\xc5' -'F\xc3' -'GK' -'Ges' -'Gj' -'G\xc4' -'Hau' -'Hv' -'H\xc3' -'ITA' -'IU' -'Ib' -'Ie' -'Il' -'Il ' -'Ir' -'Ist' -'It ' -'Ita' -'Ital' -'Iz' -'I\xc4' -'I\xc5' -'J,' -'JI' -'JM' -'Jah' -'J\xc3' -'J\xc4' -'J\xc5' -'K)' -'K,' -'KU' -'KY' -'Kh' -'Kj' -'Kla' -'Ku' -'Kun' -'K\xc4' -'L-' -'LJ' -'LL ' -'La ' -'Le ' -'Les' -'Les ' -'Ll' -'Los' -'Los ' -'L\xc4' -'Mag' -'Mit' -'M\xc3' -'M\xc4' -'M\xe1' -'NJ' -'Nac' -'Nach' -'Nd' -'Ng' -'Nh' -'Nie' -'Ny' -'ONI' -'OS ' -'O\xc4' -'PB' -'Pen' -'Pie' -'Pou' -'Pr\xc3' -'P\xc4' -'P\xc5' -'RW' -'R\xc4' -'SA ' -'Seg' -'Sj' -'Spr' -'Sv' -'Sz' -'S\xc4' -'TAL' -'TN' -'Tai' -'Tar' -'The' -'The ' -'Thi' -'This' -'Tie' -'T\xc4' -'T\xe1' -'UD' -'UJ' -'USE' -'UST' -'UV' -'Ud' -'Uf' -'Un ' -'Unt' -'Uu' -'VN' -'VR' -'Vor' -'Vy' -'V|' -'V\xc4' -'V\xc5' -'W ' -'WR' -'War' -'Wh' -'Ws' -'Wy' -'W\xc5' -'Y)' -'YC' -'YH' -'YJ' -'YV' -'Yh' -'Yl' -'Y\xc3' -'Z)' -'ZT' -'ZY' -'Za' -'Ze' -'Zei' -'Zi' -'Zm' -'Zu' -'Z\xc3' -'Z\xc4' -'[\xe0' -'a a ' -'a ac' -'a av' -'a bi' -'a bo' -'a ca' -'a ce' -'a ch' -'a co' -'a cr' -'a cu' -'a de' -'a di' -'a e' -'a e ' -'a el' -'a em' -'a en' -'a er' -'a es' -'a ex' -'a fi' -'a fu' -'a i ' -'a il' -'a jo' -'a la' -'a lo' -'a mi' -'a mu' -'a nu' -'a o ' -'a pl' -'a q' -'a qu' -'a ri' -'a un' -'a w' -'a y' -'a y ' -'a \xc3\xa9' -"a' " -'aa ' -'aad' -'aal' -'aam' -'aan' -'aan ' -'aar' -'aas' -'aat' -'ab ' -'aban' -'abb' -'aben' -'aber' -'abil' -'ab\xc3' -'ac ' -'acci' -'ach ' -'achr' -'acht' -'acio' -'aci\xc3' -'actu' -'ac\xc3' -'ad d' -'ada ' -'ada,' -'ada.' -'adan' -'adas' -'aden' -'ades' -'ado' -'ado ' -'ado.' -'ador' -'ados' -'ad\xc3' -'aes' -'af ' -'ag ' -'agg' -'aggi' -'agl' -'agre' -'ah ' -'ahe' -'ahi' -'ahl' -'ahr' -'ahre' -'ahu' -'ai p' -'ai.' -'aie' -'aik' -'aio' -'air' -'aire' -'ais' -'ais ' -'aise' -'ait' -'ait ' -'aja ' -'aji' -'ake ' -'aken' -'aks' -'alar' -'ald' -'alde' -'ale ' -'ales' -'alg' -'algu' -'alh' -'alia' -'alie' -'alim' -'all ' -'all-' -'alla' -'alli' -'ally' -'alma' -'alme' -'aln' -'alor' -'als ' -'also' -'alt ' -'alta' -'alti' -'al\xc3' -'ama ' -'amas' -'ambi' -'amis' -'aml' -'amma' -'amme' -'amn' -'amos' -'amt' -'an d' -'an h' -'an k' -'an l' -'an m' -'ana ' -'anas' -'anca' -'and ' -'ande' -'andr' -'ane ' -'ang ' -'anga' -'angi' -'angs' -'anh' -'ania' -'anie' -'anj' -'anl' -'anm' -'anno' -'ano ' -'anq' -'ans ' -'ansa' -'ansk' -'ant ' -'ante' -'anv' -'any ' -'anza' -'an\xc3\xa7' -'ao ' -'aos' -'apro' -'ap\xc3' -'aque' -'ar k' -'ar l' -'ara ' -'arad' -'arba' -'arbe' -'are ' -'are,' -'arec' -'ario' -'arj' -'arna' -'arv' -'ary ' -'arz' -'ar\xc3\xa1' -'as (' -'as d' -'as e' -'as k' -'as l' -'as p' -'as u' -'as v' -'as y' -'as)' -'as. ' -'asa ' -'asi ' -'asl' -'assa' -'asta' -'asu' -'as\xc3' -'at d' -'at f' -'at l' -'at t' -'atan' -'ated' -'aten' -'ati ' -'atie' -'ato ' -'atos' -'atre' -'ats ' -'att ' -'atti' -'atv' -'aty' -'atz' -'at\xc4' -'au ' -'au d' -'auch' -'aue' -'auf' -'auf ' -'auk' -'aun' -'aus' -'aus ' -'auti' -'aux' -'aux ' -'av ' -'ava ' -'aval' -'avan' -'ave ' -'avec' -'aven' -'avr' -'avs' -'aw ' -'awi' -'awn' -'aya' -'az ' -'aza' -'azi' -'azio' -'azo' -'azz' -'az\xc3' -'a\xc3' -'a\xc3\xa7' -'a\xc3\xad' -'a\xc3\xb1' -'a\xc3\xb1o' -'a\xc5\xa1' -'a\xc8' -'a\xe2' -'b e' -'bag' -'bai' -'baj' -'bak' -'ban ' -'bau' -'bba' -'bbi' -'bbl' -'be ' -'bee' -'been' -'bei' -'bei ' -'bek' -'ben' -'ben ' -'bera' -'beri' -'bie' -'bij' -'bile' -'bis' -'biz' -'bi\xc3' -'bni' -'bo ' -'bos' -'bra ' -'bre ' -'bres' -'bte' -'bud' -'by ' -'by s' -'by t' -'b\xc3\xa1' -'b\xc3\xa9' -'b\xc3\xad' -'b\xe1' -'b\xe2' -'c c' -'c l' -'c n' -'c t' -'ca ' -'ca d' -'caci' -'cad' -'cada' -'cado' -'cal ' -'camb' -'can ' -'cana' -'care' -'cas ' -'cato' -'caz' -'cci' -'ccio' -'cci\xc3' -'ce d' -'cea' -'ced ' -'cele' -'cerc' -'cere' -'cet' -'cett' -'ch ' -'ch d' -'ch s' -'ch v' -'ch w' -'ch z' -'cha ' -'chaf' -'che ' -'chei' -'chen' -'chl' -'cho ' -'chr' -'chri' -'chs' -'chst' -'cht' -'cht ' -'chte' -'chti' -'chts' -'chw' -'chy' -'ci ' -'cia ' -'ciar' -'cias' -'cida' -'cido' -'cie ' -'cij' -'cio' -'cio ' -'cion' -'cios' -'ciu' -'ci\xc3' -'ci\xc3\xb3' -'cj' -'co ' -'co d' -'cola' -'cole' -'como' -'con ' -'cos ' -'cta' -'cto ' -'cu ' -'cua' -'cue' -'cui' -'cun' -'cyc' -'cz' -'c\xc3\xa1' -'c\xc3\xa9' -'c\xc3\xad' -'c\xc3\xb3' -'c\xc4' -'c\xc8' -'c\xe1' -'d a ' -'d an' -'d as' -'d by' -'d f' -'d fo' -'d i' -'d in' -'d k' -'d mi' -'d o' -'d of' -'d on' -'d op' -'d t' -'d th' -'d ti' -'d to' -'d vo' -'d \xc3' -"d'u" -'da c' -'da e' -'da p' -'da v' -'da. ' -'daa' -'dab' -'dad' -'dad ' -'dade' -'dado' -'dag' -'dak' -'dal ' -'dami' -'dans' -'dari' -'das' -'das ' -'dat ' -'dau' -'dav' -'dda' -'de C' -'de J' -'de R' -'de b' -'de f' -'de g' -'de h' -'de j' -'de l' -'de q' -'de v' -'ded' -'ded ' -'dei' -'dei ' -'del ' -'dela' -'dell' -'dels' -'dem ' -'den ' -'den.' -'dend' -'deni' -'denn' -'dere' -'des,' -'des.' -'desa' -'desd' -'dese' -'desp' -'det ' -'dez' -'dha' -'di ' -'di a' -'di d' -'di p' -'di s' -'dib' -'dice' -'die' -'die ' -'dien' -'dig' -'dige' -'dim' -'din ' -'dip' -'diri' -'dit ' -'diz' -'dn\xc3' -'do ' -'do a' -'do c' -'do d' -'do e' -'do l' -'do m' -'do n' -'do o' -'do p' -'do q' -'do s' -'do u' -'do,' -'do, ' -'do. ' -'dob' -'dol' -'doo' -'door' -'dor' -'dor ' -'dore' -'dos' -'dos ' -'dpo' -'dra ' -'dre ' -'drin' -'dsk' -'dt ' -'dte' -'du ' -'dun' -'dung' -'dup' -'durc' -'dus' -'duz' -'dwa' -'dwar' -'dz' -'dzi' -'d\xc3\xa1' -'d\xc3\xa9' -'d\xc3\xa9c' -'d\xc3\xad' -'d\xc3\xb3' -'d\xe1' -'e Tr' -'e ac' -'e af' -'e au' -'e av' -'e cu' -'e d' -'e de' -'e di' -'e du' -'e d\xc3' -'e e' -'e e ' -'e ei' -'e el' -'e em' -'e er' -'e es' -'e et' -'e f\xc3' -'e ge' -'e i ' -'e il' -'e ja' -'e je' -'e ki' -'e la' -'e le' -'e lo' -'e lu' -'e nu' -'e o ' -'e oc' -'e of' -'e pe' -'e q' -'e qu' -'e ri' -'e r\xc3' -'e se' -'e th' -'e um' -'e un' -'e ve' -'e vo' -'e wa' -'e wh' -'e z' -'e zu' -'e \xc3\xa0' -'e \xc3\xa9' -'ea d' -'eab' -'eas' -'ease' -'eau ' -'ebb' -'ebe' -'eben' -'eber' -'ebn' -'ec ' -'ecc' -'ecci' -'ecer' -'eces' -'ech ' -'echt' -'ecie' -'ecla' -'ecta' -'ectu' -'ed a' -'ed b' -'ed f' -'ed i' -'ed o' -'ed t' -'ed w' -'edan' -'edd' -'ede ' -'eder' -'edes' -'edi ' -'edn' -'ed\xc3' -'ee ' -'eel ' -'een ' -'eens' -'eer' -'eer ' -'efec' -'eg ' -'ega ' -'ege' -'egel' -'egen' -'egg' -'egn' -'ego ' -'egt' -'egui' -'egun' -'egur' -'egy' -'eg\xc3' -'ehe' -'ehen' -'ehr' -'eht' -'eh\xc3' -'ei ' -'ei d' -'ei.' -'eia' -'eic' -'eich' -'eide' -'eige' -'eik' -'eil' -'eil ' -'eill' -'ein' -'ein ' -'eine' -'eins' -'eir' -'eir ' -'eiro' -'eis' -'eis ' -'eise' -'eist' -'eit' -'eit ' -'eita' -'eite' -'eits' -'eiz' -'ej ' -'eja' -'eje' -'ejo' -'ejs' -'eke' -'eken' -'ekk' -'ekl' -'eks ' -'ek\xc3' -'el a' -'el b' -'el c' -'el g' -'el l' -'el m' -'el n' -'el p' -'el r' -'el s' -'el t' -'el v' -'ela ' -'elar' -'eld ' -'elde' -'ele ' -'eler' -'elg' -'elh' -'elig' -'elj' -'ella' -'elm' -'elo ' -'els ' -'elt' -'elt ' -'elta' -'em c' -'em n' -'emal' -'emis' -'emme' -'emon' -'empr' -'en A' -'en S' -'en a' -'en b' -'en c' -'en d' -'en e' -'en g' -'en h' -'en i' -'en j' -'en k' -'en l' -'en m' -'en n' -'en s' -'en v' -'en w' -'en y' -'en z' -'en, ' -'en.' -'en. ' -'enar' -'enas' -'enb' -'enci' -'ende' -'endo' -'endr' -'endu' -'ene ' -'enes' -'enga' -'enh' -'enia' -'enie' -'enir' -'enj' -'enm' -'enna' -'enne' -'eno ' -'enos' -'enp' -'ens ' -'ensc' -'ensk' -'ento' -'enut' -'envi' -'enz' -'enza' -'en\xc3' -'en\xc3\xa9' -'en\xc3\xad' -'epri' -'er K' -'er b' -'er d' -'er e' -'er f' -'er g' -'er h' -'er i' -'er j' -'er k' -'er l' -'er m' -'er p' -'er s' -'er u' -'er v' -'era ' -'erad' -'erar' -'eras' -'erbe' -'erbi' -'erd' -'erde' -'erdi' -'erei' -'erem' -'erer' -'erh' -'erha' -'erim' -'erj' -'erk' -'erke' -'erle' -'erne' -'ero ' -'ersc' -'erst' -'erte' -'eru' -'erun' -'erw' -'erwe' -'erz' -'er\xc3\xa1' -'er\xc3\xad' -'es a' -'es d' -'es j' -'es l' -'es p' -'es q' -'es u' -'es y' -'es \xc3' -'esa' -'esa ' -'esar' -'esb' -'esch' -'esde' -'eset' -'esis' -'esit' -'eso ' -'esol' -'espr' -'ess ' -'ess\xc3' -'este' -'estu' -'esz' -'es\xc3' -'et b' -'et d' -'et e' -'et f' -'et g' -'et h' -'et i' -'et l' -'et w' -'et. ' -'ete ' -'etes' -'etk' -'etl' -'ets ' -'ett ' -'etta' -'etto' -'etz' -'etzt' -'eu ' -'eud' -'eue' -'eur' -'eur ' -'eurs' -'euts' -'euv' -'eux' -'eux ' -'ev ' -'eva ' -'evo' -'evr' -'ewi' -'ex ' -'ez ' -'eza' -'ezu' -'ezz' -'e\xc3\xa7' -'e\xc3\xb1' -'e\xc4\x8d' -'e\xc8' -'f a' -'f a ' -'f d' -'f de' -'f e' -'f t' -'f th' -'fa ' -'fai' -'fall' -'fei' -'fel' -'fen ' -'feri' -'fge' -'fi ' -'fis' -'foi' -'for ' -'fos' -'fr"' -'fr">' -'fran' -'fro' -'from' -'fr\xc3' -'fs ' -'fti' -'fue' -'f\xc3\xa4' -'f\xc3\xa9' -'f\xc3\xb6' -'f\xc3\xbc' -'f\xc3\xbcr' -'f\xc4' -'f\xc5' -'f\xe2' -'g be' -'g d' -'g de' -'g di' -'g e' -'g fr' -'g g' -'g ge' -'g ha' -'g m' -'g se' -'g t' -'g th' -'g ti' -'g to' -'g v' -'g vo' -'g="f' -'ga ' -'ga,' -'gabe' -'gad' -'gado' -'gal ' -'gan ' -'gang' -'gar ' -'gare' -'gde' -'ge d' -'geb' -'gebe' -'ged ' -'gee' -'gef' -'geg' -'gege' -'geh' -'gek' -'gel' -'gele' -'geli' -'gen ' -'gen,' -'gen.' -'geno' -'gens' -'gep' -'ger ' -'gere' -'gesc' -'gew' -'gez' -'gga' -'ggi' -'gh ' -'gha' -'gib' -'gig' -'giu' -'give' -'gi\xc3' -'gj' -'gk' -'gli' -'go ' -'go d' -'god' -'gol' -'gos ' -'grun' -'gse' -'gsm' -'gst' -'gt ' -'gte' -'gu ' -'gun' -'gund' -'gung' -'gut' -'gw' -'gy ' -'g\xc3\xa5' -'g\xc3\xa9' -'g\xc3\xb3' -'g\xc5' -'g\xe1' -'h a' -'h be' -'h d' -'h da' -'h di' -'h g' -'h se' -'h t' -'h th' -'h v' -'h w' -'ha ' -'ha d' -'ha.' -'habe' -'habi' -'haf' -'haft' -'hal' -'halt' -'har ' -'has ' -'hat' -'hat ' -'hau' -'have' -'hay' -'hde' -'he ' -'he a' -'he b' -'he c' -'he e' -'he f' -'he l' -'he m' -'he n' -'he o' -'he p' -'he r' -'he s' -'he t' -'he w' -'heb' -'hen' -'hen ' -'hend' -'her ' -'here' -'het' -'het ' -'hete' -'heu' -'hey' -'hey ' -'hez' -'hi ' -'hia' -'hic' -'hich' -'hing' -'his ' -'hj' -'hla' -'hle' -'hme' -'hne' -'ho ' -'ho d' -'ho p' -'ho s' -'hold' -'hou' -'hra' -'hren' -'hri' -'hric' -'hta' -'hte' -'hten' -'hter' -'hti' -'hto' -'htu' -'hu ' -'hui' -'hul' -'huma' -'hv' -'h\xc3\xa4' -'h\xc3\xb6' -'h\xc4' -'h\xc5' -'h\xc6' -'h\xe1' -'i M' -'i au' -'i ca' -'i co' -'i da' -'i de' -'i di' -'i ha' -'i i ' -'i in' -'i pe' -'i q' -'i qu' -'i ri' -'i so' -'i un' -'i w' -'i z' -'i \xc3' -'ia c' -'ia d' -'iac' -'iad' -'iada' -'iado' -'iai' -'iaj' -'iame' -'iano' -'iant' -'iar' -'iar ' -'ias' -'ias ' -'ic ' -'ica ' -'icac' -'icad' -'icar' -'icas' -'ich' -'ich ' -'iche' -'ichi' -'icht' -'ici' -'ici ' -'icia' -'icio' -'ici\xc3' -'ico ' -'icol' -'icos' -'id-' -'ida ' -'idad' -'idas' -'ido' -'ido ' -'idos' -'idr' -'ie ' -'ie A' -'ie m' -'ie n' -'ie o' -'ie p' -'ie s' -'ie u' -'ie v' -'ie w' -'ie z' -'ieg' -'ieh' -'iej' -'iel ' -'iele' -'ien ' -'ieni' -'ienn' -'ier ' -'iere' -'iers' -'iert' -'iese' -'iet ' -'iete' -'ieu' -'iez' -'ifa' -'ig ' -'iga ' -'ige' -'ige ' -'igen' -'iger' -'ighe' -'igl' -'igo ' -'igs' -'igt' -'igt ' -'ih' -'iha' -'ihe' -'ihr' -'ii ' -'iin' -'ij' -'ij ' -'ije' -'iji' -'ijo' -'ikke' -'ikr' -'il c' -'il d' -'il e' -'il f' -'il m' -'il p' -'il s' -'il-' -'ila ' -'ilan' -'ilb' -'ilf' -'ilh' -'ilis' -'iliz' -'ilk' -'ill ' -'ille' -'illi' -'ilo' -'im ' -'imas' -'imb' -'imis' -'imo' -'imo ' -'imul' -'in D' -'in R' -'in d' -'in e' -'in h' -'in k' -'in t' -'inam' -'inea' -'inen' -'inga' -'inge' -'ingi' -'ingu' -'ini ' -'inia' -'inie' -'inio' -'inj' -'intr' -'inua' -'io ' -'io d' -'io p' -'io s' -'io, ' -'ione' -'ioni' -'ionn' -'ions' -'ios' -'ios ' -'ipo ' -'iq' -'iqu' -'ique' -'ir ' -'ir a' -'ir e' -'ir l' -'ir p' -'ir s' -'ir t' -'ir-' -'ira ' -'iran' -'ire ' -'iret' -'iro' -'iro ' -'irr' -'ir\xc3' -'is ' -'is a' -'is v' -'is-' -'isa ' -'isan' -'isch' -'ise ' -'ised' -'isen' -'iser' -'ises' -'isie' -'isj' -'isk ' -'iska' -'isn' -'issa' -'istu' -'ist\xc3' -'is\xc3' -'is\xc3\xa9' -'it e' -'it p' -'it u' -'it v' -'itad' -'iten' -'itg' -'ith' -'ith ' -'itm' -'ito ' -'itos' -'itt' -'itt ' -'itta' -'itte' -'itti' -'itud' -'ity ' -'itz' -'it\xc3\xa0' -'it\xc3\xa4' -'it\xc3\xa9' -'it\xc4' -'iu ' -'iun' -'ius' -'ivo' -'ivo ' -'iv\xc3' -'iw' -'ixa' -'iz ' -'izad' -'izar' -'izi' -'izie' -'izz' -'izza' -'i\xc3\xa7' -'i\xc3\xa8' -'i\xc3\xa8r' -'i\xc3\xa9' -'i\xc3\xb3' -'i\xc3\xb3 ' -'i\xc3\xb3n' -'i\xc5\xa1' -'i\xc8' -'i\xe1' -'j:' -'ja e' -'jad' -'jai' -'jam' -'jap' -'jar' -'jd' -'jde' -'je v' -'jeg' -'jel' -'jer' -'jes' -'jest' -'jet' -'jf' -'ji ' -'jim' -'jin' -'jis' -'jj' -'jk' -'jm' -'jn ' -'jo ' -'joi' -'jon' -'jos' -'jos ' -'jou' -'jour' -'jr' -'jt' -'jum' -'jv' -'jz' -'j\xc3\xa1' -'j\xc3\xa4' -'j\xc4' -'j\xc4\x85' -'k (' -'k e' -'k g' -'k m' -'k me' -'k \xc3' -'k. ' -'kaa' -'kab' -'kac' -'kad' -'kai' -'kan ' -'kann' -'kau' -'ke o' -'ke s' -'ke,' -'keh' -'kei' -'keit' -'ken' -'ken ' -'kend' -'ket ' -'kh' -'kia' -'kic' -'kid' -'kie ' -'kir' -'kis' -'kj' -'kka' -'kke' -'kki' -'kko' -'klu' -'kl\xc3' -'kna' -'kni' -'know' -'koh' -'kok' -'komm' -'koo' -'kou' -'kov' -'kse' -'ksi' -'kt ' -'kt.' -'kte' -'kte ' -'kter' -'ktio' -'kt\xc3' -'kur' -'ky ' -'k\xc3\xa1' -'k\xc3\xa4' -'k\xc3\xa9' -'k\xc3\xb6' -'k\xc3\xbc' -'k\xc3\xbd' -'k\xe1' -'l U' -'l a ' -'l al' -'l ar' -'l at' -'l ba' -'l ca' -'l d' -'l de' -'l di' -'l es' -'l fi' -'l gr' -'l j' -'l mi' -'l qu' -'l se' -'l ta' -'l th' -'l un' -'l vi' -'l vo' -"l'a" -"l'i" -'l-m' -'l-p' -'la a' -'la c' -'la e' -'la f' -'la g' -'la q' -'la s' -'la v' -'laa' -'laci' -'lada' -'lade' -'lado' -'lag ' -'lage' -'lah' -'lais' -'lama' -'lan ' -'lara' -'lare' -'lari' -'las ' -'lau' -'laz' -'ld ' -'ld t' -'lda' -'lde ' -'lden' -'ldo' -'ldu' -'le d' -'le g' -'le m' -'le p' -'le \xc3' -'leas' -'led ' -'lede' -'leh' -'lei' -'lem ' -'lema' -'len ' -'lena' -'lep' -'ler ' -'lera' -'leri' -'les ' -'leu' -'leur' -'le\xc5' -'lfe' -'lge' -'lgen' -'lgu' -'lh' -'lha' -'lhe' -'lho' -'li a' -'li d' -'li t' -'lich' -'lie ' -'lig ' -'lige' -'lii' -'lij' -'lika' -'lil' -'lio' -'lio ' -'lir' -'lisa' -'lise' -'lisi' -'lit\xc3' -'liu' -'liza' -'lk ' -'ll ' -'ll d' -'ll e' -'ll f' -"ll'" -'lla' -'lla ' -'llan' -'lle ' -'lleg' -'llem' -'llen' -'lli ' -'llig' -'llis' -'llo ' -'llt' -'lly' -'lly ' -'ll\xc3' -'lma' -'lmen' -'lmi' -'lne' -'lno' -'ln\xc3' -'lo ' -'lo d' -'lo s' -'loi' -'lon ' -'lone' -'lor ' -'lors' -'los ' -'lo\xc5' -'lqu' -'ls d' -'lso' -'lso ' -'lt a' -'lta ' -'lte ' -'lten' -'lt\xc3' -'lui' -'lui ' -'luk' -'lul' -'lung' -'lus ' -'lust' -'lvo' -'ly ' -'ly a' -'ly t' -'l\xc2' -'l\xc3\xa1' -'l\xc3\xa4' -'l\xc3\xa9' -'l\xc3\xad' -'l\xc3\xb3' -'l\xc3\xb6' -'l\xc3\xbc' -'l\xe1' -'m a ' -'m co' -'m ta' -'m th' -'maa' -'maci' -'mai ' -'mais' -'mala' -'mant' -'mar ' -'mas ' -'ma\xc3' -'mba' -'mbra' -'mbre' -'me l' -'med ' -'mee' -'meg' -'mej' -'mena' -'meno' -'mer ' -'mest' -'met ' -'mett' -'mez' -'mia' -'mie' -'mien' -'mill' -'mine' -'mio' -'mis ' -'mise' -'mist' -'mit ' -'miti' -'mm ' -'mma ' -'mme ' -'mna' -'mo ' -'mo a' -'mo p' -'mo s' -'mo,' -'moc' -'moi' -'mok' -'mos ' -'mse' -'mst' -'mt ' -'mui' -'mz' -'m\xc3\xa1' -'m\xc3\xa1s' -'m\xc3\xa4' -'m\xc3\xa5' -'m\xc3\xa9' -'m\xc3\xa9r' -'m\xc3\xad' -'m\xc3\xb3' -'m\xc3\xb6' -'m\xc3\xbc' -'m\xe1' -'n La' -'n ac' -'n af' -'n as' -'n au' -'n av' -'n ba' -'n be' -'n bi' -'n ca' -'n ce' -'n d' -'n di' -'n du' -'n e' -'n ei' -'n el' -'n en' -'n es' -'n fe' -'n f\xc3' -'n g' -'n ga' -'n ge' -'n h' -'n he' -'n i ' -'n j' -'n ka' -'n ke' -'n ku' -'n k\xc3' -'n l' -'n la' -'n lo' -'n m' -'n mu' -'n na' -'n of' -'n op' -'n pe' -'n pi' -'n po' -'n pu' -'n p\xc3' -'n q' -'n qu' -'n sc' -'n si' -'n sp' -'n ta' -'n th' -'n tu' -'n u' -'n vo' -'n w' -'n we' -'n wo' -'n y' -'n z' -'n zu' -'n \xc3\xa9' -'n-n' -'n. D' -'na c' -'na f' -'naa' -'nac' -'nach' -'nado' -'nah' -'nais' -'naj' -'nal ' -'nan ' -'nar ' -'nare' -'nari' -'nas' -'nas ' -'nast' -'nat ' -'nato' -'nau' -'nba' -'nbe' -'nc ' -'nce ' -'nci' -'ncia' -'ncie' -'ncio' -'nd ' -'nd a' -'nd c' -'nd i' -'nd p' -'nd s' -'nd t' -'nd v' -'nda ' -'ndam' -'ndan' -'ndas' -'nde ' -'ndel' -'ndes' -'ndet' -'ndh' -'ndi ' -'ndid' -'ndis' -'ndli' -'ndo ' -'ndr' -'ndra' -'ndre' -'ndri' -'ndt' -'ndus' -'nd\xc3' -'ne c' -'ne d' -'ne f' -'ne g' -'ne w' -'neb' -'ned ' -'neg' -'nega' -'nego' -'neh' -'nej' -'nek ' -'nel ' -'nell' -'nem ' -'neme' -'nen ' -'nes ' -'nest' -'nete' -'nett' -'neu' -'nez' -'ng a' -'ng d' -'ng g' -'ng k' -'ng n' -'ng t' -'ng v' -'nga' -'nga ' -'ngar' -'ngen' -'nger' -'ngst' -'nha' -'nho' -'ni d' -'nia ' -'nich' -'nici' -'nid' -'nida' -'nido' -'nie ' -'nies' -'nij' -'nik' -'nika' -'nime' -'nio ' -'nir' -'nir ' -'nis ' -'nj' -'nja' -'nje' -'njo' -'nju' -'nle' -'nly' -'nly ' -'nma' -'nmi' -'nn ' -'nna ' -'nne ' -'nnen' -'nnes' -'nnet' -'nnon' -'nns' -'nnt' -'nn\xc3\xa9' -'no ' -'no a' -'no c' -'no d' -'no e' -'no i' -'no o' -'no s' -'no, ' -'nok' -'nom ' -'non ' -'nos ' -'nost' -'not ' -'novi' -'no\xc5' -'ns c' -'ns d' -'ns l' -'ns m' -'ns p' -'nsch' -'nsk' -'nska' -'nss' -'nste' -'nt d' -'nt e' -'nt g' -'nt l' -'nt p' -'nt u' -'nt \xc3' -'ntad' -'ntas' -'nte ' -'nted' -'ntel' -'ntes' -'nti ' -'ntis' -'ntli' -'ntly' -'nto ' -'ntos' -'ntw' -'nt\xc3\xa9' -'nui' -'nung' -'nuo' -'nur' -'nus' -'nva' -'ny p' -'nya' -'nyc' -'nye' -'nym' -'nza' -'nza ' -'nze' -'nzi' -'n\xc3\xa1' -'n\xc3\xa4' -'n\xc3\xa7' -'n\xc3\xa7a' -'n\xc3\xa9' -'n\xc3\xa9 ' -'n\xc3\xad' -'n\xc3\xb3' -'n\xc3\xba' -'n\xc3\xbd' -'n\xc8' -'n\xe1' -'o a' -'o a ' -'o al' -'o at' -'o ba' -'o ch' -'o co' -'o d' -'o da' -'o de' -'o di' -'o do' -'o e' -'o e ' -'o el' -'o em' -'o en' -'o es' -'o i' -'o il' -'o in' -'o l' -'o la' -'o ne' -'o no' -'o o ' -'o pa' -'o pe' -'o po' -'o q' -'o qu' -'o ri' -'o se' -'o su' -'o th' -'o un' -'o y' -'o y ' -'o \xc3' -'oa ' -'oan' -'ob ' -'obr' -'obra' -'obre' -'obt' -'och' -'och ' -'odl' -'odo ' -'odos' -'oe ' -'oen' -'oer' -'oet' -'of ' -'of a' -'of c' -'of s' -'of t' -'og ' -'ogn' -'ogu' -'oha' -'ohe' -'ohl' -'oho' -'oht' -'oi ' -'oil' -'oim' -'oir' -'oir ' -'oire' -'ois' -'ois ' -'oit' -'oit ' -'oja' -'oji' -'ojo' -'okr' -'olan' -'olar' -'oles' -'olg' -'olge' -'olis' -'oll ' -'olm' -'olo ' -'olt' -'oly' -'ol\xc3' -'om e' -'om f' -'om n' -'om s' -'oma ' -'omas' -'ombr' -'omo ' -'ompt' -'omt' -'on E' -'on d' -'on k' -'on o' -'on t' -'on u' -'on w' -'ona ' -'onar' -'onat' -'onde' -'ondo' -'one,' -'onen' -'oner' -'ones' -'oni ' -'onie' -'onna' -'ono ' -'onos' -'ons ' -'ony' -'ony ' -'oon ' -'oor ' -'opa ' -'or d' -'or h' -'or l' -'or t' -'or. ' -'ora ' -'orb' -'ore ' -'ores' -'orh' -'ori ' -'oris' -'orno' -'orra' -'orz' -'or\xc3' -'os ' -'os a' -'os b' -'os c' -'os d' -'os e' -'os f' -'os g' -'os i' -'os l' -'os m' -'os n' -'os o' -'os p' -'os q' -'os r' -'os s' -'os t' -'os v' -'os y' -'os,' -'os, ' -'os. ' -'osa ' -'osto' -'ostr' -'ostu' -'os\xc3' -'othe' -'oto ' -'otr' -'otra' -'otre' -'otro' -'otta' -'ot\xc3' -'ou p' -'oud' -'ough' -'ould' -'out ' -'ouv' -'ouve' -'ov ' -'ovan' -'ovat' -'ovo' -'ov\xc3' -'ow ' -'ow t' -'owa' -'owi' -'own ' -'oz' -'oz ' -'oza' -'o\xc4\x8d' -'o\xc5\xbe' -'o\xe1' -'p d' -'p de' -'p e' -'p h' -'p v' -'pa ' -'paga' -'pale' -'pant' -'par ' -'pas ' -'patr' -'paz' -'pa\xc3' -'pa\xc3\xb1' -'pel' -'pene' -'pent' -'per ' -'perd' -'peu' -'pien' -'pier' -'pj' -'plus' -'po ' -'poc' -'pod' -'pode' -'pok' -'polo' -'poo' -'por ' -'pos ' -'pou' -'pour' -'pov' -'pp ' -'ppel' -'ppre' -'prav' -'pre ' -'prec' -'pred' -'prem' -'pren' -'prie' -'prim' -'pros' -'pr\xc3\xa9' -'pta' -'ptu' -'pue' -'pui' -'p\xc3\xa4' -'p\xc3\xa5' -'p\xc3\xa5 ' -'p\xc3\xa9' -'p\xc3\xb3' -'p\xc5' -'q ' -'q,' -'qa' -'qb' -'qe' -'qg' -'que ' -'qui ' -'quie' -'q\xc3' -'r at' -'r av' -'r bl' -'r d' -'r de' -'r ei' -'r el' -'r en' -'r er' -'r et' -'r fo' -'r fr' -'r f\xc3' -'r ge' -'r he' -'r i ' -'r l' -'r la' -'r le' -'r mi' -'r of' -'r p\xc3' -'r qu' -'r sa' -'r sk' -'r so' -'r th' -'r ti' -'r un' -'r vo' -'r z' -'r, d' -'r, s' -'r. D' -'ra a' -'ra c' -'ra e' -'ra l' -'ra o' -'ra q' -'raa' -'rab' -'raba' -'rabi' -'rabl' -'rach' -'raci' -'rada' -'rado' -'rait' -'ran ' -'rani' -'ran\xc3' -'rar ' -'rare' -'ras ' -'rasi' -'rat ' -'ratt' -'rau' -'rava' -'raw' -'razi' -'ra\xc3' -'rbe' -'rbei' -'rca ' -'rde ' -'rden' -'rdu' -'re d' -'re l' -'re p' -'re \xc3' -'reb' -'recu' -'red ' -'rege' -'rego' -'reh' -'rei' -'reic' -'reit' -'ren ' -'ren.' -'rend' -'rens' -'renz' -'rer ' -'ret ' -'rete' -'reto' -'rett' -'reu' -'rez' -'rf\xc3' -'rha' -'ri d' -'rias' -'rich' -'rico' -'rie ' -'rime' -'rin ' -'rio ' -'rios' -'rir ' -'ris ' -'risi' -'riu' -'rivi' -'rj' -'rken' -'rkl' -'rkt' -'rmar' -'rmet' -'rmit' -'rna ' -'rne ' -'rnes' -'rno ' -'rn\xc3' -'ro ' -'ro a' -'ro c' -'ro d' -'ro e' -'ro p' -'ro s' -'roba' -'roch' -'roe' -'roi' -'rom ' -'ron ' -'ropi' -'ros ' -'rra ' -'rre ' -'rru' -'rs d' -'rs e' -'rs p' -'rsc' -'rsch' -'rsta' -'rste' -'rs\xc3' -'rt d' -'rt w' -'rte' -'rti ' -'rtid' -'rtig' -'rto ' -'rtur' -'rt\xc3' -'rug' -'rui' -'rul' -'run' -'rund' -'rung' -'rust' -'rwe' -'rwi' -'ry ' -'ryt' -'rz' -'rza' -'rze' -'r\xc3\xa1' -'r\xc3\xa1n' -'r\xc3\xa4' -'r\xc3\xa5' -'r\xc3\xa8' -'r\xc3\xa8s' -'r\xc3\xa9' -'r\xc3\xa9 ' -'r\xc3\xa9s' -'r\xc3\xaa' -'r\xc3\xad' -'r\xc3\xada' -'r\xc3\xb3' -'r\xc3\xbc' -'r\xe1' -'s 6' -'s Be' -'s K' -'s a' -'s a ' -'s ac' -'s af' -'s am' -'s an' -'s au' -'s av' -'s d' -"s d'" -'s da' -'s de' -'s do' -'s du' -'s e' -'s e ' -'s ei' -'s el' -'s em' -'s en' -'s es' -'s et' -'s fe' -'s f\xc3' -'s ga' -'s ge' -'s i ' -'s in' -'s j' -'s le' -'s m\xc3' -'s nu' -'s o' -'s of' -'s on' -'s pa' -'s pe' -'s pi' -'s po' -'s q' -'s qu' -'s sa' -'s se' -'s su' -'s th' -'s to' -'s tu' -'s u' -'s un' -'s vi' -'s vo' -'s w' -'s wh' -'s wi' -'s y' -'s y ' -'s z' -'s \xc3\xa0' -'s \xc3\xa9' -'s \xc4' -'s. L' -'sa d' -'sa p' -'sa.' -'saa' -'sah' -'samm' -'samt' -'sar ' -'sas ' -'sau' -'sce' -'sch' -'sch ' -'scha' -'sche' -'schi' -'schl' -'schr' -'sci' -'sde ' -'se (' -'se k' -'se l' -'seb' -'sed ' -'seen' -'seg' -'segu' -'seh' -'sei' -'sein' -'sell' -'semb' -'sen ' -'sen,' -'sena' -'ser ' -'sera' -'sere' -'sess' -'setz' -'seu' -'sge' -'si a' -'si d' -'sich' -'sici' -'sie ' -'sier' -'sigu' -'sij' -'sin ' -'sind' -'sir' -'sita' -'si\xc3' -'si\xc3\xb3' -'sj' -'sjo' -'sk ' -'ska' -'ska ' -'skal' -'ske ' -'skl' -'skr' -'skri' -'smo ' -'so d' -'so p' -'so s' -'sob' -'sobr' -'soi' -'sok' -'sole' -'som' -'som ' -'some' -'sont' -'sop' -'sos ' -'sov' -'sow' -'spra' -'spre' -'spu' -'sra' -'sre' -'ssa ' -'sse ' -'ssen' -'sser' -'ssi ' -'ssie' -'sso ' -'ss\xc3' -'st k' -'st:' -'sta.' -'staa' -'stad' -'stam' -'stas' -'ste ' -'stei' -'stel' -'stes' -'stf' -'stg' -'stig' -'stk' -'sto ' -'ston' -'stos' -'stv' -'stw' -'st\xc3\xa1' -'st\xc3\xa4' -'st\xc4' -'su ' -'su p' -'sud' -'sun' -'suo' -'sur ' -'sus ' -'sut' -'sve' -'sz' -'sza' -'sze' -'s\xc3\xa1' -'s\xc3\xa4' -'s\xc3\xa4t' -'s\xc3\xa5' -'s\xc3\xa9' -'s\xc3\xad' -'s\xc3\xb3' -'s\xc3\xb6' -'s\xe1' -'t af' -'t at' -'t av' -'t d' -'t de' -'t e' -'t ei' -'t en' -'t er' -'t et' -'t fe' -'t f\xc3' -'t g' -'t ge' -'t he' -'t i ' -'t im' -'t is' -'t le' -'t mi' -'t of' -'t op' -'t pa' -'t pe' -'t p\xc3' -'t q' -'t qu' -'t se' -'t th' -'t ti' -'t to' -'t tr' -'t un' -'t va' -'t vo' -'t w' -'t wo' -'t z' -'t \xc3\xa0' -'t \xc3\xa9' -'t. A' -'ta d' -'ta f' -"ta'" -'ta. ' -'taa' -'taat' -'taci' -'tado' -'tage' -'tair' -'tali' -'tama' -'tamb' -'tami' -'tan ' -'tana' -'tani' -'tar ' -'tas ' -'tato' -'tava' -'taz' -'ta\xc5' -'tbe' -'te B' -'te c' -'te d' -'te e' -'te l' -'te s' -'te \xc3' -'teen' -'teil' -'tel ' -'tels' -'ten ' -'ten.' -'teni' -'ter\xc3' -'tet ' -'tete' -'teu' -'teur' -'tev' -'tge' -'th a' -'th t' -'tha' -'that' -'the ' -'thei' -'ther' -'they' -'thin' -'thou' -'th\xc3' -'ti d' -'ti e' -'ti i' -'tici' -'tico' -'tid' -'tida' -'tido' -'tie ' -'tiem' -'tier' -'tig ' -'tige' -'tik ' -'til ' -'till' -'tin ' -'tino' -'tiq' -'tiqu' -'tir ' -'tis ' -'tisc' -'tivo' -'tiz' -'tke' -'tki' -'tlic' -'tly' -'tly ' -'tmi' -'tnin' -'tno' -'tn\xc3' -'to ' -'to b' -'to d' -'to i' -'to t' -'to. ' -'toa' -'todo' -'toe' -'toi' -'torn' -'tos ' -'tout' -'tow' -'tq' -'tra ' -'trag' -'tras' -'tre ' -'treb' -'tren' -'tres' -'tret' -'tro ' -'tros' -'ts ' -'ts a' -'ts d' -'ts e' -'ts o' -'tsc' -'tsch' -'tsi' -'tso' -'tst' -'tsu' -'ts\xc3' -'tt ' -'tt.' -'tta' -'tta ' -'tte ' -'ttes' -'tti ' -'tto ' -'tts' -'ttu' -'tty' -'tt\xc3' -'tuk' -'tul' -'tun' -'tung' -'tuo' -'tup' -'tur ' -'tuv' -'tvo' -'tv\xc3' -'twa' -'ty ' -'tyk' -'tym' -'tys' -'tyt' -'tz' -'tze' -'tzen' -'tzt' -'tzu' -'t\xc3\xa0' -'t\xc3\xa0 ' -'t\xc3\xa1' -'t\xc3\xa1 ' -'t\xc3\xa4' -'t\xc3\xa9' -'t\xc3\xa9 ' -'t\xc3\xa9s' -'t\xc3\xad' -'t\xc3\xb3' -'t\xc3\xb3r' -'t\xc3\xb6' -'t\xc3\xbc' -'t\xe1' -'u C' -'u P' -'u a ' -'u al' -'u be' -'u c' -'u co' -'u de' -'u f' -'u fi' -'u g' -'u li' -'u ma' -'u mi' -'u no' -'u ta' -'u un' -'ua ' -'uai' -'uale' -'uand' -'uar ' -'uas' -'ub ' -'ubbl' -'ubo' -'ud ' -'udg' -'ue a' -'ue d' -'ue e' -'ue n' -'ue o' -'ue p' -'ue s' -'ued' -'uen ' -'uev' -'uf' -'uf ' -'uf D' -'uf d' -'uge' -'ugh' -'ugh ' -'ugl' -'ugo' -'uh' -'uha' -'ui ' -'ui p' -'ui s' -'uie' -'uien' -'uill' -'uin' -'uit ' -'uje ' -'ujo' -'uk ' -'uka' -'uke' -'uks' -'uku' -'ula ' -'uld' -'uld ' -'ulla' -'ulte' -'ulu' -'um ' -'um d' -'uma ' -'umu' -'un ' -'un a' -'un b' -'un c' -'un d' -'un e' -'un i' -'un p' -'un t' -'una' -'una ' -'und ' -'undo' -'une ' -'ung' -'ung ' -'unge' -'ungs' -'unn' -'unne' -'uno' -'uno ' -'uns' -'unto' -'uo ' -'uol' -'uom' -'uos' -'upo' -'upu' -'uq' -'ur d' -'ur l' -'uri ' -'urie' -'urm' -'urs ' -'uru' -'ur\xc3' -'us d' -'us i' -'us n' -'us o' -'us p' -'us r' -'us. ' -'uses' -'usg' -'usl' -'ust ' -'uste' -'ut,' -'uta ' -'utan' -'utar' -'utat' -'ute ' -'utin' -'utor' -'utr' -'utre' -'uts' -'utsc' -'utz' -'uu' -'uur' -'uva' -'uve' -'uver' -'uvo' -'uw' -'uwa' -'ux' -'ux ' -'uze' -'uzi' -'u\xc3\x9f' -'u\xc3\xa9' -'u\xc3\xad' -'u\xc5\xa1' -'u\xc5\xbe' -'u\xe1' -'v d' -'v e' -'v f' -'v. ' -'vad' -'vag' -'vah' -'vall' -'van ' -'vare' -'vast' -'vea' -'veau' -'vec' -'vec ' -'ved ' -'vell' -'vend' -'vens' -'verb' -'verg' -'verk' -'verl' -'verw' -'vez' -'vid ' -'vida' -'vien' -'vii' -'vik' -'vil ' -'vim' -'ving' -'virt' -'vis ' -'visa' -'vise' -'viso' -'viz' -'vi\xc3' -'vj' -'vk' -'vne' -'vod' -'voi' -'voir' -'vom' -'vom ' -'von' -'von ' -'voo' -'vor ' -'vr' -'vre' -'vri' -'vro' -'vv' -'vy' -'vy ' -'vz' -'v\xc3\xa1' -'v\xc3\xa4' -'v\xc3\xa9' -'v\xc3\xa9 ' -'v\xc3\xad' -'v\xc3\xbd' -'v\xc4' -'v\xc5' -'v\xe1' -'w c' -'w d' -'w l' -'w o' -'w p' -'w r' -'w s' -'w w' -'wa ' -'wac' -'wan' -'war ' -'wart' -'was' -'was ' -'wc' -'weg' -'weit' -'werd' -'whe' -'when' -'wher' -'whic' -'who' -'who ' -'wie' -'wie ' -'wing' -'wir' -'wit' -'with' -'wk' -'wn ' -'wol' -'wro' -'wu' -'wur' -'wurd' -'wy' -'wz' -'w\xc3' -'w\xc3\xa4' -'w\xc4' -'w\xc5' -'x e' -'x i' -'x m' -'x t' -'xa ' -'xer' -'xx' -'x\xc3' -'x\xc4' -'y a' -'y an' -'y do' -'y el' -'y f' -'y in' -'y j' -'y na' -'y o' -'y of' -'y po' -'y t' -'y th' -'y v' -'y w' -'y z' -'ya ' -'yan' -'yar' -'ybo' -'ych' -'yda' -'yde' -'ye ' -'yel' -'yen' -'yh' -'ying' -'yj' -'yk' -'yla' -'ym ' -'yma' -'ymi' -'ymo' -'yn ' -'yor' -'ypa' -'yra' -'yta' -'yto' -'ytt' -'yy' -'y\xc3' -'y\xc4' -'y\xc5' -'y\xe1' -'z ' -'z a' -'z d' -'z e' -'z i' -'z p' -'z v' -'z. ' -'za d' -'zac' -'zaci' -'zad' -'zado' -'zah' -'zaj' -'zan' -'zap' -'zar' -'zar ' -'zas' -'zb' -'zc' -'ze,' -'zec' -'zeg' -'zei' -'zek' -'zel' -'zen' -'zen ' -'zent' -'zer' -'zes' -'zet' -'zg' -'zh' -'zia' -'zic' -'zie' -'zie ' -'zig' -'zij' -'zio' -'zion' -'zj' -'zl' -'zle' -'zli' -'zm' -'zn' -'zna' -'zne' -'zni' -'zn\xc3' -'zo ' -'zol' -'zos' -'zp' -'zr' -'zt' -'zt ' -'zte' -'zu' -'zu ' -'zum' -'zum ' -'zun' -'zur' -'zur ' -'zw' -'zwi' -'zy' -'zy ' -'zz' -'zza' -'z\xc3' -'z\xc3\xa1' -'z\xc3\xb3' -'z\xc4' -'{K' -'\x80(' -'\x80-' -'\x80.' -'\x80\x81' -'\x80\x81\xe3' -'\x80\x81\xe5' -'\x80\x82' -'\x80\x8f' -'\x80\xa2' -'\x80\xce' -'\x80\xcf' -'\x80\xd0\xb5' -'\x80\xd0\xb5\xd0' -'\x80\xd0\xb5\xd1' -'\x80\xd0\xb6' -'\x80\xd0\xbe' -'\x80\xd0\xbe\xd0' -'\x80\xd0\xbe\xd1' -'\x80\xd1\x81' -'\x80\xd1\x83' -'\x80\xd1\x8f' -'\x80\xe2' -'\x80\xe3' -'\x80\xe3\x83' -'\x80\xea' -'\x80\xeb' -'\x80\xec' -'\x80\xed' -'\x81 ' -'\x81 \xd0' -'\x81 \xd1' -'\x81,' -'\x81.' -'\x81G' -'\x81a' -'\x81c' -'\x81d' -'\x81i' -'\x81j' -'\x81k' -'\x81l' -'\x81m' -'\x81n' -'\x81p' -'\x81r' -'\x81s' -'\x81t' -'\x81u' -'\x81v' -'\x81\x82' -'\x81\x82\xe3' -'\x81\x84' -'\x81\x84\xe3' -'\x81\x86' -'\x81\x86\xe3' -'\x81\x8b' -'\x81\x8b\xe3' -'\x81\x8c' -'\x81\x8c\xe3' -'\x81\x8d' -'\x81\x8f' -'\x81\x91' -'\x81\x93' -'\x81\x93\xe3' -'\x81\x95' -'\x81\x95\xe3' -'\x81\x97' -'\x81\x97\xe3' -'\x81\x97\xe3\x81' -'\x81\x99' -'\x81\x99\xe3' -'\x81\x9f' -'\x81\x9f\xe3' -'\x81\x9f\xe3\x80' -'\x81\xa0' -'\x81\xa0\xe3' -'\x81\xa3' -'\x81\xa3\xe3' -'\x81\xa6' -'\x81\xa6\xe3' -'\x81\xa6\xe3\x81' -'\x81\xa7' -'\x81\xa7\xe3' -'\x81\xa8' -'\x81\xa8\xe3' -'\x81\xa8\xe3\x81' -'\x81\xaa' -'\x81\xaa\xe3' -'\x81\xab' -'\x81\xab\xe3' -'\x81\xab\xe5' -'\x81\xac' -'\x81\xae' -'\x81\xae\xe3' -'\x81\xae\xe4' -'\x81\xae\xe5' -'\x81\xae\xe6' -'\x81\xae\xe7' -'\x81\xae\xe8' -'\x81\xae\xe9' -'\x81\xaf' -'\x81\xaf\xe3' -'\x81\xaf\xe3\x80' -'\x81\xbe' -'\x81\xbe\xe3' -'\x81\xc4' -'\x81\xc5' -'\x81\xce' -'\x81\xcf' -'\x81\xd0' -'\x81\xd0\xb5' -'\x81\xd0\xb5\xd0' -'\x81\xd0\xb8' -'\x81\xd0\xb8\xd1' -'\x81\xd0\xba' -'\x81\xd0\xba\xd0' -'\x81\xd0\xbb\xd0' -'\x81\xd0\xbd' -'\x81\xd0\xbd\xd0' -'\x81\xd0\xbe' -'\x81\xd0\xbe\xd0' -'\x81\xd1' -'\x81\xd1\x82' -'\x81\xd1\x82\xd0' -'\x81\xd1\x82\xd1' -'\x81\xd1\x83' -'\x81\xd1\x8f' -'\x81\xd8' -'\x81\xd9' -'\x81\xe2' -'\x81\xe3' -'\x81\xe3\x81' -'\x81\xe3\x82' -'\x81\xeb' -'\x81\xec' -'\x81\xed' -'\x82 ' -'\x82 (' -'\x82 \xd0' -'\x82 \xd1' -'\x82(' -'\x82)' -'\x82,' -'\x82, ' -'\x82.' -'\x82:' -'\x82a' -'\x82c' -'\x82e' -'\x82k' -'\x82n' -'\x82o' -'\x82u' -'\x82y' -'\x82\x81' -'\x82\x82' -'\x82\x88' -'\x82\x88\xe3' -'\x82\x89' -'\x82\x89\xe3' -'\x82\x8a' -'\x82\x8a\xe3' -'\x82\x8b' -'\x82\x8b\xe3' -'\x82\x8b\xe3\x81' -'\x82\x8c' -'\x82\x8c\xe3' -'\x82\x92' -'\x82\x92\xe3' -'\x82\x92\xe5' -'\x82\x98' -'\x82\xa4' -'\x82\xa4\xe3' -'\x82\xa4\xe3\x82' -'\x82\xac' -'\x82\xaf' -'\x82\xaf\xe3' -'\x82\xb3' -'\x82\xb7' -'\x82\xb9' -'\x82\xb9\xe3' -'\x82\xbf' -'\x82\xbf\xe3' -'\x82\xbf\xe3\x83' -'\x82\xc3' -'\x82\xc4' -'\x82\xc5' -'\x82\xd0\xb0' -'\x82\xd0\xb0 ' -'\x82\xd0\xb0\xd0' -'\x82\xd0\xb2' -'\x82\xd0\xb2\xd0' -'\x82\xd0\xb5' -'\x82\xd0\xb5 ' -'\x82\xd0\xb5\xd0' -'\x82\xd0\xb5\xd1' -'\x82\xd0\xb8' -'\x82\xd0\xb8 ' -'\x82\xd0\xb8\xd1' -'\x82\xd0\xbd' -'\x82\xd0\xbd\xd0' -'\x82\xd0\xbe' -'\x82\xd0\xbe ' -'\x82\xd0\xbe\xd0' -'\x82\xd1' -'\x82\xd1\x80' -'\x82\xd1\x80\xd0' -'\x82\xd1\x81' -'\x82\xd1\x83' -'\x82\xd1\x8c' -'\x82\xd8' -'\x82\xd9' -'\x82\xe3' -'\x82\xe3\x81' -'\x82\xe3\x82' -'\x82\xe7' -'\x83 ' -'\x83 \xd0' -'\x83"' -'\x83)' -'\x83,' -'\x83-' -'\x83.' -'\x83:' -'\x83;' -'\x83O' -'\x83c' -'\x83i' -'\x83l' -'\x83m' -'\x83n' -'\x83r' -'\x83s' -'\x83t' -'\x83u' -'\x83z' -'\x83\x83' -'\x83\x88' -'\x83\x88\xe3' -'\x83\x89' -'\x83\x89\xe3' -'\x83\x8b' -'\x83\x8b\xe3' -'\x83\x8b\xe3\x83' -'\x83\x92' -'\x83\xa4' -'\x83\xa5' -'\x83\xa5\xe3' -'\x83\xa5\xe3\x83' -'\x83\xa6' -'\x83\xab' -'\x83\xab\xe3' -'\x83\xad' -'\x83\xad\xe3' -'\x83\xad\xe3\x82' -'\x83\xb3\xe3' -'\x83\xbc' -'\x83\xbc\xe3' -'\x83\xbc\xe3\x82' -'\x83\xbc\xe3\x83' -'\x83\xc5' -'\x83\xc8' -'\x83\xce' -'\x83\xcf' -'\x83\xd0\xbf' -'\x83\xd1' -'\x83\xd1\x81' -'\x83\xd1\x81\xd1' -'\x83\xd1\x82' -'\x83\xd8' -'\x83\xd9' -'\x83\xe3' -'\x83\xe5' -'\x83\xec' -'\x84 ' -"\x84'" -'\x84.' -'\x84:' -'\x84C' -'\x84D' -'\x84E' -'\x84M' -'\x84N' -'\x84c' -'\x84s' -'\x84\x9c' -'\x84\xa0' -'\x84\xa4' -'\x84\xb0' -'\x84\xb1' -'\x84\xb8' -'\x84\xc3' -'\x84\xce' -'\x84\xcf' -'\x84\xd0\xb8' -'\x84\xd1' -'\x84\xd9' -'\x84\xdb' -'\x84\xe0' -'\x84\xe2' -'\x84\xe3' -'\x84\xe3\x81' -'\x84\xe3\x82' -'\x84\xe6' -'\x84\xe7' -'\x84\xe8' -'\x84\xea' -'\x84\xeb' -'\x84\xec' -'\x84\xed' -'\x85 ' -'\x85 \xd0' -'\x85, ' -'\x85.' -'\x85;' -'\x85c' -'\x85d' -'\x85g' -'\x85j' -'\x85l' -'\x85p' -'\x85r' -'\x85s' -'\x85t' -'\x85z' -'\x85\xb6' -'\x85\xc4' -'\x85\xc5' -'\x85\xce' -'\x85\xcf' -'\x85\xd1' -'\x85\xd8' -'\x85\xd9' -'\x85\xdb' -'\x85\xe3' -'\x85\xeb' -'\x85\xec' -'\x86' -'\x86 ' -'\x86,' -'\x86.' -'\x86I' -'\x86L' -'\x86e' -'\x86\x8c' -'\x86\xa0' -'\x86\xc4' -'\x86\xce' -'\x86\xcf' -'\x86\xd0\xb5' -'\x86\xd0\xb8' -'\x86\xd0\xb8\xd0' -'\x86\xd1' -'\x86\xd9' -'\x86\xda' -'\x86\xdb' -'\x86\xe3' -'\x86\xe3\x81' -'\x87 ' -'\x87,' -'\x87a' -'\x87e' -'\x87i' -'\x87n' -'\x87u' -'\x87\xce' -'\x87\xcf' -'\x87\xd0' -'\x87\xd0\xb5' -'\x87\xd0\xb5\xd0' -'\x87\xd0\xb8' -'\x87\xd0\xbd' -'\x87\xd8' -'\x87\xd9' -'\x87\xe2' -'\x87\xe3' -'\x87\xe6' -'\x88,' -'\x88.' -'\x88a' -'\x88o' -'\x88u' -'\x88\x88' -'\x88\x98' -'\x88\xb0' -'\x88\xce' -'\x88\xcf' -'\x88\xd0\xb5' -'\x88\xdb' -'\x88\xe3' -'\x88\xe3\x81' -'\x88\xe3\x82' -'\x88\xea' -'\x88\xeb' -'\x88\xec' -'\x88\xed' -'\x89' -'\x89 ' -'\x89,' -'\x89.' -'\x89G' -'\x89K' -'\x89s' -'\x89\xb5' -'\x89\xce' -'\x89\xcf' -'\x89\xd0' -'\x89\xd0\xb5' -'\x89\xd0\xb8' -'\x89\xe3' -'\x89\xe3\x81' -'\x89\xe3\x82' -'\x89\xec' -'\x8a' -'\x8a ' -'\x8a.' -'\x8a\x94' -'\x8a\xa4' -'\x8a\xb5' -'\x8a\xb8' -'\x8a\xce' -'\x8a\xcf' -'\x8a\xd0' -'\x8a\xd1' -'\x8a\xd9' -'\x8a\xe3' -'\x8a\xe3\x81' -'\x8b)' -'\x8b-' -'\x8ba' -'\x8be' -'\x8bi' -'\x8bj' -'\x8bn' -'\x8b\x88' -'\x8b\x9c' -'\x8b\xa0' -'\x8b\xa4' -'\x8b\xa8' -'\x8b\xad' -'\x8b\xb0' -'\x8b\xc4' -'\x8b\xd0' -'\x8b\xd1' -'\x8b\xe3' -'\x8b\xe3\x80' -'\x8b\xe3\x81' -'\x8b\xe3\x82' -'\x8b\xe3\x83' -'\x8c' -'\x8c ' -'\x8c \xd0' -'\x8c)' -'\x8c,' -'\x8c-' -'\x8c.' -'\x8ce' -'\x8c\x80' -'\x8c\x8c' -'\x8c\xc3' -'\x8c\xce' -'\x8c\xcf' -'\x8c\xd0' -'\x8c\xd0\xbd' -'\x8c\xd1' -'\x8c\xd8' -'\x8c\xd9' -'\x8c\xda' -'\x8c\xe2' -'\x8c\xe3' -'\x8c\xe3\x81' -'\x8c\xe3\x82' -'\x8c\xea' -'\x8c\xeb' -'\x8c\xec' -'\x8c\xed' -'\x8d' -'\x8d ' -'\x8d,' -'\x8d:' -'\x8dZ' -'\x8da' -'\x8dc' -'\x8de' -'\x8den' -'\x8dj' -'\x8dn' -'\x8do' -'\x8dr' -'\x8dt' -'\x8du' -'\x8d\xb0' -'\x8d\xc3' -'\x8d\xc5' -'\x8d\xce' -'\x8d\xcf' -'\x8d\xe2' -'\x8d\xe3' -'\x8d\xe3\x81' -'\x8e' -'\x8e ' -'\x8e,' -'\x8e.' -'\x8en' -'\x8e\x98' -'\x8e\xce' -'\x8e\xcf' -'\x8e\xd0' -'\x8e\xd1' -'\x8e\xe0' -'\x8e\xe5' -'\x8f \xd0' -'\x8f \xd1' -'\x8f,' -'\x8f.' -'\x8f:' -'\x8fa' -'\x8f\x84' -'\x8f\x99' -'\x8f\xac' -'\x8f\xc5' -'\x8f\xd0' -'\x8f\xd0\xbd' -'\x8f\xd1' -'\x8f\xd1\x82' -'\x8f\xe3' -'\x8f\xe3\x81' -'\x8f\xe7' -'\x90 ' -'\x90)' -'\x90,' -'\x90.' -'\x90:' -'\x90\x98' -'\x90\xce' -'\x90\xe1' -'\x90\xea' -'\x90\xeb' -'\x90\xec' -'\x90\xed' -'\x91' -'\x91 ' -'\x91)' -'\x91,' -'\x91-' -'\x91.' -'\x91a' -'\x91b' -'\x91d' -'\x91e' -'\x91f' -'\x91i' -'\x91k' -'\x91l' -'\x91n' -'\x91r' -'\x91s' -'\x91t' -'\x91v' -'\x91z' -'\x91\x9c' -'\x91\xc3' -'\x91\xc6' -'\x91\xce' -'\x91\xcf' -'\x91\xd7' -'\x91\xe1' -'\x91\xe3' -'\x91\xec' -'\x92' -'\x92 ' -'\x92\xce' -'\x92\xcf' -'\x92\xd7' -'\x92\xe1' -'\x92\xe3' -'\x92\xe5' -'\x92\xe6' -'\x93,' -'\x93.' -'\x93L' -'\x93P' -'\x93c' -'\x93d' -'\x93j' -'\x93k' -'\x93l' -'\x93m' -'\x93n' -'\x93r' -'\x93s' -'\x93t' -'\x93\x9c' -'\x93\xa4' -'\x93\xb1' -'\x93\xc5' -'\x93\xce' -'\x93\xd7' -'\x93\xe1' -'\x93\xe3' -'\x93\xe3\x81' -'\x94"' -'\x94)' -'\x94,' -'\x94.' -'\x94:' -'\x94\x84' -'\x94\x94' -'\x94\xb6' -'\x94\xce' -'\x94\xd0\xbe' -'\x94\xd1' -'\x94\xd7' -'\x94\xe0' -'\x94\xea' -'\x94\xeb' -'\x94\xec' -'\x94\xed' -'\x95' -'\x95 ' -'\x95,' -'\x95-' -'\x95.' -'\x95T' -'\x95i' -'\x95\x84' -'\x95\x88' -'\x95\x8a' -'\x95\x98' -'\x95\x9c' -'\x95\xa0' -'\x95\xa9' -'\x95\xb4' -'\x95\xbc' -'\x95\xce' -'\x95\xcf' -'\x95\xd0' -'\x95\xe1' -'\x95\xe3' -'\x95\xe3\x81' -'\x95\xe3\x82' -'\x95\xeb' -'\x95\xec' -'\x95\xed' -'\x96 ' -'\x96,' -'\x96.' -'\x96:' -'\x96J' -'\x96L' -'\x96M' -'\x96R' -'\x96S' -'\x96Z' -'\x96v' -'\x96\x88' -'\x96\x89' -'\x96\xb0' -'\x96\xb4' -'\x96\xd0' -'\x96\xd1' -'\x96\xd7' -'\x96\xe3' -'\x97' -'\x97 ' -'\x97,' -'\x97.' -'\x97j' -'\x97l' -'\x97m' -'\x97n' -'\x97r' -'\x97s' -'\x97t' -'\x97\x86' -'\x97\x88' -'\x97\x90' -'\x97\xa5\xe3' -'\x97\xac' -'\x97\xb0' -'\x97\xbb' -'\x97\xc5' -'\x97\xce' -'\x97\xd0' -'\x97\xd0\xb0' -'\x97\xd1' -'\x97\xd7' -'\x97\xe3' -'\x97\xe3\x81' -'\x97\xe3\x81\xa6' -'\x98 ' -'\x98,' -'\x98.' -'\x98B' -'\x98F' -'\x98I' -'\x98L' -'\x98\x81' -'\x98\x84' -'\x98\xa4' -'\x98\xaf' -'\x98\xb8' -'\x98\xc3' -'\x98\xce' -'\x98\xd0' -'\x98\xd1' -'\x98\xd7' -'\x98\xe3' -'\x98\xea' -'\x98\xeb' -'\x98\xec' -'\x98\xed' -'\x99 ' -'\x99,' -'\x99.' -'\x99b' -'\x99c' -'\x99d' -'\x99e' -'\x99g' -'\x99h' -'\x99i' -'\x99k' -'\x99m' -'\x99n' -'\x99p' -'\x99r' -'\x99t' -'\x99z' -'\x99\x80' -'\x99\x94' -'\x99\x95' -'\x99\xc3' -'\x99\xc5' -'\x99\xce' -'\x99\xcf' -'\x99\xd0' -'\x99\xe1' -'\x99\xe3' -'\x99\xe3\x82' -'\x99\xeb' -'\x99\xec' -'\x9a' -'\x9a ' -'\x9a)' -'\x9a,' -'\x9a.' -'\x9ar' -'\x9a\x84' -'\x9a\x94' -'\x9a\xa9' -'\x9a\xb0' -'\x9a\xb4' -'\x9a\xc5' -'\x9a\xce' -'\x9a\xcf' -'\x9a\xd0\xbe' -'\x9a\xe3' -'\x9a\xe3\x81' -'\x9b' -'\x9b ' -'\x9b,' -'\x9b.' -'\x9bc' -'\x9bd' -'\x9be' -'\x9bh' -'\x9bi' -'\x9bj' -'\x9bk' -'\x9bl' -'\x9bm' -'\x9bn' -'\x9bp' -'\x9br' -'\x9bs' -'\x9bt' -'\x9bw' -'\x9b\x90' -'\x9b\xc4' -'\x9b\xc5' -'\x9b\xce' -'\x9b\xd1' -'\x9b\xd7' -'\x9c)' -'\x9c,' -'\x9c-' -'\x9c.' -'\x9c:' -'\x9cH' -'\x9cT' -'\x9ch' -'\x9cl' -'\x9c\x84' -'\x9c\x89' -'\x9c\x8b' -'\x9c\xa0' -'\x9c\xa8' -'\x9c\xac' -'\x9c\xbc' -'\x9c\xce' -'\x9c\xd1' -'\x9c\xea' -'\x9c\xeb' -'\x9c\xec' -'\x9c\xed' -'\x9d ' -'\x9d,' -'\x9d.' -'\x9d:' -'\x9dC' -'\x9di' -'\x9dn' -'\x9d\x80' -'\x9d\x84' -'\x9d\x8c' -'\x9d\x98' -'\x9d\xb4' -'\x9d\xb8' -'\x9d\xbc' -'\x9d\xce' -'\x9d\xd0' -'\x9d\xd0\xb0' -'\x9d\xd1' -'\x9d\xe3' -'\x9d\xeb' -'\x9d\xec' -'\x9d\xed' -'\x9e ' -'\x9e-' -'\x9e.' -'\x9ep' -'\x9e\x85' -'\x9e\x88' -'\x9e\x90' -'\x9e\x91' -'\x9e\x98' -'\x9e\xa5' -'\x9e\xac' -'\x9e\xce' -'\x9e\xd0' -'\x9e\xe3' -'\x9e\xe3\x83' -'\x9f ' -'\x9f,' -'\x9f.' -'\x9f:' -'\x9fa' -'\x9fe' -'\x9fi' -'\x9fk' -'\x9fl' -'\x9fm' -'\x9ft' -'\x9fu' -'\x9f\xac' -'\x9f\xc4' -'\x9f\xce' -'\x9f\xcf' -'\x9f\xe3' -'\x9f\xe3\x80' -'\x9f\xe3\x80\x82' -'\x9f\xe3\x81' -'\xa0 ' -'\xa0 c' -'\xa0 d' -'\xa0 l' -'\xa0 la' -'\xa0 t' -'\xa0,' -'\xa0.' -'\xa0U' -'\xa0a' -'\xa0c' -'\xa0e' -'\xa0i' -'\xa0l' -'\xa0m' -'\xa0n' -'\xa0o' -'\xa0r' -'\xa0s' -'\xa0t' -'\xa0u' -'\xa0y' -'\xa0\x80' -'\xa0\x84' -'\xa0\x88' -'\xa0\x95' -'\xa0\x9c' -'\xa0\xa4' -'\xa0\xa5' -'\xa0\xc4' -'\xa0\xce' -'\xa0\xcf' -'\xa0\xd0' -'\xa0\xd0\xb5' -'\xa0\xe3' -'\xa0\xe3\x81' -'\xa0\xea' -'\xa0\xeb' -'\xa0\xec' -'\xa0\xed' -'\xa1' -'\xa1b' -'\xa1c' -'\xa1ci' -'\xa1d' -'\xa1g' -'\xa1i ' -'\xa1j' -'\xa1n' -'\xa1n ' -'\xa1r' -'\xa1ri' -'\xa1s' -'\xa1s ' -'\xa1ti' -'\xa1u' -'\xa1v' -'\xa1x' -'\xa1y' -'\xa1z' -'\xa1\x9c' -'\xa1\x9d' -'\xa1\xa3' -'\xa1\xb6' -'\xa1\xc3' -'\xa1\xc5' -'\xa1\xce' -'\xa1\xd0' -'\xa1\xd1' -'\xa1\xd7' -'\xa1\xe3' -'\xa2' -'\xa2 ' -'\xa2,' -'\xa2.' -'\xa2I' -'\xa2m' -'\xa2n' -'\xa2r' -'\xa2t' -'\xa2y' -'\xa2\xb2' -'\xa2\xd0' -'\xa2\xd1' -'\xa2\xd7' -'\xa2\xd8' -'\xa2\xd9' -'\xa2\xe1' -'\xa3' -'\xa3 ' -'\xa3)' -'\xa3,' -'\xa3.' -'\xa3a' -'\xa3c' -'\xa3e' -'\xa3i' -'\xa3n' -'\xa3o' -'\xa3o ' -'\xa3\xa1' -'\xa3\xa9' -'\xa3\xac' -'\xa3\xba' -'\xa3\xbc' -'\xa3\xc4' -'\xa3\xce' -'\xa3\xcf' -'\xa3\xd0' -'\xa3\xd1' -'\xa3\xd8' -'\xa3\xd9' -'\xa3\xe1' -'\xa3\xe3' -'\xa3\xe3\x81' -'\xa4' -'\xa4 ' -'\xa4,' -'\xa4-' -'\xa4.' -'\xa4a' -'\xa4b' -'\xa4ch' -'\xa4e' -'\xa4g' -'\xa4h' -'\xa4hr' -'\xa4i' -'\xa4is' -'\xa4j' -'\xa4k' -'\xa4l' -'\xa4ll' -'\xa4lt' -'\xa4m' -'\xa4n' -'\xa4nd' -'\xa4ng' -'\xa4o' -'\xa4p' -'\xa4r' -'\xa4s' -'\xa4si' -'\xa4t' -'\xa4te' -'\xa4tt' -'\xa4v' -'\xa4x' -'\xa4y' -'\xa4z' -'\xa4\x81' -'\xa4\x82' -'\xa4\x85' -'\xa4\x86' -'\xa4\x87' -'\xa4\x89' -'\xa4\x8f' -'\xa4\x91' -'\xa4\x95' -'\xa4\x97' -'\xa4\x9c' -'\xa4\x9f' -'\xa4\xa2' -'\xa4\xa4' -'\xa4\xa5' -'\xa4\xa6' -'\xa4\xaa' -'\xa4\xab' -'\xa4\xac' -'\xa4\xad' -'\xa4\xae' -'\xa4\xaf' -'\xa4\xb6' -'\xa4\xb9' -'\xa4\xbc' -'\xa4\xc3' -'\xa4\xc3\xa4' -'\xa4\xc4' -'\xa4\xc5' -'\xa4\xce' -'\xa4\xcf' -'\xa4\xd0' -'\xa4\xd1' -'\xa4\xd9' -'\xa4\xe1' -'\xa4\xe3' -'\xa4\xe3\x82' -'\xa4\xeb' -'\xa4\xec' -'\xa4\xed' -'\xa5' -'\xa5 ' -'\xa5 d' -'\xa5 s' -'\xa5,' -'\xa5.' -'\xa5a' -'\xa5b' -'\xa5c' -'\xa5d' -'\xa5de' -'\xa5e' -'\xa5g' -'\xa5k' -'\xa5l' -'\xa5n' -'\xa5o' -'\xa5p' -'\xa5r' -'\xa5r ' -'\xa5s' -'\xa5t' -'\xa5u' -'\xa5v' -'\xa5\x80' -'\xa5\x81' -'\xa5\x82' -'\xa5\x87' -'\xa5\x88' -'\xa5\x89' -'\xa5\x8b' -'\xa5\x98' -'\xa5\xa7' -'\xa5\xbc' -'\xa5\xce' -'\xa5\xcf' -'\xa5\xd8' -'\xa5\xd9' -'\xa5\xe1' -'\xa5\xe3' -'\xa5\xe3\x83' -'\xa5\xe3\x83\xbc' -'\xa5\xec' -'\xa6' -'\xa6d' -'\xa6f' -'\xa6g' -'\xa6i' -'\xa6k' -'\xa6l' -'\xa6n' -'\xa6r' -'\xa6s' -'\xa6t' -'\xa6v' -'\xa6\x84' -'\xa6\x86' -'\xa6\x87' -'\xa6\x89' -'\xa6\x8f' -'\xa6\x95' -'\xa6\x9a' -'\xa6\xa4' -'\xa6\xa5' -'\xa6\xa8' -'\xa6\xaa' -'\xa6\xad' -'\xa6\xae' -'\xa6\xaf' -'\xa6\xb2' -'\xa6\xb6' -'\xa6\xb7' -'\xa6\xb9' -'\xa6\xbf' -'\xa6\xce' -'\xa6\xcf' -'\xa6\xd0' -'\xa6\xd1' -'\xa6\xd7' -'\xa6\xd8' -'\xa6\xd9' -'\xa6\xe3' -'\xa6\xe3\x81' -'\xa7' -'\xa7 ' -'\xa7)' -'\xa7,' -'\xa7.' -'\xa7a' -'\xa7a ' -'\xa7ai' -'\xa7ais' -'\xa7d' -'\xa7e' -'\xa7h' -'\xa7i' -'\xa7l' -'\xa7m' -'\xa7n' -'\xa7o' -'\xa7r' -'\xa7s' -'\xa7t' -'\xa7u' -'\xa7\x80' -'\xa7\x82' -'\xa7\x84' -'\xa7\x87' -'\xa7\x88' -'\xa7\x8c' -'\xa7\xc3' -'\xa7\xc3\xa3' -'\xa7\xc4' -'\xa7\xce' -'\xa7\xcf' -'\xa7\xd0' -'\xa7\xd7' -'\xa7\xda' -'\xa7\xdb' -'\xa7\xe3' -'\xa7\xe3\x81' -'\xa8' -'\xa8 ' -'\xa8,' -'\xa8.' -'\xa8d' -'\xa8l' -'\xa8m' -'\xa8n' -'\xa8r' -'\xa8re' -'\xa8re ' -'\xa8s' -'\xa8s ' -'\xa8t' -'\xa8\x82' -'\xa8\x93' -'\xa8\xa4' -'\xa8\xaa' -'\xa8\xad' -'\xa8\xae' -'\xa8\xb0' -'\xa8\xb2' -'\xa8\xbe' -'\xa8\xbf' -'\xa8\xd0' -'\xa8\xd8' -'\xa8\xd9' -'\xa8\xdb' -'\xa8\xe3' -'\xa8\xe3\x81' -'\xa8\xea' -'\xa8\xeb' -'\xa8\xec' -'\xa8\xed' -'\xa9' -'\xa9 ' -'\xa9 a' -'\xa9 d' -'\xa9 de' -'\xa9 l' -'\xa9 m' -'\xa9 n' -'\xa9 p' -'\xa9 s' -'\xa9 t' -'\xa9 v' -'\xa9a' -'\xa9b' -'\xa9c' -'\xa9ci' -'\xa9d' -'\xa9e' -'\xa9e ' -'\xa9es' -'\xa9f' -'\xa9g' -'\xa9h' -'\xa9ho' -'\xa9i' -'\xa9k' -'\xa9l' -'\xa9le' -'\xa9l\xc3' -'\xa9m' -'\xa9m ' -'\xa9n' -'\xa9n ' -'\xa9n\xc3' -'\xa9p' -'\xa9r' -'\xa9ra' -'\xa9re' -'\xa9ri' -'\xa9r\xc3' -'\xa9s' -'\xa9s ' -'\xa9s e' -'\xa9se' -'\xa9si' -'\xa9t' -'\xa9ta' -'\xa9te' -'\xa9t\xc3' -'\xa9t\xc3\xa9' -'\xa9u' -'\xa9v' -'\xa9z' -'\xa9\x82' -'\xa9\x8b' -'\xa9\x8d' -'\xa9\x94' -'\xa9\xb0' -'\xa9\xb1' -'\xa9\xc3' -'\xa9\xce' -'\xa9\xcf' -'\xa9\xd0' -'\xa9\xd7' -'\xa9\xd8' -'\xa9\xd9' -'\xa9\xe0' -'\xa9\xe3' -'\xa9\xeb' -'\xa9\xec' -'\xa9\xed' -'\xaa' -'\xaa)' -'\xaa,' -'\xaa-' -'\xaa.' -'\xaa:' -'\xaam' -'\xaan' -'\xaas' -'\xaat' -'\xaatr' -'\xaa\x82' -'\xaa\x85' -'\xaa\x95' -'\xaa\xa4' -'\xaa\xa8' -'\xaa\xa9' -'\xaa\xaa' -'\xaa\xac' -'\xaa\xad' -'\xaa\xbe' -'\xaa\xca' -'\xaa\xce' -'\xaa\xcf' -'\xaa\xd7' -'\xaa\xd8' -'\xaa\xd9' -'\xaa\xdb' -'\xaa\xe3' -'\xaa\xe3\x81' -'\xaa\xe3\x82' -'\xaa\xe5' -'\xab' -'\xab ' -'\xab,' -'\xab.' -'\xaba' -'\xabb' -'\xabc' -'\xabd' -'\xabg' -'\xabj' -'\xabk' -'\xabm' -'\xabp' -'\xabr' -'\xabs' -'\xabt' -'\xabv' -'\xabz' -'\xab\x87' -'\xab\xc5' -'\xab\xd8' -'\xab\xd9' -'\xab\xe3' -'\xab\xe3\x81' -'\xab\xe3\x82' -'\xac' -'\xac ' -'\xac,' -'\xac.' -'\xacn' -'\xac\xa7' -'\xac\xbe' -'\xac\xce' -'\xac\xcf' -'\xac\xd8' -'\xac\xd9' -'\xac\xe3' -'\xac\xea' -'\xac\xeb' -'\xac\xec' -'\xac\xed' -'\xad' -'\xad ' -'\xad p' -'\xad s' -'\xad)' -'\xad,' -'\xad.' -'\xad:' -'\xada' -'\xada ' -'\xadb' -'\xadc' -'\xadci' -'\xadd' -'\xade' -'\xadh' -'\xadj' -'\xadk' -'\xadl' -'\xadm' -'\xadm ' -'\xadn' -'\xado' -'\xadp' -'\xadr' -'\xads' -'\xadst' -'\xadt' -'\xadv' -'\xadz' -'\xad\x8b' -'\xad\xa3' -'\xad\xc4' -'\xad\xc5' -'\xad\xce' -'\xad\xcf' -'\xad\xd8' -'\xad\xd9' -'\xad\xe3' -'\xad\xe3\x82' -'\xad\xe6' -'\xad\xec' -'\xae' -'\xae\n' -'\xae ' -'\xae)' -'\xae,' -'\xae.' -'\xael' -'\xaem' -'\xaen' -'\xaet' -'\xae\x85' -'\xae\x87' -'\xae\x8a' -'\xae\x8c' -'\xae\x95' -'\xae\x9f' -'\xae\xa4' -'\xae\xa9' -'\xae\xaa' -'\xae\xae' -'\xae\xaf' -'\xae\xb1' -'\xae\xb2' -'\xae\xb3' -'\xae\xc5' -'\xae\xce' -'\xae\xcf' -'\xae\xd8' -'\xae\xd9' -'\xae\xe3' -'\xae\xe3\x81' -'\xae\xe4' -'\xae\xe5' -'\xaf' -'\xaf ' -'\xaf,' -'\xaf.' -'\xafd' -'\xafg' -'\xafl' -'\xafm' -'\xafn' -'\xafp' -'\xafr' -'\xafs' -'\xaft' -'\xafv' -'\xaf\x80' -'\xaf\x81' -'\xaf\x82' -'\xaf\x86' -'\xaf\x87' -'\xaf\xb8' -'\xaf\xc5' -'\xaf\xce' -'\xaf\xcf' -'\xaf\xd0' -'\xaf\xd8' -'\xaf\xd9' -'\xaf\xdb' -'\xaf\xe3' -'\xaf\xe3\x80' -'\xaf\xe3\x80\x81' -'\xaf\xe3\x81' -'\xaf\xe9' -'\xb0' -'\xb0 \xd0' -'\xb0 \xd0\xb2' -'\xb0 \xd0\xbd' -'\xb0 \xd0\xbf' -'\xb0 \xd1' -'\xb0 \xd1\x81' -'\xb0,' -'\xb0, ' -'\xb0.' -'\xb0a' -'\xb0n' -'\xb0s' -'\xb0u' -'\xb0\x80' -'\xb0\x82' -'\xb0\x84' -'\xb0\x86' -'\xb0\x88' -'\xb0\x8e' -'\xb0\x95' -'\xb0\x97' -'\xb0\x98' -'\xb0\x9a' -'\xb0\x9b' -'\xb0\x9c' -'\xb0\xa3' -'\xb0\xa4' -'\xb0\xa5' -'\xb0\xa8' -'\xb0\xa9' -'\xb0\xc3' -'\xb0\xc6' -'\xb0\xd0\xb2' -'\xb0\xd0\xb2\xd0' -'\xb0\xd0\xb6' -'\xb0\xd0\xb7' -'\xb0\xd0\xb7\xd0' -'\xb0\xd0\xb9' -'\xb0\xd0\xba\xd1' -'\xb0\xd0\xbb\xd1' -'\xb0\xd0\xbd\xd0' -'\xb0\xd1\x81\xd1' -'\xb0\xd1\x82' -'\xb0\xd1\x85' -'\xb0\xd1\x8f' -'\xb0\xd8' -'\xb0\xe1' -'\xb0\xe2' -'\xb0\xe3' -'\xb0\xe3\x81' -'\xb0\xe5' -'\xb0\xe6' -'\xb0\xe9' -'\xb0\xea' -'\xb0\xeb' -'\xb0\xed' -'\xb1' -'\xb1 ' -"\xb1'" -'\xb1)' -'\xb1,' -'\xb1.' -'\xb1:' -'\xb1a' -'\xb1c' -'\xb1e' -'\xb1k' -'\xb1l' -'\xb1m' -'\xb1n' -'\xb1o' -'\xb1r' -'\xb1s' -'\xb1t' -'\xb1v' -'\xb1z' -'\xb1\x80' -'\xb1\x86' -'\xb1\x8a' -'\xb1\x8b' -'\xb1\x8d' -'\xb1\xa3' -'\xb1\xa8' -'\xb1\xbb' -'\xb1\xbc' -'\xb1\xbe' -'\xb1\xc2' -'\xb1\xc3' -'\xb1\xc4' -'\xb1\xc5' -'\xb1\xce' -'\xb1\xcf' -'\xb1\xd0' -'\xb1\xd0\xbb' -'\xb1\xd1' -'\xb1\xd1\x83' -'\xb1\xda' -'\xb1\xdb' -'\xb1\xe4' -'\xb1\xec' -'\xb1\xed' -'\xb2' -'\xb2 \xd0' -'\xb2 \xd1' -"\xb2'" -'\xb2,' -'\xb2-' -'\xb2n' -'\xb2\x83' -'\xb2\x84' -'\xb2\x88' -'\xb2\x8c' -'\xb2\x92' -'\xb2\x95' -'\xb2\x9c' -'\xb2\xa4' -'\xb2\xab' -'\xb2\xac' -'\xb2\xbb' -'\xb2\xbf' -'\xb2\xce' -'\xb2\xcf' -'\xb2\xd0' -'\xb2\xd0\xb0' -'\xb2\xd0\xb0 ' -'\xb2\xd0\xb0\xd0' -'\xb2\xd0\xb5' -'\xb2\xd0\xb5\xd0' -'\xb2\xd0\xb5\xd1' -'\xb2\xd0\xb8' -'\xb2\xd0\xb8\xd0' -'\xb2\xd0\xb8\xd1' -'\xb2\xd0\xbb' -'\xb2\xd0\xbd' -'\xb2\xd0\xbe' -'\xb2\xd0\xbe\xd0' -'\xb2\xd0\xbe\xd1' -'\xb2\xd1' -'\xb2\xd1\x80' -'\xb2\xd1\x81' -'\xb2\xd9' -'\xb2\xdb' -'\xb2\xe5' -'\xb2\xe9' -'\xb2\xfa' -'\xb3' -'\xb3 ' -'\xb3 a' -'\xb3 d' -'\xb3 e' -'\xb3 l' -'\xb3 p' -'\xb3 s' -'\xb3,' -'\xb3, ' -'\xb3a' -'\xb3b' -'\xb3c' -'\xb3d' -'\xb3i' -'\xb3j' -'\xb3k' -'\xb3l' -'\xb3n' -'\xb3n ' -'\xb3n d' -'\xb3n,' -'\xb3p' -'\xb3r' -'\xb3ri' -'\xb3s' -'\xb3t' -'\xb3v' -'\xb3w' -'\xb3\x80' -'\xb3\x81' -'\xb3\x84' -'\xb3\xa0' -'\xb3\xa3' -'\xb3\xb4' -'\xb3\xb8' -'\xb3\xc5' -'\xb3\xc9' -'\xb3\xcc' -'\xb3\xce' -'\xb3\xcf' -'\xb3\xd0' -'\xb3\xd0\xb0' -'\xb3\xd0\xbb' -'\xb3\xd0\xbe' -'\xb3\xd0\xbe\xd0' -'\xb3\xd1' -'\xb3\xd8' -'\xb3\xd9' -'\xb3\xdb' -'\xb3\xe3' -'\xb3\xe3\x82' -'\xb3\xe3\x83' -'\xb3\xf6' -'\xb4' -'\xb4 \xd0' -"\xb4'" -'\xb4,' -'\xb4-' -'\xb4.' -'\xb4l' -'\xb4n' -'\xb4r' -'\xb4s' -'\xb4v' -'\xb4\x80' -'\xb4\x82' -'\xb4\x85' -'\xb4\x88' -'\xb4\x8e' -'\xb4\x97' -'\xb4\x99' -'\xb4\x9a' -'\xb4\xa4' -'\xb4\xa8' -'\xb4\xac' -'\xb4\xaf' -'\xb4\xb1' -'\xb4\xb2' -'\xb4\xb3' -'\xb4\xb6' -'\xb4\xb7' -'\xb4\xbd' -'\xb4\xbe' -'\xb4\xc5' -'\xb4\xce' -'\xb4\xcf' -'\xb4\xd0' -'\xb4\xd0\xb0 ' -'\xb4\xd0\xb0\xd0' -'\xb4\xd0\xb2' -'\xb4\xd0\xb5' -'\xb4\xd0\xb5\xd0' -'\xb4\xd0\xb8\xd0' -'\xb4\xd0\xbb' -'\xb4\xd0\xbd' -'\xb4\xd0\xbd\xd0' -'\xb4\xd0\xbe' -'\xb4\xd0\xbe\xd0' -'\xb4\xd1' -'\xb4\xd8' -'\xb4\xd9' -'\xb4\xdb' -'\xb4\xe6' -'\xb4\xea' -'\xb4\xeb' -'\xb4\xec' -'\xb4\xf2' -'\xb4\xf3' -'\xb4\xfa' -'\xb5' -'\xb5 ' -'\xb5 \xd0' -'\xb5 \xd0\xb2' -'\xb5 \xd0\xbd' -'\xb5 \xd0\xbf' -'\xb5 \xd1' -'\xb5 \xd1\x81' -'\xb5,' -'\xb5.' -'\xb5e' -'\xb5es' -'\xb5h' -'\xb5i' -'\xb5l' -'\xb5n' -'\xb5p' -'\xb5r' -'\xb5t' -'\xb5u' -'\xb5\x81' -'\xb5\x8b' -'\xb5\x8d' -'\xb5\x9c' -'\xb5\xac' -'\xb5\xb1' -'\xb5\xb7' -'\xb5\xbc' -'\xb5\xbd' -'\xb5\xc0' -'\xb5\xc4' -'\xb5\xce' -'\xb5\xcf' -'\xb5\xd0\xb3' -'\xb5\xd0\xb4' -'\xb5\xd0\xb4\xd0' -'\xb5\xd0\xb6' -'\xb5\xd0\xb9' -'\xb5\xd0\xba\xd1' -'\xb5\xd0\xbb' -'\xb5\xd0\xbb\xd0' -'\xb5\xd0\xbb\xd1' -'\xb5\xd0\xbc' -'\xb5\xd0\xbd' -'\xb5\xd0\xbd\xd0' -'\xb5\xd1' -'\xb5\xd1\x80' -'\xb5\xd1\x80\xd1' -'\xb5\xd1\x81' -'\xb5\xd1\x82' -'\xb5\xd1\x82\xd0' -'\xb5\xd1\x86' -'\xb5\xd8' -'\xb5\xd9' -'\xb5\xe7' -'\xb5\xeb' -'\xb5\xec' -'\xb6' -'\xb6.' -'\xb6b' -'\xb6d' -'\xb6f' -'\xb6g' -'\xb6i' -'\xb6j' -'\xb6k' -'\xb6l' -'\xb6m' -'\xb6n' -'\xb6p' -'\xb6r' -'\xb6r ' -'\xb6rd' -'\xb6re' -'\xb6rs' -'\xb6s' -'\xb6t' -'\xb6v' -'\xb6y' -'\xb6z' -'\xb6\x80' -'\xb6\x84' -'\xb6\x94' -'\xb6\x9c' -'\xb6\xa8' -'\xb6\xad' -'\xb6\xaf' -'\xb6\xc3' -'\xb6\xc4' -'\xb6\xce' -'\xb6\xcf' -'\xb6\xd0' -'\xb6\xd0\xb0' -'\xb6\xd0\xb5' -'\xb6\xd0\xb5\xd0' -'\xb6\xd1' -'\xb6\xd4' -'\xb6\xd8' -'\xb6\xd9' -'\xb6\xe4' -'\xb6\xe5' -'\xb6\xe7' -'\xb7' -'\xb7 ' -'\xb7 \xd0' -'\xb7)' -'\xb7,' -'\xb7.' -'\xb7:' -'\xb7c' -'\xb7i' -'\xb7l' -'\xb7t' -'\xb7\x9a' -'\xb7\xa2' -'\xb7\xa8' -'\xb7\xaf' -'\xb7\xb1' -'\xb7\xb2' -'\xb7\xb8' -'\xb7\xbd' -'\xb7\xc4' -'\xb7\xce' -'\xb7\xcf' -'\xb7\xd0' -'\xb7\xd0\xb0' -'\xb7\xd0\xb0\xd0' -'\xb7\xd0\xb0\xd1' -'\xb7\xd0\xb4' -'\xb7\xd0\xb5' -'\xb7\xd0\xb8' -'\xb7\xd0\xbd' -'\xb7\xd1' -'\xb7\xd6' -'\xb7\xd8' -'\xb7\xd9' -'\xb7\xdb' -'\xb7\xe3' -'\xb7\xe7' -'\xb7\xfe' -'\xb8' -'\xb8 ' -'\xb8 \xd0' -'\xb8 \xd0\xb2' -'\xb8 \xd0\xbd' -'\xb8 \xd0\xbf' -'\xb8 \xd1' -'\xb8 \xd1\x81' -'\xb8,' -'\xb8, ' -'\xb8.' -'\xb8. ' -'\xb8b' -'\xb8d' -'\xb8g' -'\xb8j' -'\xb8k' -'\xb8l' -'\xb8m' -'\xb8n' -'\xb8r' -'\xb8s' -'\xb8t' -'\xb8v' -'\xb8y' -'\xb8\x81' -'\xb8\x84' -'\xb8\x87' -'\xb8\x88' -'\xb8\x93' -'\xb8\x94' -'\xb8\x95' -'\xb8\x96' -'\xb8\x97' -'\xb8\x9a' -'\xb8\x9b' -'\xb8\x9c' -'\xb8\x9e' -'\xb8\xa0' -'\xb8\xa2' -'\xb8\xa5' -'\xb8\xa7' -'\xb8\xa8' -'\xb8\xa9' -'\xb8\xaa' -'\xb8\xab' -'\xb8\xb1' -'\xb8\xb3' -'\xb8\xb6' -'\xb8\xb7' -'\xb8\xb9' -'\xb8\xba' -'\xb8\xbb' -'\xb8\xbd' -'\xb8\xce' -'\xb8\xcf' -'\xb8\xd0\xb0' -'\xb8\xd0\xb2' -'\xb8\xd0\xb4' -'\xb8\xd0\xb5' -'\xb8\xd0\xb7' -'\xb8\xd0\xb7\xd0' -'\xb8\xd0\xb8' -'\xb8\xd0\xb9' -'\xb8\xd0\xbb\xd0' -'\xb8\xd0\xbc' -'\xb8\xd0\xbc\xd0' -'\xb8\xd0\xbe' -'\xb8\xd1\x80' -'\xb8\xd1\x80\xd0' -'\xb8\xd1\x81' -'\xb8\xd1\x81\xd1' -'\xb8\xd1\x82' -'\xb8\xd1\x82\xd0' -'\xb8\xd1\x85' -'\xb8\xd1\x86' -'\xb8\xd1\x87' -'\xb8\xd1\x87\xd0' -'\xb8\xd1\x8f' -'\xb8\xd1\x8f ' -'\xb8\xd8' -'\xb8\xd9' -'\xb8\xdf' -'\xb8\xe6' -'\xb8\xea' -'\xb8\xeb' -'\xb8\xec' -'\xb8\xed' -'\xb8\xf6' -'\xb8\xfc' -'\xb9' -'\xb9 ' -'\xb9 \xd0' -'\xb9 \xd1' -"\xb9'" -'\xb9,' -'\xb9.' -'\xb9:' -'\xb9n' -'\xb9\x81' -'\xb9\x82' -'\xb9\x84' -'\xb9\x88' -'\xb9\x89' -'\xb9\x8c' -'\xb9\x98' -'\xb9\xab' -'\xb9\xce' -'\xb9\xcf' -'\xb9\xd0' -'\xb9\xd1' -'\xb9\xd1\x81' -'\xb9\xd1\x82' -'\xb9\xd9' -'\xb9\xe3\x81' -'\xb9\xe6' -'\xb9\xec' -'\xb9\xfa' -'\xb9\xfd' -'\xba' -'\xba ' -'\xba \xd0' -'\xba,' -'\xba<' -'\xbaa' -'\xbab' -'\xbac' -'\xbad' -'\xbag' -'\xbah' -'\xbaj' -'\xbak' -'\xbal' -'\xban ' -'\xbap' -'\xbar' -'\xbas' -'\xbat' -'\xbav' -'\xbaz' -'\xba\xa1' -'\xba\xa3' -'\xba\xa5' -'\xba\xa7' -'\xba\xab' -'\xba\xad' -'\xba\xaf' -'\xba\xb1' -'\xba\xb7' -'\xba\xba' -'\xba\xbf' -'\xba\xc3' -'\xba\xc4' -'\xba\xc5' -'\xba\xcd' -'\xba\xce' -'\xba\xcf' -'\xba\xd0' -'\xba\xd0\xb0 ' -'\xba\xd0\xb0\xd1' -'\xba\xd0\xb5' -'\xba\xd0\xb8' -'\xba\xd0\xb8 ' -'\xba\xd0\xbe' -'\xba\xd0\xbe\xd0' -'\xba\xd0\xbe\xd1' -'\xba\xd1' -'\xba\xd1\x80' -'\xba\xd1\x81' -'\xba\xd1\x82' -'\xba\xd1\x83' -'\xba\xd8' -'\xba\xd9' -'\xba\xf3' -'\xbb' -'\xbb \xd0' -'\xbb-' -'\xbbr' -'\xbb\x81' -'\xbb\x83' -'\xbb\x85' -'\xbb\x87' -'\xbb\x89' -'\xbb\x8b' -'\xbb\x8d' -'\xbb\x8f' -'\xbb\x91' -'\xbb\x93' -'\xbb\x95' -'\xbb\x97' -'\xbb\x99' -'\xbb\x9b' -'\xbb\x9d' -'\xbb\x9f' -'\xbb\xa3' -'\xbb\xa5' -'\xbb\xa7' -'\xbb\xa9' -'\xbb\xab' -'\xbb\xad' -'\xbb\xaf' -'\xbb\xb0' -'\xbb\xb1' -'\xbb\xb6' -'\xbb\xba' -'\xbb\xbb' -'\xbb\xbd' -'\xbb\xce' -'\xbb\xcf' -'\xbb\xd0\xb5' -'\xbb\xd0\xb5\xd0' -'\xbb\xd0\xb8' -'\xbb\xd0\xbd' -'\xbb\xd1' -'\xbb\xd1\x8c' -'\xbb\xd1\x8e' -'\xbb\xd1\x8f' -'\xbb\xd6' -'\xbb\xe1' -'\xbb\xfa' -'\xbc' -'\xbc ' -'\xbc \xd0' -"\xbc'" -'\xbc,' -'\xbc-' -'\xbc.' -'\xbca' -'\xbcb' -'\xbcbe' -'\xbcber' -'\xbcc' -'\xbcck' -'\xbce' -'\xbcg' -'\xbch' -'\xbchr' -'\xbci' -'\xbcj' -'\xbck' -'\xbcl' -'\xbcm' -'\xbco' -'\xbcp' -'\xbcr' -'\xbcr ' -'\xbcr d' -'\xbcs' -'\xbct' -'\xbcu' -'\xbcv' -'\xbcy' -'\xbcz' -'\xbc\x8b' -'\xbc\x8d' -'\xbc\xba' -'\xbc\xc3' -'\xbc\xc4' -'\xbc\xc5' -'\xbc\xc6' -'\xbc\xca' -'\xbc\xce' -'\xbc\xcf' -'\xbc\xd0' -'\xbc\xd0\xb0\xd0' -'\xbc\xd0\xb5' -'\xbc\xd0\xb5\xd0' -'\xbc\xd0\xb8' -'\xbc\xd1' -'\xbc\xd1\x83' -'\xbc\xd2' -'\xbc\xd3' -'\xbc\xe3' -'\xbc\xe3\x82' -'\xbc\xe3\x82\xb9' -'\xbc\xe3\x83' -'\xbc\xe4' -'\xbc\xea' -'\xbc\xeb' -'\xbc\xec' -'\xbc\xed' -'\xbd' -'\xbd ' -'\xbd \xd0' -'\xbd)' -'\xbd,' -'\xbd, ' -'\xbd.' -'\xbd:' -'\xbdb' -'\xbdc' -'\xbdch' -'\xbdi' -'\xbdk' -'\xbdl' -'\xbdm' -'\xbdn' -'\xbdr' -'\xbds' -'\xbdt' -'\xbdv' -'\xbdz' -'\xbd\x81' -'\xbd\x82' -'\xbd\x86' -'\xbd\x89' -'\xbd\x8f' -'\xbd\x90' -'\xbd\x94' -'\xbd\x99' -'\xbd\x9e' -'\xbd\xa0' -'\xbd\xa2' -'\xbd\xa4' -'\xbd\xa6' -'\xbd\xb1' -'\xbd\xbb' -'\xbd\xc5' -'\xbd\xce' -'\xbd\xcf' -'\xbd\xd0\xb0' -'\xbd\xd0\xb0 ' -'\xbd\xd0\xb0\xd0' -'\xbd\xd0\xb5' -'\xbd\xd0\xb5 ' -'\xbd\xd0\xb5\xd0' -'\xbd\xd0\xb5\xd1' -'\xbd\xd0\xb8' -'\xbd\xd0\xb8\xd0' -'\xbd\xd0\xb8\xd1' -'\xbd\xd0\xba' -'\xbd\xd0\xbd' -'\xbd\xd0\xbe' -'\xbd\xd0\xbe ' -'\xbd\xd0\xbe\xd0' -'\xbd\xd0\xbe\xd1' -'\xbd\xd1' -'\xbd\xd1\x83' -'\xbd\xd1\x8b' -'\xbd\xd1\x8f' -'\xbd\xd3' -'\xbd\xe3' -'\xbd\xe7' -'\xbd\xe9' -'\xbd\xec' -'\xbd\xf0' -'\xbd\xf8' -'\xbe' -'\xbe \xd0' -'\xbe \xd0\xb2' -'\xbe \xd1' -'\xbe \xd1\x81' -'\xbe,' -'\xbe-' -'\xbe<' -'\xbea' -'\xbeb' -'\xbed' -'\xbee' -'\xbee ' -'\xbeen' -'\xbek' -'\xbem' -'\xben' -'\xbeo' -'\xbes' -'\xbet' -'\xbeu' -'\xbev' -'\xbey' -'\xbe\xa4' -'\xbe\xad' -'\xbe\xb1' -'\xbe\xb5' -'\xbe\xc3' -'\xbe\xc4' -'\xbe\xcd' -'\xbe\xce' -'\xbe\xcf' -'\xbe\xd0\xb1' -'\xbe\xd0\xb1\xd1' -'\xbe\xd0\xb2' -'\xbe\xd0\xb2\xd0' -'\xbe\xd0\xb2\xd1' -'\xbe\xd0\xb3' -'\xbe\xd0\xb3\xd0' -'\xbe\xd0\xb4' -'\xbe\xd0\xb4\xd0' -'\xbe\xd0\xb5' -'\xbe\xd0\xb6\xd0' -'\xbe\xd0\xb7' -'\xbe\xd0\xb7\xd0' -'\xbe\xd0\xb8' -'\xbe\xd0\xb9' -'\xbe\xd0\xbb' -'\xbe\xd0\xbc' -'\xbe\xd0\xbc\xd0' -'\xbe\xd0\xbd\xd1' -'\xbe\xd0\xbf' -'\xbe\xd0\xbf\xd0' -'\xbe\xd1' -'\xbe\xd1\x80\xd1' -'\xbe\xd1\x81' -'\xbe\xd1\x81\xd0' -'\xbe\xd1\x81\xd1' -'\xbe\xd1\x82' -'\xbe\xd1\x82\xd0' -'\xbe\xd8' -'\xbe\xdb' -'\xbe\xe3' -'\xbe\xe3\x81' -'\xbf' -'\xbf ' -'\xbf)' -'\xbf,' -'\xbf.' -'\xbf:' -'\xbfn' -'\xbft' -'\xbf\xaa' -'\xbf\xb4' -'\xbf\xc9' -'\xbf\xce' -'\xbf\xcf' -'\xbf\xd0' -'\xbf\xd0\xbe' -'\xbf\xd0\xbe\xd0' -'\xbf\xd0\xbe\xd1' -'\xbf\xd1' -'\xbf\xd1\x80' -'\xbf\xd1\x80\xd0' -'\xbf\xe3' -'\xbf\xe3\x83' -'\xbf\xe3\x83\xbc' -'\xc0\xb4' -'\xc0\xed' -'\xc1 ' -'\xc1\xaa' -'\xc1\xcb' -'\xc2\xbc' -'\xc2\xd2' -'\xc2\xdb' -'\xc3\x81' -'\xc3\x82' -'\xc3\x83' -'\xc3\x85' -'\xc3\x86' -'\xc3\x87' -'\xc3\x8d' -'\xc3\x8e' -'\xc3\x94' -'\xc3\x95' -'\xc3\x96' -'\xc3\x98' -'\xc3\x9a' -'\xc3\x9c' -'\xc3\x9d' -'\xc3\x9f' -'\xc3\x9f ' -'\xc3\x9fe' -'\xc3\xa0' -'\xc3\xa0 ' -'\xc3\xa0 c' -'\xc3\xa0 d' -'\xc3\xa0 l' -'\xc3\xa1' -'\xc3\xa1 ' -'\xc3\xa1 s' -'\xc3\xa1b' -'\xc3\xa1c' -'\xc3\xa1d' -'\xc3\xa1g' -'\xc3\xa1k' -'\xc3\xa1l' -'\xc3\xa1m' -'\xc3\xa1n' -'\xc3\xa1n ' -'\xc3\xa1r' -'\xc3\xa1s' -'\xc3\xa1s ' -'\xc3\xa1t' -'\xc3\xa1v' -'\xc3\xa1z' -'\xc3\xa2' -'\xc3\xa2n' -'\xc3\xa3' -'\xc3\xa3o' -'\xc3\xa3o ' -'\xc3\xa4' -'\xc3\xa4 ' -'\xc3\xa4g' -'\xc3\xa4h' -'\xc3\xa4i' -'\xc3\xa4k' -'\xc3\xa4l' -'\xc3\xa4ll' -'\xc3\xa4m' -'\xc3\xa4n' -'\xc3\xa4nd' -'\xc3\xa4ng' -'\xc3\xa4r' -'\xc3\xa4s' -'\xc3\xa4t' -'\xc3\xa4tt' -'\xc3\xa4\xc3' -'\xc3\xa5' -'\xc3\xa5 ' -'\xc3\xa5d' -'\xc3\xa5n' -'\xc3\xa5r' -'\xc3\xa6' -'\xc3\xa7' -'\xc3\xa7a' -'\xc3\xa7ai' -'\xc3\xa7o' -'\xc3\xa7\xc3' -'\xc3\xa8' -'\xc3\xa8 ' -'\xc3\xa8m' -'\xc3\xa8r' -'\xc3\xa8re' -'\xc3\xa8s' -'\xc3\xa8s ' -'\xc3\xa9 ' -'\xc3\xa9 a' -'\xc3\xa9 d' -'\xc3\xa9 l' -'\xc3\xa9 m' -'\xc3\xa9 n' -'\xc3\xa9 p' -'\xc3\xa9 s' -'\xc3\xa9 v' -'\xc3\xa9a' -'\xc3\xa9b' -'\xc3\xa9c' -'\xc3\xa9ci' -'\xc3\xa9d' -'\xc3\xa9e' -'\xc3\xa9e ' -'\xc3\xa9es' -'\xc3\xa9f' -'\xc3\xa9g' -'\xc3\xa9h' -'\xc3\xa9l' -'\xc3\xa9m' -'\xc3\xa9n' -'\xc3\xa9n ' -'\xc3\xa9p' -'\xc3\xa9r' -'\xc3\xa9ra' -'\xc3\xa9re' -'\xc3\xa9ri' -'\xc3\xa9s' -'\xc3\xa9s ' -'\xc3\xa9se' -'\xc3\xa9si' -'\xc3\xa9t' -'\xc3\xa9ta' -'\xc3\xa9t\xc3' -'\xc3\xa9v' -'\xc3\xaa' -'\xc3\xaam' -'\xc3\xaat' -'\xc3\xab' -'\xc3\xac' -'\xc3\xad' -'\xc3\xad ' -'\xc3\xada' -'\xc3\xada ' -'\xc3\xadc' -'\xc3\xadd' -'\xc3\xadl' -'\xc3\xadm' -'\xc3\xadr' -'\xc3\xads' -'\xc3\xadt' -'\xc3\xadv' -'\xc3\xae' -'\xc3\xaf' -'\xc3\xb1' -'\xc3\xb1a' -'\xc3\xb1o' -'\xc3\xb2' -'\xc3\xb3' -'\xc3\xb3 ' -'\xc3\xb3 e' -'\xc3\xb3d' -'\xc3\xb3l' -'\xc3\xb3n' -'\xc3\xb3n ' -'\xc3\xb3p' -'\xc3\xb3r' -'\xc3\xb3s' -'\xc3\xb4' -'\xc3\xb5' -'\xc3\xb5e' -'\xc3\xb6' -'\xc3\xb6d' -'\xc3\xb6f' -'\xc3\xb6g' -'\xc3\xb6k' -'\xc3\xb6l' -'\xc3\xb6n' -'\xc3\xb6r' -'\xc3\xb6s' -'\xc3\xb6t' -'\xc3\xb6v' -'\xc3\xb6\xc3' -'\xc3\xb8' -'\xc3\xb8r' -'\xc3\xb9' -'\xc3\xb9 ' -'\xc3\xba' -'\xc3\xba ' -'\xc3\xbab' -'\xc3\xbal' -'\xc3\xban' -'\xc3\xbas' -'\xc3\xbc' -'\xc3\xbcb' -'\xc3\xbcbe' -'\xc3\xbcc' -'\xc3\xbcg' -'\xc3\xbch' -'\xc3\xbcl' -'\xc3\xbcn' -'\xc3\xbcr' -'\xc3\xbcr ' -'\xc3\xbcs' -'\xc3\xbd' -'\xc3\xbd ' -'\xc3\xbdc' -'\xc3\xbe' -'\xc3\xc7' -'\xc3\xd1' -'\xc3\xdf' -'\xc3\xe4' -'\xc3\xf7' -'\xc3\xfb' -'\xc4\x81' -'\xc4\x82' -'\xc4\x83' -'\xc4\x85' -'\xc4\x85 ' -'\xc4\x87' -'\xc4\x8a' -'\xc4\x8b' -'\xc4\x8c' -'\xc4\x8d' -'\xc4\x8da' -'\xc4\x8de' -'\xc4\x8dn' -'\xc4\x8f' -'\xc4\x90' -'\xc4\x91' -'\xc4\x92' -'\xc4\x93' -'\xc4\x96' -'\xc4\x97' -'\xc4\x98' -'\xc4\x99' -'\xc4\x9b' -'\xc4\x9f' -'\xc4\xa0' -'\xc4\xa1' -'\xc4\xa3' -'\xc4\xa6' -'\xc4\xa7' -'\xc4\xa9' -'\xc4\xaa' -'\xc4\xab' -'\xc4\xae' -'\xc4\xaf' -'\xc4\xb0' -'\xc4\xb1' -'\xc4\xb7' -'\xc4\xba' -'\xc4\xbc' -'\xc4\xbe' -'\xc4\xcf' -'\xc4\xd0' -'\xc4\xda' -'\xc4\xdc' -'\xc4\xea' -'\xc4\xfa' -'\xc5\x81' -'\xc5\x82' -'\xc5\x84' -'\xc5\x86' -'\xc5\x88' -'\xc5\x91' -'\xc5\x98' -'\xc5\x99' -'\xc5\x9a' -'\xc5\x9b' -'\xc5\x9e' -'\xc5\x9f' -'\xc5\xa0' -'\xc5\xa1e' -'\xc5\xa1k' -'\xc5\xa2' -'\xc5\xa3' -'\xc5\xa5' -'\xc5\xa9' -'\xc5\xab' -'\xc5\xad' -'\xc5\xaf' -'\xc5\xb1' -'\xc5\xb2' -'\xc5\xb3' -'\xc5\xba' -'\xc5\xbb' -'\xc5\xbc' -'\xc5\xbd' -'\xc5\xbe' -'\xc5\xbe ' -'\xc5\xbee' -'\xc5\xe1' -'\xc6\xa1' -'\xc6\xb0' -'\xc6\xb7' -'\xc6\xed' -'\xc7\xb0' -'\xc7\xcc' -'\xc7\xd1' -'\xc7\xd8' -'\xc7\xda' -'\xc7\xdd' -'\xc7\xde' -'\xc7\xdf' -'\xc7\xe1' -'\xc7\xe3' -'\xc7\xe4' -'\xc7\xe6' -'\xc7\xe9' -'\xc7\xeb' -'\xc7\xed' -'\xc7\xf8' -'\xc8 ' -'\xc8\x99' -'\xc8\x9b' -'\xc8\xa8' -'\xc8\xab' -'\xc8\xc7' -'\xc8\xcb' -'\xc8\xcd' -'\xc8\xd1' -'\xc8\xd5' -'\xc8\xd8' -'\xc8\xdd' -'\xc8\xde' -'\xc8\xdf' -'\xc8\xe1' -'\xc8\xe4' -'\xc8\xe6' -'\xc8\xe7' -'\xc8\xeb' -'\xc8\xfd' -'\xc9 ' -'\xc9"' -'\xc9,' -'\xc9\xcc' -'\xc9\xcf' -'\xc9\xe8' -'\xc9\xfa' -'\xca ' -'\xca\xb1' -'\xca\xb5' -'\xca\xbd' -'\xca\xc8' -'\xca\xd7' -'\xca\xda' -'\xca\xdd' -'\xca\xde' -'\xca\xdf' -'\xca\xe3' -'\xca\xe5' -'\xca\xe6' -'\xca\xed' -'\xcb\xf7' -'\xcb\xf9' -'\xcc\xc7' -'\xcc\xcf' -'\xcc\xe2' -'\xcc\xe3' -'\xcc\xe4' -'\xcc\xe6' -'\xcc\xec' -'\xcc\xed' -'\xcd\xa8' -'\xcd\xbc' -'\xcd\xc6' -'\xcd\xc7' -'\xcd\xde' -'\xcd\xdf' -'\xcd\xe1' -'\xcd\xe3' -'\xcd\xf8' -'\xce\x86' -'\xce\x88' -'\xce\x8c' -'\xce\x90' -'\xce\x91' -'\xce\x92' -'\xce\x93' -'\xce\x94' -'\xce\x95' -'\xce\x97' -'\xce\x98' -'\xce\x99' -'\xce\x9a' -'\xce\x9b' -'\xce\x9c' -'\xce\x9d' -'\xce\x9e' -'\xce\x9f' -'\xce\xa0' -'\xce\xa1' -'\xce\xa3' -'\xce\xa4' -'\xce\xa5' -'\xce\xa6' -'\xce\xa7' -'\xce\xa9' -'\xce\xaa' -'\xce\xac' -'\xce\xad' -'\xce\xae' -'\xce\xaf' -'\xce\xb1' -'\xce\xb2' -'\xce\xb3' -'\xce\xb4' -'\xce\xb5' -'\xce\xb6' -'\xce\xb7' -'\xce\xb8' -'\xce\xb9' -'\xce\xba' -'\xce\xbb' -'\xce\xbc' -'\xce\xbd' -'\xce\xbe' -'\xce\xbf' -'\xce\xc4' -'\xce\xc7' -'\xce\xd1' -'\xce\xd2' -'\xce\xde' -'\xce\xe1' -'\xce\xe6' -'\xce\xf1' -'\xcf' -'\xcf ' -'\xcf\x80' -'\xcf\x81' -'\xcf\x82' -'\xcf\x83' -'\xcf\x84' -'\xcf\x85' -'\xcf\x86' -'\xcf\x87' -'\xcf\x88' -'\xcf\x89' -'\xcf\x8a' -'\xcf\x8c' -'\xcf\x8d' -'\xcf\x8e' -'\xcf\xa2' -'\xcf\xb5' -'\xcf\xc2' -'\xcf\xc7' -'\xcf\xd1' -'\xcf\xd6' -'\xcf\xe0' -'\xcf\xe1' -'\xcf\xe6' -'\xd0\x86' -'\xd0\x88' -'\xd0\x92' -'\xd0\x95' -'\xd0\x95\xd0' -'\xd0\x97' -'\xd0\x97\xd0' -'\xd0\x98' -'\xd0\x98\xd0' -'\xd0\x99' -'\xd0\x9d' -'\xd0\x9d\xd0' -'\xd0\x9d\xd0\xb0' -'\xd0\x9e\xd0' -'\xd0\x9f\xd0\xbe' -'\xd0\xa0' -'\xd0\xa0\xd0' -'\xd0\xa1' -'\xd0\xa1\xd0' -'\xd0\xa1\xd1' -'\xd0\xa2' -'\xd0\xa2\xd0' -'\xd0\xa3' -'\xd0\xa4' -'\xd0\xa6' -'\xd0\xa7' -'\xd0\xaf' -'\xd0\xb0 ' -'\xd0\xb0 \xd0' -'\xd0\xb0 \xd1' -'\xd0\xb0,' -'\xd0\xb0, ' -'\xd0\xb0.' -'\xd0\xb0\xd0\xb2' -'\xd0\xb0\xd0\xb7' -'\xd0\xb0\xd0\xb9' -'\xd0\xb0\xd1\x82' -'\xd0\xb1' -'\xd0\xb1\xd0' -'\xd0\xb1\xd0\xbb' -'\xd0\xb1\xd1' -'\xd0\xb2' -'\xd0\xb2 ' -'\xd0\xb2 \xd0' -'\xd0\xb2 \xd1' -'\xd0\xb2\xd0\xb5' -'\xd0\xb2\xd0\xb8' -'\xd0\xb2\xd0\xbd' -'\xd0\xb2\xd0\xbe' -'\xd0\xb2\xd1' -'\xd0\xb3' -'\xd0\xb3\xd0' -'\xd0\xb3\xd0\xb0' -'\xd0\xb3\xd0\xbe' -'\xd0\xb4' -'\xd0\xb4 ' -'\xd0\xb4 \xd0' -'\xd0\xb4\xd0' -'\xd0\xb4\xd0\xb5' -'\xd0\xb4\xd0\xbd' -'\xd0\xb4\xd0\xbe' -'\xd0\xb5 ' -'\xd0\xb5 \xd0' -'\xd0\xb5 \xd1' -'\xd0\xb5,' -'\xd0\xb5\xd0\xb4' -'\xd0\xb5\xd0\xb9' -'\xd0\xb5\xd0\xbb' -'\xd0\xb5\xd0\xbc' -'\xd0\xb5\xd0\xbd' -'\xd0\xb5\xd1\x80' -'\xd0\xb5\xd1\x81' -'\xd0\xb5\xd1\x82' -'\xd0\xb6' -'\xd0\xb6\xd0' -'\xd0\xb6\xd0\xb5' -'\xd0\xb7' -'\xd0\xb7 ' -'\xd0\xb7\xd0' -'\xd0\xb7\xd0\xb0' -'\xd0\xb7\xd0\xb8' -'\xd0\xb7\xd1' -'\xd0\xb8 ' -'\xd0\xb8 \xd0' -'\xd0\xb8 \xd1' -'\xd0\xb8,' -'\xd0\xb8.' -'\xd0\xb8\xd0\xb2' -'\xd0\xb8\xd0\xb4' -'\xd0\xb8\xd0\xb5' -'\xd0\xb8\xd0\xb7' -'\xd0\xb8\xd0\xb9' -'\xd0\xb8\xd0\xbc' -'\xd0\xb8\xd1\x80' -'\xd0\xb8\xd1\x81' -'\xd0\xb8\xd1\x82' -'\xd0\xb8\xd1\x86' -'\xd0\xb8\xd1\x87' -'\xd0\xb8\xd1\x8f' -'\xd0\xb9' -'\xd0\xb9 ' -'\xd0\xb9 \xd0' -'\xd0\xb9\xd0' -'\xd0\xb9\xd1' -'\xd0\xba\xd0\xb5' -'\xd0\xba\xd0\xb8' -'\xd0\xba\xd0\xbe' -'\xd0\xba\xd1\x80' -'\xd0\xba\xd1\x81' -'\xd0\xba\xd1\x82' -'\xd0\xba\xd1\x83' -'\xd0\xbb ' -'\xd0\xbb \xd0' -'\xd0\xbb\xd0\xb5' -'\xd0\xbb\xd0\xb8' -'\xd0\xbb\xd1\x8c' -'\xd0\xbb\xd1\x8f' -'\xd0\xbc' -'\xd0\xbc ' -'\xd0\xbc \xd0' -'\xd0\xbc\xd0\xb5' -'\xd0\xbc\xd0\xb8' -'\xd0\xbd\xd0\xb0' -'\xd0\xbd\xd0\xb5' -'\xd0\xbd\xd0\xb8' -'\xd0\xbd\xd0\xbd' -'\xd0\xbd\xd0\xbe' -'\xd0\xbe ' -'\xd0\xbe \xd0' -'\xd0\xbe \xd1' -'\xd0\xbe,' -'\xd0\xbe\xd0\xb1' -'\xd0\xbe\xd0\xb2' -'\xd0\xbe\xd0\xb3' -'\xd0\xbe\xd0\xb4' -'\xd0\xbe\xd0\xb7' -'\xd0\xbe\xd0\xb9' -'\xd0\xbe\xd0\xbc' -'\xd0\xbe\xd0\xbf' -'\xd0\xbe\xd1\x81' -'\xd0\xbe\xd1\x82' -'\xd0\xbf' -'\xd0\xbf\xd0' -'\xd0\xbf\xd0\xbe' -'\xd0\xbf\xd1' -'\xd0\xbf\xd1\x80' -'\xd0\xc2' -'\xd0\xc4' -'\xd0\xc5' -'\xd0\xce' -'\xd0\xd0' -'\xd0\xd4' -'\xd0\xe5' -'\xd1 ' -'\xd1\x80 ' -'\xd1\x80\xd0\xb5' -'\xd1\x80\xd0\xbe' -'\xd1\x80\xd1\x83' -'\xd1\x81 ' -'\xd1\x81 \xd0' -'\xd1\x81\xd0' -'\xd1\x81\xd0\xb5' -'\xd1\x81\xd0\xb8' -'\xd1\x81\xd0\xba' -'\xd1\x81\xd0\xbb' -'\xd1\x81\xd0\xbd' -'\xd1\x81\xd0\xbe' -'\xd1\x81\xd1' -'\xd1\x81\xd1\x82' -'\xd1\x81\xd1\x8f' -'\xd1\x82 ' -'\xd1\x82 \xd0' -'\xd1\x82 \xd1' -'\xd1\x82\xd0\xb0' -'\xd1\x82\xd0\xb2' -'\xd1\x82\xd0\xb5' -'\xd1\x82\xd0\xb8' -'\xd1\x82\xd0\xbd' -'\xd1\x82\xd0\xbe' -'\xd1\x82\xd1' -'\xd1\x82\xd1\x80' -'\xd1\x82\xd1\x83' -'\xd1\x83 ' -'\xd1\x83 \xd0' -'\xd1\x83\xd1' -'\xd1\x83\xd1\x81' -'\xd1\x85' -'\xd1\x85 ' -'\xd1\x86\xd0\xb5' -'\xd1\x86\xd0\xb8' -'\xd1\x87' -'\xd1\x87\xd0' -'\xd1\x87\xd0\xb5' -'\xd1\x87\xd0\xb8' -'\xd1\x87\xd0\xbd' -'\xd1\x88\xd0\xb5' -'\xd1\x89' -'\xd1\x89\xd0' -'\xd1\x8a' -'\xd1\x8b' -'\xd1\x8b\xd0' -'\xd1\x8c' -'\xd1\x8c ' -'\xd1\x8c\xd0' -'\xd1\x8c\xd1' -'\xd1\x8d' -'\xd1\x8e' -'\xd1\x8e\xd0' -'\xd1\x8e\xd1' -'\xd1\x8f' -'\xd1\x8f ' -'\xd1\x8f \xd0' -'\xd1\x8f \xd1' -'\xd1\x8f\xd0' -'\xd1\x8f\xd1' -'\xd1\x94' -'\xd1\x97' -'\xd1\x98' -'\xd1\x9a' -'\xd1\xc8' -'\xd1\xc9' -'\xd1\xd6' -'\xd1\xde' -'\xd1\xdf' -'\xd1\xe3' -'\xd1\xe6' -'\xd1\xed' -'\xd2 ' -'\xd2\xaa' -'\xd2\xb3' -'\xd2\xb5' -'\xd2\xbb' -'\xd2\xc3' -'\xd2\xd4' -'\xd2\xe2' -'\xd2\xed' -'\xd3 ' -'\xd3\xc3' -'\xd3\xc7' -'\xd3\xd0' -'\xd3\xe1' -'\xd3\xe3' -'\xd3\xe6' -'\xd3\xeb' -'\xd3\xed' -'\xd4\xb1' -'\xd4\xb4' -'\xd4\xc2' -'\xd4\xc7' -'\xd4\xd1' -'\xd4\xda' -'\xd5\xbe' -'\xd5\xd1' -'\xd5\xdd' -'\xd5\xdf' -'\xd5\xe2' -'\xd5\xfd' -'\xd6\xae' -'\xd6\xbb' -'\xd6\xc3' -'\xd6\xd0' -'\xd6\xe6' -'\xd6\xed' -'\xd6\xf7' -'\xd7\x91' -'\xd7\x92' -'\xd7\x93' -'\xd7\x96' -'\xd7\x97' -'\xd7\x98' -'\xd7\x9a' -'\xd7\x9b' -'\xd7\x9d' -'\xd7\x9f' -'\xd7\xa3' -'\xd7\xa4' -'\xd7\xa5' -'\xd7\xa6' -'\xd7\xa7' -'\xd7\xa9' -'\xd7\xca' -'\xd7\xd2' -'\xd7\xd3' -'\xd7\xd4' -'\xd7\xd6' -'\xd7\xee' -'\xd7\xf7' -'\xd8\x8c' -'\xd8\xa1' -'\xd8\xa2' -'\xd8\xa3' -'\xd8\xa4' -'\xd8\xa5' -'\xd8\xa6' -'\xd8\xa9' -'\xd8\xab' -'\xd8\xac' -'\xd8\xad' -'\xd8\xae' -'\xd8\xb0' -'\xd8\xb6' -'\xd8\xb7' -'\xd8\xb8' -'\xd8\xb9' -'\xd8\xba' -'\xd8\xd1' -'\xd8\xd6' -'\xd9 ' -'\xd9\x81' -'\xd9\x82' -'\xd9\x83' -'\xd9\x87' -'\xd9\x89' -'\xd9\xbe' -'\xda\x86' -'\xda\xa9' -'\xda\xaf' -'\xda\xc7' -'\xda\xd1' -'\xda\xe1' -'\xda\xe4' -'\xda\xe6' -'\xdb\xb2' -'\xdc\xc7' -'\xdd ' -'\xdd\xc7' -'\xdd\xc9' -'\xdd\xcd' -'\xdd\xd1' -'\xdd\xed' -'\xde ' -'\xde\xc7' -'\xde\xc8' -'\xde\xc9' -'\xde\xca' -'\xde\xcf' -'\xde\xd1' -'\xde\xd3' -'\xde\xe6' -'\xde\xed' -'\xdf\xc7' -'\xdf\xc9' -'\xdf\xd1' -'\xdf\xe1' -'\xdf\xe3' -'\xdf\xe6' -'\xe0\xb0' -'\xe0\xb1' -'\xe0\xb4' -'\xe0\xbd' -'\xe0\xbe' -'\xe1,' -'\xe1\xba' -'\xe1\xbb' -'\xe1\xc2' -'\xe1\xc3' -'\xe1\xc5' -'\xe1\xc7' -'\xe1\xc8' -'\xe1\xc9' -'\xe1\xcc' -'\xe1\xcd' -'\xe1\xce' -'\xe1\xcf' -'\xe1\xd1' -'\xe1\xd3' -'\xe1\xd5' -'\xe1\xda' -'\xe1\xdd' -'\xe1\xde' -'\xe1\xdf' -'\xe1\xe1' -'\xe1\xe3' -'\xe1\xe4' -'\xe1\xe5' -'\xe1\xe6' -'\xe1\xec' -'\xe1\xed' -'\xe3 ' -'\xe3"' -'\xe3\x80' -'\xe3\x80\x81' -'\xe3\x80\x81\xe3' -'\xe3\x80\x81\xe5' -'\xe3\x80\x82' -'\xe3\x81' -'\xe3\x81\x82' -'\xe3\x81\x84' -'\xe3\x81\x84\xe3' -'\xe3\x81\x86' -'\xe3\x81\x8b' -'\xe3\x81\x8c' -'\xe3\x81\x8d' -'\xe3\x81\x8f' -'\xe3\x81\x93' -'\xe3\x81\x95' -'\xe3\x81\x97' -'\xe3\x81\x97\xe3' -'\xe3\x81\x99' -'\xe3\x81\x9f' -'\xe3\x81\x9f\xe3' -'\xe3\x81\xa0' -'\xe3\x81\xa3' -'\xe3\x81\xa6' -'\xe3\x81\xa6\xe3' -'\xe3\x81\xa7' -'\xe3\x81\xa7\xe3' -'\xe3\x81\xa8' -'\xe3\x81\xa8\xe3' -'\xe3\x81\xaa' -'\xe3\x81\xaa\xe3' -'\xe3\x81\xab' -'\xe3\x81\xab\xe3' -'\xe3\x81\xae' -'\xe3\x81\xae\xe3' -'\xe3\x81\xae\xe5' -'\xe3\x81\xaf' -'\xe3\x81\xaf\xe3' -'\xe3\x81\xbe' -'\xe3\x82\x82' -'\xe3\x82\x88' -'\xe3\x82\x89' -'\xe3\x82\x8a' -'\xe3\x82\x8b' -'\xe3\x82\x8b\xe3' -'\xe3\x82\x8c' -'\xe3\x82\x92' -'\xe3\x82\xa4' -'\xe3\x82\xa4\xe3' -'\xe3\x82\xaf' -'\xe3\x82\xaf\xe3' -'\xe3\x82\xb9' -'\xe3\x82\xbf' -'\xe3\x82\xbf\xe3' -'\xe3\x83\x88' -'\xe3\x83\x88\xe3' -'\xe3\x83\x89' -'\xe3\x83\x8b' -'\xe3\x83\x8b\xe3' -'\xe3\x83\xa5' -'\xe3\x83\xa5\xe3' -'\xe3\x83\xab' -'\xe3\x83\xab\xe3' -'\xe3\x83\xad' -'\xe3\x83\xad\xe3' -'\xe3\x83\xb3' -'\xe3\x83\xb3\xe3' -'\xe3\x83\xbc' -'\xe3\x83\xbc\xe3' -'\xe3\xc7' -'\xe3\xc8' -'\xe3\xc9' -'\xe3\xcc' -'\xe3\xcd' -'\xe3\xd1' -'\xe3\xd3' -'\xe3\xd4' -'\xe3\xd5' -'\xe3\xda' -'\xe3\xdf' -'\xe3\xe1' -'\xe3\xe4' -'\xe3\xe5' -'\xe3\xe6' -'\xe3\xed' -'\xe4' -'\xe4 ' -'\xe4\xb9' -'\xe4\xbd' -'\xe4\xc7' -'\xe4\xc9' -'\xe4\xcc' -'\xe4\xcf' -'\xe4\xd3' -'\xe4\xd4' -'\xe4\xdf' -'\xe4\xe5' -'\xe4\xe6' -'\xe4\xed' -'\xe5' -'\xe5 ' -'\xe5\x90' -'\xe5\x9c' -'\xe5\x9c\xa8' -'\xe5\xb0' -'\xe5\xb8' -'\xe5\xc7' -'\xe5\xcf' -'\xe5\xd0' -'\xe6' -'\xe6\x96\xb0' -'\xe6\x9c\xac' -'\xe6\xb5' -'\xe6\xb7' -'\xe6\xbc' -'\xe6\xc7' -'\xe6\xcc' -'\xe6\xcf' -'\xe6\xd1' -'\xe6\xd3' -'\xe6\xda' -'\xe6\xde' -'\xe6\xe1' -'\xe6\xe3' -'\xe6\xe4' -'\xe6\xed' -'\xe7' -'\xe7\x82' -'\xe7\x9a' -'\xe7\x9a\x84' -'\xe7\xb9' -'\xe7\xba' -'\xe7\xbb' -'\xe7\xbe' -'\xe8' -'\xe8\xaa' -'\xe8\xae' -'\xe8\xb5' -'\xe8\xb7' -'\xe9' -'\xe9\x96' -'\xe9\x97' -'\xe9\xa1' -'\xe9\xbb' -'\xea\xb0' -'\xea\xb1' -'\xea\xb2' -'\xea\xb3' -'\xea\xb7' -'\xea\xb8' -'\xeb\x82' -'\xeb\x84' -'\xeb\x85' -'\xeb\x8a' -'\xeb\x8b' -'\xeb\x8c' -'\xeb\x8d' -'\xeb\x8f' -'\xeb\x90' -'\xeb\x93' -'\xeb\x94' -'\xeb\x9d' -'\xeb\x9e' -'\xeb\x9f' -'\xeb\xa0' -'\xeb\xa1' -'\xeb\xa5' -'\xeb\xa6' -'\xeb\xa7' -'\xeb\xa9' -'\xeb\xaa' -'\xeb\xac' -'\xeb\xaf' -'\xeb\xb0' -'\xeb\xb3' -'\xeb\xb6' -'\xec ' -'\xec\x82' -'\xec\x83' -'\xec\x84' -'\xec\x85' -'\xec\x86' -'\xec\x88' -'\xec\x8a' -'\xec\x8b' -'\xec\x95' -'\xec\x96' -'\xec\x97' -'\xec\x98' -'\xec\x99' -'\xec\x9a' -'\xec\x9b' -'\xec\x9c' -'\xec\x9d' -'\xec\x9e' -'\xec\xa0' -'\xec\xa2' -'\xec\xa4' -'\xec\xa6' -'\xec\xa7' -'\xec\xb0' -'\xec\xb2' -'\xec\xb6' -'\xed\x81' -'\xed\x83' -'\xed\x84' -'\xed\x85' -'\xed\x86' -'\xed\x8a' -'\xed\x8b' -'\xed\x8c' -'\xed\x8e' -'\xed\x8f' -'\xed\x91' -'\xed\x94' -'\xed\x95' -'\xed\x96' -'\xed\x98' -'\xed\x99' -'\xed\x9a' -'\xed\x9b' -'\xed\xc7' -'\xed\xc9' -'\xed\xcb' -'\xed\xcf' -'\xed\xd1' -'\xed\xda' -'\xed\xdd' -'\xed\xde' -'\xed\xdf' -'\xed\xe1' -'\xed\xe3' -'\xed\xe4' -'\xed\xe5' -'\xed\xe6' -'\xee\xd0' -'\xef' -'\xf8<' -'\xf8\xd5' -'\xf9\xd3' -'\xfa\xb5' -'\xfd' -'\xfe' diff --git a/HISTORY.rst b/HISTORY.rst index b6464318..4b130f5c 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -2,6 +2,23 @@ History ======= +0.4.0 +----- + +* New model: 139 languages + ``zxx`` (was 97) +* Reproducible training pipeline (``py3langid.train``) +* Faster inference; input NFC-normalized as in training +* Length-calibrated confidence normalization, with ``min_confidence`` + to return ``und`` below a threshold + +Breaking: + +* ``nb`` merged into ``no``; ``set_languages(["nb"])`` raises +* ``npz``\ +LZMA is the only model format (pickle removed) +* WSGI moved to ``py3langid.server:application`` +* invalid ``-m`` path raises instead of silent fallback +* ``--dist`` CSV gained a ``language`` column + 0.3.0 ----- diff --git a/MANIFEST.in b/MANIFEST.in new file mode 100644 index 00000000..19ccd66b --- /dev/null +++ b/MANIFEST.in @@ -0,0 +1,3 @@ +# the test suite and the training package are repository-only, not released +prune tests +prune py3langid/train diff --git a/README.rst b/README.rst index 37ad9deb..ea922576 100644 --- a/README.rst +++ b/README.rst @@ -12,70 +12,51 @@ Original license: BSD-2-Clause. Fork license: BSD-3-Clause. Changes in this fork -------------------- -Execution speed has been improved and the code base has been optimized for Python 3.6+: +Execution speed has been improved and the code base has been modernized for Python 3.10+: -- Import: Loading the package (``import py3langid``) is about 30% faster -- Startup: Loading the default classification model is 25-30x faster -- Execution: Language detection with ``langid.classify`` is 5-6x faster on paragraphs (less on longer texts) +- Import: Loading the package (``import py3langid``) is about 25% faster +- Execution: Language detection with ``langid.classify`` is 10x faster on single sentences and 3-4x faster on paragraphs (less on longer texts, about 1.4x at 100 kB) +- Startup: Loading the default classification model is 2-3x faster, with a model six times larger For implementation details see this blog post: `How to make language detection with langid.py faster `_. -For more information and older Python versions see `changelog `_. +The fork also ships a retrained model covering **139 languages** (up from 97) +and a fully rewritten, reproducible training pipeline (see `Training a model`_). + +For version history see the `changelog `_. Usage ----- -Drop-in replacement -~~~~~~~~~~~~~~~~~~~ - - -1. Install the package: - - * ``pip3 install py3langid`` (or ``pip`` where applicable) - -2. Use it: - - * with Python: ``import py3langid as langid`` - * on the command-line: ``langid`` - +Install: ``pip install py3langid`` — use as ``import py3langid as langid`` +or on the command-line as ``langid``. With Python ~~~~~~~~~~~ -Basics: - .. code-block:: python >>> import py3langid as langid - - >>> text = 'This text is in English.' - # identified language and probability - >>> langid.classify(text) - ('en', -56.77429) - # unpack the result tuple in variables - >>> lang, prob = langid.classify(text) - # all potential languages - >>> langid.rank(text) - -More options: - -.. code-block:: python + >>> langid.classify('This text is in English.') + ('en', -68.562286) + >>> langid.rank('This text is in English.') # all languages, most likely first >>> from py3langid.langid import LanguageIdentifier, MODEL_FILE - - # subset of target languages - >>> identifier = LanguageIdentifier.from_pickled_model(MODEL_FILE) + >>> identifier = LanguageIdentifier.from_model_file(MODEL_FILE, norm_probs=True) >>> identifier.set_languages(['de', 'en', 'fr']) - # this won't work well... - >>> identifier.classify('这样不好') - ('en', -81.831665) - - # normalization of probabilities to an interval between 0 and 1 - >>> identifier = LanguageIdentifier.from_pickled_model(MODEL_FILE, norm_probs=True) >>> identifier.classify('This should be enough text.') - ('en', 1.0) + ('en', 0.9999628) + + # abstention: return ('und', confidence) below a threshold + >>> identifier = LanguageIdentifier.from_model_file(MODEL_FILE, norm_probs=True, + ... min_confidence=0.2) + >>> identifier.classify('ok') + ('und', 0.0140845) + +Input can be ``str`` or UTF-8 ``bytes``; input is NFC-normalized before +classification, and all-uppercase text is case-folded. On the command-line @@ -85,233 +66,91 @@ On the command-line # basic usage with probability normalization $ echo "This should be enough text." | langid -n - ('en', 1.0) + ('en', 0.9935992) # define a subset of target languages $ echo "This won't be recognized properly." | langid -n -l fr,it,tr - ('it', 0.97038305) - - -Legacy documentation --------------------- - - -**The docs below are provided for reference, only part of the functions are currently tested and maintained.** - - -Introduction ------------- - -``langid.py`` is a standalone Language Identification (LangID) tool. - -The design principles are as follows: - -1. Fast -2. Pre-trained over a large number of languages (currently 97) -3. Not sensitive to domain-specific features (e.g. HTML/XML markup) -4. Single .py file with minimal dependencies -5. Deployable as a web service - -All that is required to run ``langid.py`` is Python >= 3.6 and numpy. - -The accompanying training tools are still Python2-only. - -``langid.py`` is WSGI-compliant. ``langid.py`` will use ``fapws3`` as a web server if -available, and default to ``wsgiref.simple_server`` otherwise. + ('fr', 0.4838270) -``langid.py`` comes pre-trained on 97 languages (ISO 639-1 codes given): +Run ``langid`` without input to get an interactive prompt, pipe text into it +to classify a whole document, or add ``--line`` to classify each line +separately. ``langid -u URL`` downloads and classifies a web page. See +``langid --help`` for all options. - af, am, an, ar, as, az, be, bg, bn, br, - bs, ca, cs, cy, da, de, dz, el, en, eo, - es, et, eu, fa, fi, fo, fr, ga, gl, gu, - he, hi, hr, ht, hu, hy, id, is, it, ja, - jv, ka, kk, km, kn, ko, ku, ky, la, lb, - lo, lt, lv, mg, mk, ml, mn, mr, ms, mt, - nb, ne, nl, nn, no, oc, or, pa, pl, ps, - pt, qu, ro, ru, rw, se, si, sk, sl, sq, - sr, sv, sw, ta, te, th, tl, tr, ug, uk, - ur, vi, vo, wa, xh, zh, zu -The training data was drawn from 5 different sources: - -* JRC-Acquis -* ClueWeb 09 -* Wikipedia -* Reuters RCV2 -* Debian i18n - - -Usage ------ - - langid [options] - -optional arguments: - -h, --help show this help message and exit - -s, --serve launch web service - --host=HOST host/ip to bind to - --port=PORT port to listen on - -v increase verbosity (repeat for greater effect) - -m MODEL load model from file - -l LANGS, --langs=LANGS - comma-separated set of target ISO639 language codes - (e.g en,de) - -r, --remote auto-detect IP address for remote access - -b, --batch specify a list of files on the command line - -d, --dist show full distribution over languages - -u URL, --url=URL langid of URL - --line process pipes line-by-line rather than as a document - -n, --normalize normalize confidence scores to probability values - - -The simplest way to use ``langid.py`` is as a command-line tool, and you can -invoke using ``python langid.py``. If you installed ``langid.py`` as a Python -module (e.g. via ``pip install langid``), you can invoke ``langid`` instead of -``python langid.py -n`` (the two are equivalent). This will cause a prompt to -display. Enter text to identify, and hit enter:: - - >>> This is a test - ('en', -54.41310358047485) - >>> Questa e una prova - ('it', -35.41771221160889) - - -``langid.py`` can also detect when the input is redirected (only tested under Linux), and in this -case will process until EOF rather than until newline like in interactive mode:: - - python langid.py < README.rst - ('en', -22552.496054649353) - - -The value returned is the unnormalized probability estimate for the language. Calculating -the exact probability estimate is disabled by default, but can be enabled through a flag:: - - python langid.py -n < README.rst - ('en', 1.0) - -More details are provided in this README in the section on `Probability Normalization`. - -You can also use ``langid.py`` as a Python library:: - - # python - Python 2.7.2+ (default, Oct 4 2011, 20:06:09) - [GCC 4.6.1] on linux2 - Type "help", "copyright", "credits" or "license" for more information. - >>> import langid - >>> langid.classify("This is a test") - ('en', -54.41310358047485) - -Finally, ``langid.py`` can use Python's built-in ``wsgiref.simple_server`` (or ``fapws3`` if available) to -provide language identification as a web service. To do this, launch ``python langid.py -s``, and -access http://localhost:9008/detect . The web service supports GET, POST and PUT. If GET is performed -with no data, a simple HTML forms interface is displayed. - -The response is generated in JSON, here is an example:: - - {"responseData": {"confidence": -54.41310358047485, "language": "en"}, "responseDetails": null, "responseStatus": 200} +Languages +--------- -A utility such as curl can be used to access the web service:: +The shipped model knows 139 languages plus ``zxx`` (ISO 639 codes):: - # curl -d "q=This is a test" localhost:9008/detect - {"responseData": {"confidence": -54.41310358047485, "language": "en"}, "responseDetails": null, "responseStatus": 200} + ace, af, am, an, ar, ary, arz, as, az, ba, bcl, be, bg, bn, br, bs, ca, + crh, cs, cy, da, de, dz, el, en, eo, es, et, eu, ext, fa, fi, fo, fr, + fuv, fy, ga, gcf, gcr, gd, gl, gom, grc, gu, gug, guw, ha, hbo, he, hi, + hr, ht, hu, hy, id, ig, is, it, ja, jv, ka, kab, kik, kk, km, kn, ko, + ku, ky, la, lb, lg, lij, ln, lo, lt, ltg, lv, mg, mk, ml, mn, mr, ms, + mt, my, ne, nl, nn, no, nso, oc, om, or, pa, pcm, pl, ps, pt, qu, ro, + ru, rw, sa, sdh, se, si, sk, sl, sn, so, sq, sr, st, sv, sw, ta, te, tg, + th, tk, tl, tr, tt, ug, uk, ur, uz, uzs, vec, vi, vo, wa, wuu, xh, yo, + yue, zh, zu, zxx -You can also use HTTP PUT:: +``zxx`` is a synthetic "not a language" class that catches numbers, markup, +identifiers, and similar non-linguistic content. With ``min_confidence`` +set, low-confidence predictions are returned as ``und`` (undetermined). - # curl -T readme.rst localhost:9008/detect - % Total % Received % Xferd Average Speed Time Time Time Current - Dload Upload Total Spent Left Speed - 100 2871 100 119 100 2752 117 2723 0:00:01 0:00:01 --:--:-- 2727 - {"responseData": {"confidence": -22552.496054649353, "language": "en"}, "responseDetails": null, "responseStatus": 200} -If no "q=XXX" key-value pair is present in the HTTP POST payload, ``langid.py`` will interpret the entire -file as a single query. This allows for redirection via curl:: +Batch mode +---------- - # echo "This is a test" | curl -d @- localhost:9008/detect - {"responseData": {"confidence": -54.41310358047485, "language": "en"}, "responseDetails": null, "responseStatus": 200} +``langid -b`` reads file paths from ``stdin`` (one per line) and classifies +the files in parallel, writing CSV to ``stdout``: -``langid.py`` will attempt to discover the host IP address automatically. Often, this is set to localhost(127.0.1.1), even -though the machine has a different external IP address. ``langid.py`` can attempt to automatically discover the external -IP address. To enable this functionality, start ``langid.py`` with the ``-r`` flag. +.. code-block:: bash -``langid.py`` supports constraining of the output language set using the ``-l`` flag and a comma-separated list of ISO639-1 -language codes (the ``-n`` flag enables probability normalization):: + $ find corpus -name "*.txt" | langid -b + corpus/a.txt,en,-127.32 + corpus/b.txt,de,-81.15 - # python langid.py -n -l it,fr - >>> Io non parlo italiano - ('it', 0.99999999988965627) - >>> Je ne parle pas français - ('fr', 1.0) - >>> I don't speak english - ('it', 0.92210605672341062) +With ``-d``, the output is one CSV row per file with the full score +distribution over all languages (one column per language). -When using ``langid.py`` as a library, the set_languages method can be used to constrain the language set:: - - python - Python 2.7.2+ (default, Oct 4 2011, 20:06:09) - [GCC 4.6.1] on linux2 - Type "help", "copyright", "credits" or "license" for more information. - >>> import langid - >>> langid.classify("I do not speak english") - ('en', 0.57133487679900674) - >>> langid.set_languages(['de','fr','it']) - >>> langid.classify("I do not speak english") - ('it', 0.99999835791478453) - >>> langid.set_languages(['en','it']) - >>> langid.classify("I do not speak english") - ('en', 0.99176190378750373) +Web service +----------- -Batch Mode ----------- +``langid -s`` serves language identification over HTTP (default port 9008). +Use the ``langid`` console script; ``python -m py3langid.langid`` is not +supported. +Endpoints ``/detect`` and ``/rank`` accept GET, POST, and PUT: -``langid.py`` supports batch mode processing, which can be invoked with the ``-b`` flag. -In this mode, ``langid.py`` reads a list of paths to files to classify as arguments. -If no arguments are supplied, ``langid.py`` reads the list of paths from ``stdin``, -this is useful for using ``langid.py`` with UNIX utilities such as ``find``. +.. code-block:: bash -In batch mode, ``langid.py`` uses ``multiprocessing`` to invoke multiple instances of -the classifier, utilizing all available CPUs to classify documents in parallel. + $ curl -d "q=This is a test" localhost:9008/detect +For production, use ``py3langid.server:application`` under a WSGI server. -Probability Normalization -------------------------- -The probabilistic model implemented by ``langid.py`` involves the multiplication of a -large number of probabilities. For computational reasons, the actual calculations are -implemented in the log-probability space (a common numerical technique for dealing with -vanishingly small probabilities). One side-effect of this is that it is not necessary to -compute a full probability in order to determine the most probable language in a set -of candidate languages. However, users sometimes find it helpful to have a "confidence" -score for the probability prediction. Thus, ``langid.py`` implements a re-normalization -that produces an output in the 0-1 range. +Custom models +------------- -``langid.py`` disables probability normalization by default. For -command-line usages of ``langid.py``, it can be enabled by passing the ``-n`` flag. For -probability normalization in library use, the user must instantiate their own -``LanguageIdentifier``. An example of such usage is as follows:: - - >> from py3langid.langid import LanguageIdentifier, MODEL_FILE - >> identifier = LanguageIdentifier.from_pickled_model(MODEL_FILE, norm_probs=True) - >> identifier.classify("This is a test") - ('en', 0.9999999909903544) +``langid -m FILE`` (or ``LanguageIdentifier.from_modelpath(path)``) loads a +model trained with this package (``model.npz.xz``). Models from the +original ``langid.py`` are not supported. Training a model ---------------- -So far Python 2.7 only, see the `original instructions `_. +``python -m py3langid.train.train -m model_dir corpus_dir``, run from a +clone of the repository (the training code is not part of the PyPI +package) — see +`TRAINING.md `_ +for corpus layout, data gathering, hygiene, and pipeline design. Training +is deterministic: the same corpus and settings reproduce the model byte for +byte. Read more --------- -``langid.py`` is based on published research. [1] describes the LD feature selection technique in detail, -and [2] provides more detail about the module ``langid.py`` itself. - -[1] Lui, Marco and Timothy Baldwin (2011) Cross-domain Feature Selection for Language Identification, -In Proceedings of the Fifth International Joint Conference on Natural Language Processing (IJCNLP 2011), -Chiang Mai, Thailand, pp. 553—561. Available from http://www.aclweb.org/anthology/I11-1062 - -[2] Lui, Marco and Timothy Baldwin (2012) langid.py: An Off-the-shelf Language Identification Tool, -In Proceedings of the 50th Annual Meeting of the Association for Computational Linguistics (ACL 2012), -Demo Session, Jeju, Republic of Korea. Available from www.aclweb.org/anthology/P12-3005 +| [1] Lui & Baldwin (2011) `Cross-domain Feature Selection for Language Identification `_, IJCNLP 2011. +| [2] Lui & Baldwin (2012) `langid.py: An Off-the-shelf Language Identification Tool `_, ACL 2012 Demo. diff --git a/TRAINING.md b/TRAINING.md new file mode 100644 index 00000000..54ad9435 --- /dev/null +++ b/TRAINING.md @@ -0,0 +1,142 @@ +# Training a Model + +The pipeline follows Lui & Baldwin (2011): a multi-domain corpus, LD +feature selection (per-language information gain minus domain information +gain), and Multinomial Naive Bayes over byte n-grams. Training is +deterministic — the same corpus and settings reproduce the model byte for +byte. + +The `py3langid.train` package ships in the repository, not in the PyPI +wheel, so run the commands below from a clone: + +```bash +git clone https://github.com/adbar/py3langid.git +cd py3langid +``` + +Training itself requires only `numpy` (already a dependency of py3langid). +Data gathering additionally needs `huggingface_hub` and `datasets` for the +top-up sources: `pip install huggingface_hub datasets`. + +## Corpus + +Layout: `corpus/{domain}/{lang}/docNNNN.txt`, UTF-8 bytes. A domain is a +text register (news, web, encyclopedia, ...); LD selection needs ≥2 domains +per language to separate language signal from domain signal, so more +domains generalize better. + +The released model is trained on 130,914 docs, capped at 300 docs per +language per domain and 3,000 bytes per doc (`DOC_CAP` in `common.py`, the +pipeline's one doc byte budget — gathering, tokenization, the verifier and +`zxx` all read it): + +| Domain | Source | Docs | Register | +|-------------|---------------------------|--------|----------------| +| wiki | cirrus dumps | 37,901 | encyclopedia | +| leipzig | Leipzig Corpora (news) | 30,538 | news | +| cc100 | CommonCrawl 2018 filtered | 29,006 | web | +| tatoeba | user sentences (CC BY) | 18,864 | conversational | +| glotcc | GlotCC-V1 (topup) | 9,381 | web | +| glot500 | Glot500 (topup) | 4,624 | mixed | +| glotsparse | GlotSparse (topup) | 600 | web | + +The last three are top-up sources: `topup.py` fills classes that fall +below 600 docs or 2 domains from GlotCC, Glot500, GlotSparse and, as a last +resort, UDHR; they are not gathered for every language. Two classes (`sdh`, +`uzs`) exist only in GlotSparse. + +Classes: 139 languages + `zxx` (synthetic not-a-language) + two internal +script-split classes = 142 NB classes / 140 public labels. A language +written in two scripts trains as two classes and is merged back to one +label at model assembly (`sr` Cyrillic + internal `srl` Latin; same for +`uz`/`uzc`) — a single class spanning two scripts dilutes its weights. +Adding a split language is one `SplitScript` entry in `common.py`. + +## Gathering data + +``` +python -m py3langid.train.gather_data \ + --output corpus \ + --langs af,am,...,zu # default: model languages + --domains tatoeba,cc100,wiki,leipzig # add 'topup' to fill thin classes + --max-docs-per-lang 300 + --sentences-per-doc 50 + --jobs 4 +``` + +`topup` is not in the default domain list, so pass `--domains` explicitly +for a corpus that covers every shipped label. Downloads are cached in +`raw_downloads/` and re-gathers resume where they stopped. Run the stages +below in order on the result, then evaluate before adopting the model. + +## Corpus hygiene + +On a fresh corpus, in order: + +``` +python -m py3langid.train.zxx corpus # generate the not-a-language class +python -m py3langid.train.dedup corpus # cross-domain exact line dedup +python -m py3langid.train.verify --model VERIFIER_MODEL corpus +``` + +`verify` classifies every doc and moves docs predicted as a *different, +non-confusable* language to a sibling `corpus_dropped/` tree (nothing is +deleted). Confusable pairs (bs/hr/sr, ms/id, no/nn/da, ...) are protected: +dropping there would push the pair boundary toward the verifier's own +bias. Use an independent verifier for the first round (e.g. the previous +release model or any model not trained on this corpus) — a model trained +on the corpus itself accepts the contamination it learned. After one clean +round, the retrained model is a valid self-verifier. `--paragraphs` +strips foreign paragraphs in place instead of dropping docs. + +## Training run + +``` +python -m py3langid.train.train -m model_dir corpus +``` + +Bare defaults reproduce the release config, byte for byte. A run with a +warm shard cache takes about a minute at `-j 10`; a cold one adds the +tokenization pass. + +`--feats_per_lang` is the only sweep knob as a flag. Tokenization +constants live in `common.py` (`MIN_NGRAM_ORDER`, `MAX_NGRAM_ORDER`, +`SELECT_ORDERS`, `DF_TOKENS`, `DOC_CAP`) — sweep by editing; the shard +cache invalidates itself. + +How it works: + +- Tokenization writes per-(domain, lang) document-frequency shards cached + at `CORPUS_DIR.shards` (`--shards` overrides). Only changed directories + are re-tokenized; all later stages are algebra over the shards. +- Order 6 is restricted to CJK codepoint bigrams (byte order 5 cannot + span two 3-byte codepoints). +- Feature selection: top `DF_TOKENS` terms per order as candidates, then + top `feats_per_lang` per language by LD weight, restricted to terms + appearing in ≥2 domains. +- Confusable clusters (`CLUSTERS` in `common.py`) each add 150 features + by cluster-restricted IG. Change them only with a full re-evaluation. +- Class priors are `log(per-class doc counts)`, no smoothing knobs. +- The compiled Aho-Corasick scanner emits only the **longest** matching + feature at each byte position (`out_feat` in the model), avoiding + duplicate votes from nested n-grams. NB numerators come from a scanner + pass over the corpus (`feature_counts`) that runs the runtime's own walk, + so training counts exactly what classification accumulates. + +## Evaluation + +The eval harness is not part of the repository; the datasets are large and +licensed separately. The split used for this model: + +- Dev (tuned on): WiLI-2018 and OpenLID, scored against a stored baseline + per language so a run reports deltas rather than absolute numbers. +- Held-out (run once per adoption): FLORES-200 and CommonLID + (noisy-register web text — decides ties). +- Caveats: dev sets are formal register; some confusable pairs (bs/hr + especially) are genuinely multi-valid with an intrinsic accuracy ceiling. + +## Results + +Shipped model: **WiLI 95.51 / OpenLID 94.55**, 140 labels, 100,053 +features, 4.6 MB. The pre-fork langid.py model scores 91.21/88.41 on the +same harnesses. diff --git a/py3langid/__init__.py b/py3langid/__init__.py index 8b1baeb3..4e213b01 100644 --- a/py3langid/__init__.py +++ b/py3langid/__init__.py @@ -1,3 +1,3 @@ from .langid import classify, rank, set_languages -__version__ = '0.3.0' +__version__ = '0.4.0' diff --git a/py3langid/data/model.npz.xz b/py3langid/data/model.npz.xz new file mode 100644 index 00000000..2221d7ba Binary files /dev/null and b/py3langid/data/model.npz.xz differ diff --git a/py3langid/data/model.plzma b/py3langid/data/model.plzma deleted file mode 100644 index 5220a59d..00000000 Binary files a/py3langid/data/model.plzma and /dev/null differ diff --git a/py3langid/examples/_twokenize.py b/py3langid/examples/_twokenize.py deleted file mode 100644 index cf22088b..00000000 --- a/py3langid/examples/_twokenize.py +++ /dev/null @@ -1,297 +0,0 @@ -# -*- coding: utf-8 -*- -""" -Twokenize -- a tokenizer designed for Twitter text in English and some other European languages. -This tokenizer code has gone through a long history: - -(1) Brendan O'Connor wrote original version in Python, http://github.com/brendano/tweetmotif - TweetMotif: Exploratory Search and Topic Summarization for Twitter. - Brendan O'Connor, Michel Krieger, and David Ahn. - ICWSM-2010 (demo track), http://brenocon.com/oconnor_krieger_ahn.icwsm2010.tweetmotif.pdf -(2a) Kevin Gimpel and Daniel Mills modified it for POS tagging for the CMU ARK Twitter POS Tagger -(2b) Jason Baldridge and David Snyder ported it to Scala -(3) Brendan bugfixed the Scala port and merged with POS-specific changes - for the CMU ARK Twitter POS Tagger -(4) Tobi Owoputi ported it back to Java and added many improvements (2012-06) - -Current home is http://github.com/brendano/ark-tweet-nlp and http://www.ark.cs.cmu.edu/TweetNLP - -There have been at least 2 other Java ports, but they are not in the lineage for the code here. - -Ported to Python by Myle Ott . -""" - -from __future__ import print_function - -import operator -import re -import HTMLParser - -def regex_or(*items): - return '(?:' + '|'.join(items) + ')' - -Contractions = re.compile(u"(?i)(\w+)(n['’′]t|['’′]ve|['’′]ll|['’′]d|['’′]re|['’′]s|['’′]m)$", re.UNICODE) -Whitespace = re.compile(u"[\s\u0020\u00a0\u1680\u180e\u202f\u205f\u3000\u2000-\u200a]+", re.UNICODE) - -punctChars = r"['\"“”‘’.?!…,:;]" -#punctSeq = punctChars+"+" #'anthem'. => ' anthem '. -punctSeq = r"['\"“”‘’]+|[.?!,…]+|[:;]+" #'anthem'. => ' anthem ' . -entity = r"&(?:amp|lt|gt|quot);" -# URLs - - -# BTO 2012-06: everyone thinks the daringfireball regex should be better, but they're wrong. -# If you actually empirically test it the results are bad. -# Please see https://github.com/brendano/ark-tweet-nlp/pull/9 - -urlStart1 = r"(?:https?://|\bwww\.)" -commonTLDs = r"(?:com|org|edu|gov|net|mil|aero|asia|biz|cat|coop|info|int|jobs|mobi|museum|name|pro|tel|travel|xxx)" -ccTLDs = r"(?:ac|ad|ae|af|ag|ai|al|am|an|ao|aq|ar|as|at|au|aw|ax|az|ba|bb|bd|be|bf|bg|bh|bi|bj|bm|bn|bo|br|bs|bt|" + \ -r"bv|bw|by|bz|ca|cc|cd|cf|cg|ch|ci|ck|cl|cm|cn|co|cr|cs|cu|cv|cx|cy|cz|dd|de|dj|dk|dm|do|dz|ec|ee|eg|eh|" + \ -r"er|es|et|eu|fi|fj|fk|fm|fo|fr|ga|gb|gd|ge|gf|gg|gh|gi|gl|gm|gn|gp|gq|gr|gs|gt|gu|gw|gy|hk|hm|hn|hr|ht|" + \ -r"hu|id|ie|il|im|in|io|iq|ir|is|it|je|jm|jo|jp|ke|kg|kh|ki|km|kn|kp|kr|kw|ky|kz|la|lb|lc|li|lk|lr|ls|lt|" + \ -r"lu|lv|ly|ma|mc|md|me|mg|mh|mk|ml|mm|mn|mo|mp|mq|mr|ms|mt|mu|mv|mw|mx|my|mz|na|nc|ne|nf|ng|ni|nl|no|np|" + \ -r"nr|nu|nz|om|pa|pe|pf|pg|ph|pk|pl|pm|pn|pr|ps|pt|pw|py|qa|re|ro|rs|ru|rw|sa|sb|sc|sd|se|sg|sh|si|sj|sk|" + \ -r"sl|sm|sn|so|sr|ss|st|su|sv|sy|sz|tc|td|tf|tg|th|tj|tk|tl|tm|tn|to|tp|tr|tt|tv|tw|tz|ua|ug|uk|us|uy|uz|" + \ -r"va|vc|ve|vg|vi|vn|vu|wf|ws|ye|yt|za|zm|zw)" #TODO: remove obscure country domains? -urlStart2 = r"\b(?:[A-Za-z\d-])+(?:\.[A-Za-z0-9]+){0,3}\." + regex_or(commonTLDs, ccTLDs) + r"(?:\."+ccTLDs+r")?(?=\W|$)" -urlBody = r"(?:[^\.\s<>][^\s<>]*?)?" -urlExtraCrapBeforeEnd = regex_or(punctChars, entity) + "+?" -urlEnd = r"(?:\.\.+|[<>]|\s|$)" -url = regex_or(urlStart1, urlStart2) + urlBody + "(?=(?:"+urlExtraCrapBeforeEnd+")?"+urlEnd+")" - - -# Numeric -timeLike = r"\d+(?::\d+){1,2}" -#numNum = r"\d+\.\d+" -numberWithCommas = r"(?:(?|>)[\._-]+(?:<|<|>|>)" -s5 = "(?:[.][_]+[.])" -# myleott: in Python the (?i) flag affects the whole expression -#basicface = "(?:(?i)" +bfLeft+bfCenter+bfRight+ ")|" +s3+ "|" +s4+ "|" + s5 -basicface = "(?:" +bfLeft+bfCenter+bfRight+ ")|" +s3+ "|" +s4+ "|" + s5 - -eeLeft = r"[\\\ƪԄ\((<>;ヽ\-=~\*]+" -eeRight= u"[\\-=\\);'\u0022<>ʃ)//ノノ丿╯σっµ~\\*]+".encode('utf-8') -eeSymbol = r"[^A-Za-z0-9\s\(\)\*:=-]" -eastEmote = eeLeft + "(?:"+basicface+"|" +eeSymbol+")+" + eeRight - -oOEmote = r"(?:[oO]" + bfCenter + r"[oO])" - - -emoticon = regex_or( - # Standard version :) :( :] :D :P - "(?:>|>)?" + regex_or(normalEyes, wink) + regex_or(noseArea,"[Oo]") + regex_or(tongue+r"(?=\W|$|RT|rt|Rt)", otherMouths+r"(?=\W|$|RT|rt|Rt)", sadMouths, happyMouths), - - # reversed version (: D: use positive lookbehind to remove "(word):" - # because eyes on the right side is more ambiguous with the standard usage of : ; - regex_or("(?<=(?: ))", "(?<=(?:^))") + regex_or(sadMouths,happyMouths,otherMouths) + noseArea + regex_or(normalEyes, wink) + "(?:<|<)?", - - #inspired by http://en.wikipedia.org/wiki/User:Scapler/emoticons#East_Asian_style - eastEmote.replace("2", "1", 1), basicface, - # iOS 'emoji' characters (some smileys, some symbols) [\ue001-\uebbb] - # TODO should try a big precompiled lexicon from Wikipedia, Dan Ramage told me (BTO) he does this - - # myleott: o.O and O.o are two of the biggest sources of differences - # between this and the Java version. One little hack won't hurt... - oOEmote -) - -Hearts = "(?:<+/?3+)+" #the other hearts are in decorations - -Arrows = regex_or(r"(?:<*[-―—=]*>+|<+[-―—=]*>*)", u"[\u2190-\u21ff]+".encode('utf-8')) - -# BTO 2011-06: restored Hashtag, AtMention protection (dropped in original scala port) because it fixes -# "hello (#hashtag)" ==> "hello (#hashtag )" WRONG -# "hello (#hashtag)" ==> "hello ( #hashtag )" RIGHT -# "hello (@person)" ==> "hello (@person )" WRONG -# "hello (@person)" ==> "hello ( @person )" RIGHT -# ... Some sort of weird interaction with edgepunct I guess, because edgepunct -# has poor content-symbol detection. - -# This also gets #1 #40 which probably aren't hashtags .. but good as tokens. -# If you want good hashtag identification, use a different regex. -Hashtag = "#[a-zA-Z0-9_]+" #optional: lookbehind for \b -#optional: lookbehind for \b, max length 15 -AtMention = "[@@][a-zA-Z0-9_]+" - -# I was worried this would conflict with at-mentions -# but seems ok in sample of 5800: 7 changes all email fixes -# http://www.regular-expressions.info/email.html -Bound = r"(?:\W|^|$)" -Email = regex_or("(?<=(?:\W))", "(?<=(?:^))") + r"[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,4}(?=" +Bound+")" - -# We will be tokenizing using these regexps as delimiters -# Additionally, these things are "protected", meaning they shouldn't be further split themselves. -Protected = re.compile( - unicode(regex_or( - Hearts, - url, - Email, - timeLike, - #numNum, - numberWithCommas, - numComb, - emoticon, - Arrows, - entity, - punctSeq, - arbitraryAbbrev, - separators, - decorations, - embeddedApostrophe, - Hashtag, - AtMention - ).decode('utf-8')), re.UNICODE) - -# Edge punctuation -# Want: 'foo' => ' foo ' -# While also: don't => don't -# the first is considered "edge punctuation". -# the second is word-internal punctuation -- don't want to mess with it. -# BTO (2011-06): the edgepunct system seems to be the #1 source of problems these days. -# I remember it causing lots of trouble in the past as well. Would be good to revisit or eliminate. - -# Note the 'smart quotes' (http://en.wikipedia.org/wiki/Smart_quotes) -#edgePunctChars = r"'\"“”‘’«»{}\(\)\[\]\*&" #add \\p{So}? (symbols) -edgePunctChars = u"'\"“”‘’«»{}\\(\\)\\[\\]\\*&" #add \\p{So}? (symbols) -edgePunct = "[" + edgePunctChars + "]" -notEdgePunct = "[a-zA-Z0-9]" # content characters -offEdge = r"(^|$|:|;|\s|\.|,)" # colon here gets "(hello):" ==> "( hello ):" -EdgePunctLeft = re.compile(offEdge + "("+edgePunct+"+)("+notEdgePunct+")", re.UNICODE) -EdgePunctRight = re.compile("("+notEdgePunct+")("+edgePunct+"+)" + offEdge, re.UNICODE) - -def splitEdgePunct(input): - input = EdgePunctLeft.sub(r"\1\2 \3", input) - input = EdgePunctRight.sub(r"\1 \2\3", input) - return input - -# The main work of tokenizing a tweet. -def simpleTokenize(text): - - # Do the no-brainers first - splitPunctText = splitEdgePunct(text) - - textLength = len(splitPunctText) - - # BTO: the logic here got quite convoluted via the Scala porting detour - # It would be good to switch back to a nice simple procedural style like in the Python version - # ... Scala is such a pain. Never again. - - # Find the matches for subsequences that should be protected, - # e.g. URLs, 1.0, U.N.K.L.E., 12:53 - bads = [] - badSpans = [] - for match in Protected.finditer(splitPunctText): - # The spans of the "bads" should not be split. - if (match.start() != match.end()): #unnecessary? - bads.append( [splitPunctText[match.start():match.end()]] ) - badSpans.append( (match.start(), match.end()) ) - - # Create a list of indices to create the "goods", which can be - # split. We are taking "bad" spans like - # List((2,5), (8,10)) - # to create - # List(0, 2, 5, 8, 10, 12) - # where, e.g., "12" here would be the textLength - # has an even length and no indices are the same - indices = [0] - for (first, second) in badSpans: - indices.append(first) - indices.append(second) - indices.append(textLength) - - # Group the indices and map them to their respective portion of the string - splitGoods = [] - for i in range(0, len(indices), 2): - goodstr = splitPunctText[indices[i]:indices[i+1]] - splitstr = goodstr.strip().split(" ") - splitGoods.append(splitstr) - - # Reinterpolate the 'good' and 'bad' Lists, ensuring that - # additonal tokens from last good item get included - zippedStr = [] - for i in range(len(bads)): - zippedStr = addAllnonempty(zippedStr, splitGoods[i]) - zippedStr = addAllnonempty(zippedStr, bads[i]) - zippedStr = addAllnonempty(zippedStr, splitGoods[len(bads)]) - - # BTO: our POS tagger wants "ur" and "you're" to both be one token. - # Uncomment to get "you 're" - #splitStr = [] - #for tok in zippedStr: - # splitStr.extend(splitToken(tok)) - #zippedStr = splitStr - - return zippedStr - -def addAllnonempty(master, smaller): - for s in smaller: - strim = s.strip() - if (len(strim) > 0): - master.append(strim) - return master - -# "foo bar " => "foo bar" -def squeezeWhitespace(input): - return Whitespace.sub(" ", input).strip() - -# Final pass tokenization based on special patterns -def splitToken(token): - m = Contractions.search(token) - if m: - return [m.group(1), m.group(2)] - return [token] - -# Assume 'text' has no HTML escaping. -def tokenize(text): - return simpleTokenize(squeezeWhitespace(text)) - - -# Twitter text comes HTML-escaped, so unescape it. -# We also first unescape &'s, in case the text has been buggily double-escaped. -def normalizeTextForTagger(text): - text = text.replace("&", "&") - text = HTMLParser.HTMLParser().unescape(text) - return text - -# This is intended for raw tweet text -- we do some HTML entity unescaping before running the tagger. -# -# This function normalizes the input text BEFORE calling the tokenizer. -# So the tokens you get back may not exactly correspond to -# substrings of the original text. -def tokenizeRawTweetText(text): - return tokenize(normalizeTextForTagger(text)) diff --git a/py3langid/examples/process_twitter.py b/py3langid/examples/process_twitter.py deleted file mode 100644 index f79719d9..00000000 --- a/py3langid/examples/process_twitter.py +++ /dev/null @@ -1,63 +0,0 @@ -""" -Example for using langid.py to identify the language of messages -on a twitter livestream. Optionally, it can also filter messages -and display only those in a target language(s). - -Expects a Twitterstream on STDIN, such as the one provided by: - -# curl https://stream.twitter.com/1/statuses/sample.json -u -s - -Outputs lang:message one-per-line to STDOUT - -Marco Lui, June 2012 -""" - -import sys -import langid -import json -import optparse -import re - -import _twokenize - - -to_clean = re.compile(_twokenize.regex_or( - _twokenize.Hearts, - _twokenize.url, - _twokenize.Email, - _twokenize.emoticon, - _twokenize.Arrows, - _twokenize.entity, - _twokenize.decorations, - _twokenize.Hashtag, - _twokenize.AtMention, -).decode('utf8'), re.UNICODE) - - -def clean_tweet(text): - return to_clean.sub('', text) - - -def squeeze_whitespace(text): - return re.sub('\s+', ' ', text) - - -if __name__ == "__main__": - parser = optparse.OptionParser() - parser.add_option('-l', '--langs', dest='langs', help='comma-separated set of target ISO639 language codes (e.g en,de)') - opts, args = parser.parse_args() - - lang_set = set(opts.langs.split(",")) if opts.langs else None - - try: - for line in sys.stdin: - j = json.loads(line) - if j.get('retweet_count') == 0: - text = j.get('text') - if text: - lang, conf = langid.classify(clean_tweet(text)) - if lang_set is None or lang in lang_set: - print "{0}: {1}".format(lang, squeeze_whitespace(text).encode('utf8')) - except (IOError, KeyboardInterrupt): - # Terminate on broken pipe or ^C - pass diff --git a/py3langid/langid.py b/py3langid/langid.py index 76224b13..d711eeee 100755 --- a/py3langid/langid.py +++ b/py3langid/langid.py @@ -1,54 +1,63 @@ #!/usr/bin/env python3 -""" -This file bundles language identification functions. +"""Language identification (fork of langid.py by Marco Lui).""" -Modifications (fork): Copyright (c) 2021, Adrien Barbaresi. - -Original code: Copyright (c) 2011 Marco Lui . -Based on research by Marco Lui and Tim Baldwin. - -See LICENSE file for more info. -""" - -import bz2 -import json import logging -import lzma -import pickle -from base64 import b64decode +import math +import unicodedata from collections import Counter -from http import HTTPStatus from operator import itemgetter from pathlib import Path -from urllib.parse import parse_qs import numpy as np +from .modelio import load_model as _load_model_file + LOGGER = logging.getLogger(__name__) IDENTIFIER = None -MODEL_FILE = 'data/model.plzma' +MODEL_FILE = 'data/model.npz.xz' MODEL_DIR = Path(__file__).parent +RAW_FLOOR = float(np.finfo(np.float32).min) # finite floor for featureless input + + +def decode_trimmed(data): + """Decode UTF-8, trimming ≤3 partial trailing bytes; None if undecodable. + Shared train/inference contract (also used by train.common.nfc_bytes).""" + for trim in range(4): + chunk = data[:len(data) - trim] if trim else data + try: + return chunk.decode('utf8') + except UnicodeDecodeError as e: + if e.start < len(data) - 3: # not fixable by trimming the tail + return None + return None + + +def visit_counts(nm, rowbase, out, text): + """DFA-walk feature counts over bytes; None if none. + Shared by inference (_raw_score) and training (train.stages).""" + state, indexes = 0, [] + append = indexes.append + for letter in text: + state = nm[rowbase[state] + letter] + f = out[state] + if f >= 0: + append(f) + return Counter(indexes) if indexes else None def _load_identifier(model_path=None, norm_probs=False, langs=None): - """Load an identifier: external model if given, else the bundled one.""" - identifier = None if model_path: - try: - identifier = LanguageIdentifier.from_modelpath(model_path, norm_probs=norm_probs) - LOGGER.info("Using external model: %s", model_path) - except OSError as e: - LOGGER.warning("Failed to load %s: %s", model_path, e) - if identifier is None: - identifier = LanguageIdentifier.from_pickled_model(MODEL_FILE, norm_probs=norm_probs) + identifier = LanguageIdentifier.from_modelpath(model_path, norm_probs=norm_probs) + LOGGER.info("Using external model: %s", model_path) + else: + identifier = LanguageIdentifier.from_model_file(MODEL_FILE, norm_probs=norm_probs) if langs: identifier.set_languages(langs) return identifier def _get_identifier(): - """Return the global identifier, loading the default model if needed.""" global IDENTIFIER if IDENTIFIER is None: LOGGER.debug('initializing identifier') @@ -57,24 +66,21 @@ def _get_identifier(): def set_languages(langs=None): - """Set the language subset used by the global identifier.""" return _get_identifier().set_languages(langs) def classify(instance): - """Classify a text string, returning (language, confidence).""" return _get_identifier().classify(instance) def rank(instance): - """Rank all languages by likelihood, returning [(language, confidence), ...].""" return _get_identifier().rank(instance) def _init_worker(model_path, norm_probs, langs): - # spawned Pool workers get a fresh module: rebuild the parent's identifier global IDENTIFIER - IDENTIFIER = _load_identifier(model_path, norm_probs, langs) + if IDENTIFIER is None: # forked workers inherit the parent's identifier + IDENTIFIER = _load_identifier(model_path, norm_probs, langs) def _process_file(path, dist=False): @@ -85,54 +91,69 @@ def _process_file(path, dist=False): class LanguageIdentifier: __slots__ = [ + '_alias_pairs', '_full_model', '_norm_probs', + '_rowbase', + 'min_confidence', 'nb_classes', + 'nb_pc', 'nb_ptc', 'tk_nextmove', 'tk_output', + 'tk_row', ] @classmethod - def _from_model_data(cls, nb_ptc, _nb_pc, nb_classes, tk_nextmove, tk_output, *args, **kwargs): - n_classes = len(nb_classes) - nb_ptc = np.array(nb_ptc).reshape(len(nb_ptc) // n_classes, n_classes) - # {state: features} dict -> dense list; 256 transitions per DFA state - output_list = [None] * (len(tk_nextmove) // 256) - for s, v in tk_output.items(): - output_list[s] = v - return cls(nb_ptc, nb_classes, tk_nextmove, output_list, *args, **kwargs) - - @classmethod - def from_pickled_model(cls, pickled_file, *args, **kwargs): - with lzma.open(MODEL_DIR / pickled_file) as f: - data = pickle.load(f) - return cls._from_model_data(*data, *args, **kwargs) - - @classmethod - def from_modelstring(cls, string, *args, **kwargs): - data = pickle.loads(bz2.decompress(b64decode(string))) - return cls._from_model_data(*data, *args, **kwargs) + def from_model_file(cls, model_file, *args, **kwargs): + filepath = Path(model_file) + if not filepath.is_absolute(): + filepath = MODEL_DIR / filepath + ptc, pc, classes, nextmove, row, output = _load_model_file(filepath) + return cls(np.asarray(ptc), np.asarray(pc), classes, nextmove, output, + *args, tk_row=row, **kwargs) @classmethod def from_modelpath(cls, path, *args, **kwargs): - with open(path, 'rb') as f: - return cls.from_modelstring(f.read(), *args, **kwargs) + return cls.from_model_file(Path(path).absolute(), *args, **kwargs) - def __init__(self, nb_ptc, nb_classes, tk_nextmove, tk_output, norm_probs=False): + def __init__(self, nb_ptc, nb_pc, nb_classes, tk_nextmove, tk_output, + norm_probs=False, min_confidence=None, *, tk_row): + if min_confidence is not None and not norm_probs: + raise ValueError("min_confidence requires norm_probs=True") + self.min_confidence = min_confidence self.nb_ptc = nb_ptc + self.nb_pc = nb_pc self.nb_classes = nb_classes self.tk_nextmove = tk_nextmove + self.tk_row = tk_row + self._rowbase = [r << 8 for r in tk_row] # pre-shifted row offsets self.tk_output = tk_output self._norm_probs = norm_probs - self._full_model = nb_ptc, nb_classes + self._full_model = nb_ptc, nb_pc, nb_classes + self._set_alias_pairs() + + def _set_alias_pairs(self): + """(first, dupe) column pairs for labels appearing more than once.""" + first, pairs = {}, [] + for i, c in enumerate(self.nb_classes): + if c in first: + pairs.append((first[c], i)) + else: + first[c] = i + self._alias_pairs = pairs + + @property + def labels(self): + "Distinct output labels; script aliases (srl->sr) share one." + return list(dict.fromkeys(self.nb_classes)) def set_languages(self, langs=None): + """Restrict classification to *langs* (ISO 639 codes), or reset to all.""" LOGGER.debug("restricting languages to: %s", langs) - nb_ptc, nb_classes = self._full_model - + nb_ptc, nb_pc, nb_classes = self._full_model if langs is None: - self.nb_classes, self.nb_ptc = nb_classes, nb_ptc + self.nb_classes, self.nb_ptc, self.nb_pc = nb_classes, nb_ptc, nb_pc else: lang_set = set(langs) unknown = lang_set - set(nb_classes) @@ -142,119 +163,74 @@ def set_languages(self, langs=None): indices = [i for i, c in enumerate(nb_classes) if c in lang_set] self.nb_classes = [nb_classes[i] for i in indices] self.nb_ptc = nb_ptc[:, indices] + self.nb_pc = nb_pc[indices] + self._set_alias_pairs() - def _score(self, text): + @staticmethod + def _encode(text): if isinstance(text, bytes): - # decode so case normalization applies uniformly to str and bytes - try: - text = text.decode('utf8') - except UnicodeDecodeError: - pass + decoded = decode_trimmed(text) + if decoded is not None: + text = decoded if isinstance(text, str): if text.isupper(): text = text.lower() + text = unicodedata.normalize('NFC', text) text = text.encode('utf8', errors='surrogatepass') - - # DFA walk - state, indexes = 0, [] - extend = indexes.extend - nm, out = self.tk_nextmove, self.tk_output - - for letter in text: - state = nm[(state << 8) + letter] - v = out[state] - if v: - extend(v) - - if indexes: - feat_counts = Counter(indexes) - idx = np.fromiter(feat_counts.keys(), dtype=np.intp, count=len(feat_counts)) - counts = np.fromiter(feat_counts.values(), dtype=np.float32, count=len(feat_counts)) - probs = counts @ self.nb_ptc[idx] - else: - # no features: minimal confidence (softmax turns this into uniform) - fill = 0.0 if self._norm_probs else -np.inf - probs = np.full(len(self.nb_classes), fill, dtype=np.float32) - + return text + + def _sparse_score(self, visits, table): + """NB log-posterior from sparse {feature: count}.""" + idx = np.fromiter(visits.keys(), dtype=np.intp, count=len(visits)) + counts = np.fromiter(visits.values(), dtype=np.float32, count=len(visits)) + return np.log1p(counts) @ table[idx] + self.nb_pc + + def _raw_score(self, text): + """Raw NB scores via DFA walk over encoded bytes.""" + visits = visit_counts(self.tk_nextmove, self._rowbase, self.tk_output, + text) + if visits: + return self._sparse_score(visits, self.nb_ptc) + + # no features: 0.0 under norm_probs (uniform → abstain), RAW_FLOOR otherwise + fill = 0.0 if self._norm_probs else RAW_FLOOR + return np.full(len(self.nb_classes), fill, dtype=np.float32) + + def _decide(self, text): + """Score per class, optionally normalized to probabilities.""" + text = self._encode(text) + scores = self._raw_score(text) if self._norm_probs: - e = np.exp(probs - probs.max()) - probs = e / e.sum() - - return probs + # T = sqrt(bytes) keeps softmax calibrated across lengths + scores *= 1.0 / math.sqrt(len(text) or 1) + np.exp(scores - scores.max(), out=scores) + scores /= scores.sum() + # aliased columns (srl->sr): fold the dupe into the first occurrence and + # mask it, so argmax and rank agree on one score per label + for i, j in self._alias_pairs: + if self._norm_probs: + scores[i] += scores[j] + scores[j] = 0.0 + else: + scores[i] = max(scores[i], scores[j]) + scores[j] = RAW_FLOOR + return scores def classify(self, text): - probs = self._score(text) - cl = probs.argmax() - return self.nb_classes[cl], float(probs[cl]) + """Return *(language, confidence)* for *text* (str or UTF-8 bytes).""" + scores = self._decide(text) + i = int(scores.argmax()) + conf = float(scores[i]) + if self.min_confidence is not None and conf < self.min_confidence: + return 'und', conf + return self.nb_classes[i], conf def rank(self, text): - probs = self._score(text) - return sorted( - ((lang, float(p)) for lang, p in zip(self.nb_classes, probs)), - key=itemgetter(1), reverse=True, - ) - - -def _detect(data): - lang, conf = classify(data) - return {'language': lang, 'confidence': conf} - - -_ROUTES = {'detect': _detect, 'rank': rank} - - -def application(environ, start_response): - """WSGI-compatible langid web service.""" - path = environ.get('PATH_INFO', '').strip('/').partition('/')[0] - handler = _ROUTES.get(path) - if handler is None: - return _return_response(start_response, 404, None, 'Not found') - - method = environ['REQUEST_METHOD'] - if method not in ('GET', 'POST', 'PUT'): - return _return_response(start_response, 405, None, f'{method} not allowed') - - data = _get_data(environ) - if data is None: - return _return_response(start_response, 400, None, 'No data provided') - - return _return_response(start_response, 200, handler(data), None) - - -def _get_data(environ): - method = environ['REQUEST_METHOD'] - if method in ('PUT', 'POST'): - try: - length = int(environ.get('CONTENT_LENGTH', 0)) - except ValueError: - return None - if length <= 0: - return None - data = environ['wsgi.input'].read(length) - if method == 'POST': - try: - data = parse_qs(data)[b'q'][0] - except KeyError: - pass - return data - if method == 'GET': - try: - return parse_qs(environ.get('QUERY_STRING', ''))['q'][0] - except KeyError: - return None - return None - - -def _return_response(start_response, status_code, response_data, response_details): - status = HTTPStatus(status_code) - response = { - 'responseData': response_data, - 'responseStatus': status_code, - 'responseDetails': response_details, - } - headers = [('Content-type', 'application/json; charset=utf-8')] - start_response(f"{status.value} {status.phrase}", headers) - return [json.dumps(response).encode('utf-8')] + """All languages by likelihood, best first, one entry per label.""" + merged = {} + for lang, score in zip(self.nb_classes, self._decide(text).tolist()): + merged.setdefault(lang, score) # first column holds the merged score + return sorted(merged.items(), key=itemgetter(1), reverse=True) def main(): @@ -270,9 +246,9 @@ def main(): parser.add_argument('-m', dest='model', help='load model from file') parser.add_argument('-l', '--langs', help='comma-separated set of target ISO639 language codes (e.g en,de)') parser.add_argument('-r', '--remote', action='store_true', help='auto-detect IP address for remote access') - parser.add_argument('-b', '--batch', action='store_true', help='specify a list of files on the command line') + parser.add_argument('-b', '--batch', action='store_true', help='read file paths from stdin and classify in parallel') parser.add_argument('-d', '--dist', action='store_true', help='show full distribution over languages') - parser.add_argument('-u', '--url', help='langid of URL') + parser.add_argument('-u', '--url', help='classify text from URL') parser.add_argument('--line', action='store_true', help='process pipes line-by-line rather than as a document') parser.add_argument('-n', '--normalize', action='store_true', help='normalize confidence scores to probability values') options = parser.parse_args() @@ -303,6 +279,8 @@ def main(): import socket from wsgiref.simple_server import make_server + from .server import application + if options.remote and options.host is None: with socket.socket(socket.AF_INET, socket.SOCK_DGRAM) as s: s.connect(("google.com", 80)) @@ -322,30 +300,32 @@ def main(): elif options.batch: import csv + import multiprocessing as mp from functools import partial - from multiprocessing import Pool - def generate_paths(): + def paths(): for line in sys.stdin: p = line.strip() if p and Path(p).is_file(): yield p writer = csv.writer(sys.stdout, lineterminator='\n') - with Pool(initializer=_init_worker, - initargs=(options.model, options.normalize, langs)) as pool: + ctx = mp.get_context('fork') if sys.platform == 'darwin' else mp + with ctx.Pool(processes=mp.cpu_count(), + initializer=_init_worker, + initargs=(options.model, options.normalize, langs)) as pool: if options.dist: - writer.writerow(['path'] + IDENTIFIER.nb_classes) - for path, ranking in pool.imap_unordered(partial(_process_file, dist=True), generate_paths()): - ranking = dict(ranking) - row = [path] + [ranking[c] for c in IDENTIFIER.nb_classes] + header = IDENTIFIER.labels + writer.writerow(['path', 'language'] + header) + for path, ranking in pool.imap_unordered(partial(_process_file, dist=True), paths()): + scores = dict(ranking) + row = [path, ranking[0][0]] + [scores[c] for c in header] writer.writerow(row) else: - for path, (lang, conf) in pool.imap_unordered(_process_file, generate_paths()): + for path, (lang, conf) in pool.imap_unordered(_process_file, paths()): writer.writerow((path, lang, conf)) else: if sys.stdin.isatty(): - # Interactive mode while True: try: print(">>>", end=' ') @@ -354,7 +334,6 @@ def generate_paths(): break print(_process(text)) else: - # Redirected if options.line: for line in sys.stdin: print(_process(line)) diff --git a/py3langid/modelio.py b/py3langid/modelio.py new file mode 100644 index 00000000..caf680fa --- /dev/null +++ b/py3langid/modelio.py @@ -0,0 +1,77 @@ +"""Model serialization: npz inside LZMA, no pickle. + +DFA rows are deduplicated: `nextmove` holds distinct 256-byte rows, +`nextmove_row` maps state → row. Both keys required; legacy models rejected. +""" + +import io +import lzma +import shutil +import tempfile +from array import array + +import numpy as np + + +def _canonical_rows(rows, row_index): + """Sort transition rows for reproducibility; narrow to uint16 if possible.""" + uniq, index = np.unique(np.asarray(rows).reshape(-1, 256), axis=0, + return_inverse=True) + dtype = np.uint16 if len(uniq) < 1 << 16 else np.uint32 + return uniq.ravel(), index.ravel()[np.asarray(row_index)].astype(dtype) + + +def expand_nextmove(rows, row_index): + """Undo row sharing: one 256-entry row per state.""" + return _to_array(np.asarray(rows).reshape(-1, 256)[np.asarray(row_index)]) + + +def save_model(path, model): + """Write (nb_ptc, nb_pc, nb_classes, tk_nextmove, tk_row, tk_output).""" + nb_ptc, nb_pc, nb_classes, tk_nextmove, tk_row, tk_output = model + nextmove = np.asarray(tk_nextmove) + dtype = np.uint16 if not nextmove.size or nextmove.max() < 1 << 16 else np.uint32 + rows, row_index = _canonical_rows(nextmove, tk_row) + out_feat = np.asarray(tk_output, dtype=np.int32) + if len(out_feat) != len(row_index): + raise ValueError("one output slot per DFA state") + arrays = { + "ptc": np.asarray(nb_ptc, dtype=np.float16).reshape(-1, len(nb_pc)), + "pc": np.asarray(nb_pc, dtype=np.float32), + "classes": np.array(nb_classes), + "nextmove": rows.astype(dtype), + "nextmove_row": row_index, + "out_feat": out_feat, + } + buffer = io.BytesIO() + np.savez(buffer, **arrays) + with open(path, "wb") as f: + f.write(lzma.compress(buffer.getvalue(), preset=6)) + + +_TYPECODE = {2: "H", 4: "I", 8: "L"} + + +def _to_array(arr): + """NumPy unsigned int array → stdlib array('H'/'I'/'L').""" + out = array(_TYPECODE[arr.dtype.itemsize]) + out.frombytes(memoryview(np.ascontiguousarray(arr)).cast("B")) + return out + + +def load_model(path): + """Load npz+LZMA model → (nb_ptc, nb_pc, nb_classes, tk_nextmove, tk_row, tk_output).""" + # stream LZMA to a temp file so the uncompressed npz is never fully resident + with tempfile.TemporaryFile(suffix=".npz") as tmp: + with lzma.open(path) as src: + shutil.copyfileobj(src, tmp, length=1 << 20) + tmp.seek(0) + with np.load(tmp, allow_pickle=False) as data: + missing = {"nextmove_row", "out_feat"}.difference(data.files) + if missing: + raise ValueError( + f"{path}: unsupported model layout, missing " + f"{sorted(missing)}; retrain with py3langid.train.train") + return (data["ptc"], data["pc"], data["classes"].tolist(), + _to_array(data["nextmove"]), _to_array(data["nextmove_row"]), + data["out_feat"].tolist()) diff --git a/py3langid/server.py b/py3langid/server.py new file mode 100644 index 00000000..9700c14c --- /dev/null +++ b/py3langid/server.py @@ -0,0 +1,66 @@ +"""WSGI-compatible langid web service (`langid -s` serves it).""" + +import json +from http import HTTPStatus +from urllib.parse import parse_qs + +from .langid import classify, rank + + +def _detect(data): + lang, conf = classify(data) + return {'language': lang, 'confidence': conf} + + +_ROUTES = {'detect': _detect, 'rank': rank} + + +def application(environ, start_response): + """WSGI-compatible langid web service.""" + path = environ.get('PATH_INFO', '').strip('/').partition('/')[0] + handler = _ROUTES.get(path) + if handler is None: + return _return_response(start_response, 404, None, 'Not found') + + method = environ['REQUEST_METHOD'] + if method not in ('GET', 'POST', 'PUT'): + return _return_response(start_response, 405, None, f'{method} not allowed') + + data = _get_data(environ, method) + if data is None: + return _return_response(start_response, 400, None, 'No data provided') + + return _return_response(start_response, 200, handler(data), None) + + +def _get_data(environ, method): + if method == 'GET': + try: + return parse_qs(environ.get('QUERY_STRING', ''))['q'][0] + except KeyError: + return None + try: + length = int(environ.get('CONTENT_LENGTH', 0)) + except ValueError: + return None + if length <= 0: + return None + data = environ['wsgi.input'].read(length) + if method == 'POST': + try: + data = parse_qs(data)[b'q'][0] + except KeyError: + pass + return data + + +def _return_response(start_response, status_code, response_data, response_details): + status = HTTPStatus(status_code) + response = { + 'responseData': response_data, + 'responseStatus': status_code, + 'responseDetails': response_details, + } + headers = [('Content-type', 'application/json; charset=utf-8')] + start_response(f"{status.value} {status.phrase}", headers) + return [json.dumps(response).encode('utf-8')] diff --git a/py3langid/tools/__init__.py b/py3langid/tools/__init__.py deleted file mode 100644 index e69de29b..00000000 diff --git a/py3langid/tools/featWeights.py b/py3langid/tools/featWeights.py deleted file mode 100644 index 8c69a01e..00000000 --- a/py3langid/tools/featWeights.py +++ /dev/null @@ -1,123 +0,0 @@ -""" -Tabulate feature weight data into a single CSV for -further analysis using other tools. This produces -a CSV with header. The features themselves are not -included. - -Marco Lui, February 2013 -""" - -import argparse, os, csv, sys -import numpy as np -import bz2, base64 -from cPickle import loads - -from langid.train.common import read_weights, read_features - -if __name__ == "__main__": - parser = argparse.ArgumentParser() - parser.add_argument('model', metavar="MODEL_DIR", help="path to langid.py training model dir") - parser.add_argument('output', metavar="OUTPUT", help = "write to OUTPUT") - parser.add_argument('-f','--features', metavar="FILE", help = 'only output features from FILE') - parser.add_argument('--raw', action='store_true', help="include raw features") - parser.add_argument('--bin', action='store_true', help="include ig for lang-bin") - args = parser.parse_args() - - def model_file(name): - return os.path.join(args.model, name) - - # Try to determine the set of features to consider - if args.features: - # Use a pre-determined feature list - print >>sys.stderr, "using user-supplied feature list:", args.features - feats = read_features(args.features) - elif os.path.exists(model_file('LDfeats')): - # Use LDfeats - print >>sys.stderr, "using LDfeats" - feats = read_features(model_file('LDfeats')) - else: - raise ValueError("no suitable feature list") - - print >>sys.stderr, "considering {0} features".format(len(feats)) - - records = dict( (k, {}) for k in feats ) - headers = [] - - headers.append('len') - for k in feats: - records[k]['len'] = len(k) - - - # Document Frequency - if os.path.exists(model_file('DF_all')): - print >>sys.stderr, "found weights for document frequency" - w = read_weights(model_file('DF_all')) - headers.append('DF') - for k in feats: - records[k]['DF'] = w[k][0] - - # IG weights for the all-languages event - if os.path.exists(model_file('IGweights.lang')): - print >>sys.stderr, "found weights for lang" - w = read_weights(model_file('IGweights.lang')) - headers.append('IGlang') - for k in feats: - records[k]['IGlang'] = w[k][0] - - # IG weights for the all-domains event - if os.path.exists(model_file('IGweights.domain')): - print >>sys.stderr, "found weights for domain" - w = read_weights(model_file('IGweights.domain')) - headers.append('IGdomain') - for k in feats: - records[k]['IGdomain'] = w[k][0] - - # IG weights for language-binarized - if args.bin and os.path.exists(model_file('IGweights.lang.bin')) and os.path.exists(model_file('lang_index')): - print >>sys.stderr, "found weights for lang.bin" - w = read_weights(model_file('IGweights.lang.bin')) - - # find the list of langs in-order - with open(os.path.join(args.model, "lang_index")) as f: - reader = csv.reader(f) - langs = zip(*reader)[0] - - r_h = ['IGlang.bin.{0}'.format(l) for l in langs] - headers.extend( r_h ) - for k in feats: - records[k].update( dict(zip(r_h, w[k])) ) - - if os.path.exists(model_file('LDfeats.scanner')) and os.path.exists(model_file('model')): - print >>sys.stderr, "found weights for P(t|c)" - with open(model_file('model')) as f: - model = loads(bz2.decompress(base64.b64decode(f.read()))) - with open(model_file('LDfeats.scanner')) as f: - _, _, nb_feats = loads(f.read()) - nb_ptc, nb_pc, nb_classes, tk_nextmove, tk_output = model - nb_numfeats = len(nb_ptc) / len(nb_pc) - nb_ptc = np.array(nb_ptc).reshape(len(nb_ptc)/len(nb_pc), len(nb_pc)) - - # Normalize to 1 on the term axis - for i in range(nb_ptc.shape[1]): - nb_ptc[:,i] = (1/np.exp(nb_ptc[:,i][None,:] - nb_ptc[:,i][:,None]).sum(1)) - w = dict(zip(nb_feats, nb_ptc)) - - r_h = ['ptc.{0}'.format(l) for l in nb_classes] - headers.extend( r_h ) - for k in feats: - records[k].update( dict(zip(r_h, w[k])) ) - - if args.raw: - headers.append('feat') - for k in feats: - records[k]['feat'] = k - - - - print >>sys.stderr, "writing output" - with open(args.output, 'w') as f: - writer = csv.DictWriter(f,headers) - writer.writeheader() - writer.writerows(records.values()) - - print >>sys.stderr, "done" diff --git a/py3langid/tools/printfeats.py b/py3langid/tools/printfeats.py deleted file mode 100644 index 48dd8aa5..00000000 --- a/py3langid/tools/printfeats.py +++ /dev/null @@ -1,41 +0,0 @@ -""" -Print features out in order of their weights - -Marco Lui, November 2013 -""" - -import argparse, os, csv, sys - -from langid.train.common import read_weights - -if __name__ == "__main__": - parser = argparse.ArgumentParser() - parser.add_argument('file', help="file to read") - parser.add_argument('-c','--column',help="project a specific column", type=int) - parser.add_argument('-n','--number',help="output top N features", type=int) - parser.add_argument('-v','--value',help="output the value used for ranking", action="store_true") - parser.add_argument('-p','--printfeat',help="print the actual feature (default is to print repr)", action="store_true") - parser.add_argument('--output', "-o", default=sys.stdout, type=argparse.FileType('w'), help = "write to OUTPUT") - args = parser.parse_args() - - w = read_weights(args.file) - n = args.number if args.number is not None else len(w) - - def show(feat): - if args.printfeat: - return feat - else: - return repr(feat) - - if args.column is not None: - for key in sorted(w, key=lambda x:w[x][args.column], reverse=True)[:n]: - if args.value: - args.output.write("{0},{1}\n".format(show(key),w[key][args.column])) - else: - args.output.write("{0}\n".format(show(key))) - else: - for key in sorted(w, key=w.get, reverse=True)[:n]: - if args.value: - args.output.write("{0},{1}\n".format(show(key),w[key])) - else: - args.output.write("{0}\n".format(show(key))) diff --git a/py3langid/train/BLweight.py b/py3langid/train/BLweight.py deleted file mode 100644 index 6e89a566..00000000 --- a/py3langid/train/BLweight.py +++ /dev/null @@ -1,136 +0,0 @@ -""" -Implementing the "blacklist" feature weighting metric proposed by -Tiedemann & Ljubesic. - -Marco Lui, February 2013 -""" - -NUM_BUCKETS = 64 # number of buckets to use in k-v pair generation -CHUNKSIZE = 50 # maximum size of chunk (number of files tokenized - less = less memory use) - -import argparse -import os - -import numpy as np - -from .common import read_features, makedir, write_weights -from .scanner import build_scanner -from .index import CorpusIndexer -from .NBtrain import generate_cm, learn_ptc - - -if __name__ == "__main__": - - parser = argparse.ArgumentParser() - parser.add_argument("-o","--output", metavar="DIR", help = "write weights to DIR") - parser.add_argument('-f','--features', metavar="FILE", help = 'only output features from FILE') - parser.add_argument("-t", "--temp", metavar='TEMP_DIR', help="store buckets in TEMP_DIR instead of in MODEL_DIR/buckets") - parser.add_argument("-j","--jobs", type=int, metavar='N', help="spawn N processes (set to 1 for no paralleization)") - parser.add_argument("-m","--model", help="save output to MODEL_DIR", metavar="MODEL_DIR") - parser.add_argument("--buckets", type=int, metavar='N', help="distribute features into N buckets", default=NUM_BUCKETS) - parser.add_argument("--chunksize", type=int, help="max chunk size (number of files to tokenize at a time - smaller should reduce memory use)", default=CHUNKSIZE) - parser.add_argument("--no_norm", default=False, action="store_true", help="do not normalize difference in p(t|C) by sum p(t|C)") - parser.add_argument("corpus", help="read corpus from CORPUS_DIR", metavar="CORPUS_DIR") - parser.add_argument("pairs", metavar='LANG_PAIR', nargs="*", help="language pairs to compute BL weights for") - args = parser.parse_args() - - # Work out where our model directory is - corpus_name = os.path.basename(args.corpus) - if args.model: - model_dir = args.model - else: - model_dir = os.path.join('.', corpus_name+'.model') - - def m_path(name): - return os.path.join(model_dir, name) - - # Try to determine the set of features to consider - if args.features: - # Use a pre-determined feature list - feat_path = args.features - elif os.path.exists(m_path('DFfeats')): - # Use LDfeats - feat_path = m_path('DFfeats') - else: - raise ValueError("no suitable feature list") - - # Where temp files go - if args.temp: - buckets_dir = args.temp - else: - buckets_dir = m_path('buckets') - makedir(buckets_dir) - - all_langs = set() - pairs = [] - for p in args.pairs: - try: - lang1, lang2 = p.split(',') - except ValueError: - # Did not unpack to two values - parser.error("{0} is not a lang-pair".format(p)) - all_langs.add(lang1) - all_langs.add(lang2) - pairs.append((lang1, lang2)) - - if args.output: - makedir(args.output) - out_dir = args.output - else: - out_dir = model_dir - - langs = sorted(all_langs) - - # display paths - print("languages({1}): {0}".format(langs, len(langs))) - print("model path:", model_dir) - print("feature path:", feat_path) - print("output path:", out_dir) - print("temp (buckets) path:", buckets_dir) - - feats = read_features(feat_path) - - indexer = CorpusIndexer(args.corpus, langs = langs) - items = [ (d,l,p) for (d,l,n,p) in indexer.items ] - if len(items) == 0: - raise ValueError("found no files!") - - print("will process {0} features across {1} paths".format(len(feats), len(items))) - print("will process {0} features across {1} paths".format(len(feats), len(items))) - - # produce a scanner over all the features - tk_nextmove, tk_output = build_scanner(feats) - - # Generate a class map over all the languages we are dealing with - cm = generate_cm([ (l,p) for d,l,p in items], len(langs)) - - # Compute P(t|C) - print("learning P(t|C)") - paths = zip(*items)[2] - nb_ptc = learn_ptc(paths, tk_nextmove, tk_output, cm, buckets_dir, args) - nb_ptc = np.array(nb_ptc).reshape(len(feats), len(langs)) - - # Normalize to 1 on the term axis - print("renormalizing P(t|C)") - for i in range(nb_ptc.shape[1]): - # had to de-vectorize this due to memory consumption - newval = np.empty_like(nb_ptc[:,i]) - for j in range(newval.shape[0]): - newval[j] = (1/np.exp(nb_ptc[:,i] - nb_ptc[j,i]).sum()) - nb_ptc[:,i] = newval - assert (1.0 - newval.sum()) < 0.0001 - - print("doing per-pair output") - for lang1, lang2 in pairs: - # Where to do output - if args.no_norm: - weights_path = os.path.join(out_dir, ('BLfeats.no_norm.{0}.{1}'.format(lang1, lang2))) - else: - weights_path = os.path.join(out_dir, ('BLfeats.{0}.{1}'.format(lang1, lang2))) - - i1 = indexer.lang_index[lang1] - i2 = indexer.lang_index[lang2] - - w = dict(zip(feats, np.abs((nb_ptc[:,i1] - nb_ptc[:,i2]) / (nb_ptc.sum(1) if not args.no_norm else 1)))) - write_weights(w, weights_path) - print("wrote weights to {0}".format(weights_path)) diff --git a/py3langid/train/DFfeatureselect.py b/py3langid/train/DFfeatureselect.py deleted file mode 100644 index 2ad7554b..00000000 --- a/py3langid/train/DFfeatureselect.py +++ /dev/null @@ -1,164 +0,0 @@ -""" -DFfeatureselect.py - -First step in the LD feature selection process, select features based on document -frequency. - -Marco Lui January 2013 - -Copyright 2013 Marco Lui . All rights reserved. - -Redistribution and use in source and binary forms, with or without modification, are -permitted provided that the following conditions are met: - - 1. Redistributions of source code must retain the above copyright notice, this list of - conditions and the following disclaimer. - - 2. Redistributions in binary form must reproduce the above copyright notice, this list - of conditions and the following disclaimer in the documentation and/or other materials - provided with the distribution. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDER ``AS IS'' AND ANY EXPRESS OR IMPLIED -WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND -FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR -CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR -CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR -SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON -ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING -NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF -ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - -The views and conclusions contained in the software and documentation are those of the -authors and should not be interpreted as representing official policies, either expressed -or implied, of the copyright holder. -""" - -###### -# Default values -# Can be overriden with command-line options -###### -MAX_NGRAM_ORDER = 4 # largest order of n-grams to consider -TOKENS_PER_ORDER = 15000 # number of tokens to consider for each order - -import argparse -import os -import marshal - -from collections import defaultdict - -from .common import unmarshal_iter, MapPool, write_features, write_weights - - -def pass_sum_df(bucket): - """ - Compute document frequency (df) by summing up (key,domain,count) triplets - over all domains. - """ - doc_count = defaultdict(int) - count = 0 - with open(os.path.join(bucket, "docfreq"),'wb') as docfreq: - for path in os.listdir(bucket): - # We use the domain buckets as there are usually less domains - if path.endswith('.domain'): - for key, _, value in unmarshal_iter(os.path.join(bucket,path)): - doc_count[key] += value - count += 1 - - for item in doc_count.iteritems(): - docfreq.write(marshal.dumps(item)) - return count - -def tally(bucketlist, jobs=None): - """ - Sum up the counts for each feature across all buckets. This - builds a full mapping of feature->count. This is stored in-memory - and thus could be an issue for large feature sets. - """ - - with MapPool(jobs) as f: - pass_sum_df_out = f(pass_sum_df, bucketlist) - - for i, keycount in enumerate(pass_sum_df_out): - print("processed bucket (%d/%d) [%d keys]" % (i+1, len(bucketlist), keycount)) - - # build the global term->df mapping - doc_count = {} - for bucket in bucketlist: - for key, value in unmarshal_iter(os.path.join(bucket, 'docfreq')): - doc_count[key] = value - - return doc_count - - - -def ngram_select(doc_count, max_order=MAX_NGRAM_ORDER, tokens_per_order=TOKENS_PER_ORDER): - """ - DF feature selection for byte-ngram tokenization - """ - # Work out the set of features to compute IG - features = set() - for i in range(1, max_order+1): - d = dict( (k, doc_count[k]) for k in doc_count if len(k) == i) - features |= set(sorted(d, key=d.get, reverse=True)[:tokens_per_order]) - features = sorted(features) - - return features - - - -if __name__ == "__main__": - parser = argparse.ArgumentParser() - parser.add_argument("-j","--jobs", type=int, metavar='N', help="spawn N processes (set to 1 for no paralleization)") - parser.add_argument("-f","--features", metavar='FEATURE_FILE', help="output features to FEATURE_FILE") - parser.add_argument("--tokens_per_order", metavar='N', type=int, help="consider top N tokens per ngram order") - parser.add_argument("--tokens", metavar='N', type=int, help="consider top N tokens") - parser.add_argument("--max_order", type=int, help="highest n-gram order to use", default=MAX_NGRAM_ORDER) - parser.add_argument("--doc_count", nargs='?', const=True, metavar='DOC_COUNT_PATH', help="output full mapping of feature->frequency to DOC_COUNT_PATH") - parser.add_argument("model", metavar='MODEL_DIR', help="read index and produce output in MODEL_DIR") - - args = parser.parse_args() - - if args.tokens and args.tokens_per_order: - parser.error("--tokens and --tokens_per_order are mutually exclusive") - - # if neither --tokens nor --tokens_per_order is given, default behaviour is tokens_per_order - if not(args.tokens) and not(args.tokens_per_order): - args.tokens_per_order = TOKENS_PER_ORDER - - if args.features: - feature_path = args.features - else: - feature_path = os.path.join(args.model, 'DFfeats') - - bucketlist_path = os.path.join(args.model, 'bucketlist') - - # display paths - print("buckets path:", bucketlist_path) - print("features output path:", feature_path) - if args.tokens_per_order: - print("max ngram order:", args.max_order) - print("tokens per order:", args.tokens_per_order) - else: - print("tokens:", args.tokens) - - with open(bucketlist_path) as f: - bucketlist = map(str.strip, f) - - doc_count = tally(bucketlist, args.jobs) - print("unique features:", len(doc_count)) - if args.doc_count: - # The constant true is used to indicate output to default location - doc_count_path = os.path.join(args.model, 'DF_all') if args.doc_count == True else args.doc_count - write_weights(doc_count, doc_count_path) - print("wrote DF counts for all features to:", doc_count_path) - - if args.tokens_per_order: - # Choose a number of features for each length of token - feats = ngram_select(doc_count, args.max_order, args.tokens_per_order) - else: - # Choose a number of features overall - feats = sorted( sorted(doc_count, key=doc_count.get, reverse=True)[:args.tokens] ) - - print("selected features: ", len(feats)) - - write_features(feats, feature_path) - print('wrote features to "%s"' % feature_path) diff --git a/py3langid/train/IGweight.py b/py3langid/train/IGweight.py deleted file mode 100644 index 6a4b4647..00000000 --- a/py3langid/train/IGweight.py +++ /dev/null @@ -1,239 +0,0 @@ -""" -IGWeight.py - -Compute IG Weights given a set of tokenized buckets and a feature set - -Marco Lui, January 2013 - -Based on research by Marco Lui and Tim Baldwin. - -Copyright 2013 Marco Lui . All rights reserved. - -Redistribution and use in source and binary forms, with or without modification, are -permitted provided that the following conditions are met: - - 1. Redistributions of source code must retain the above copyright notice, this list of - conditions and the following disclaimer. - - 2. Redistributions in binary form must reproduce the above copyright notice, this list - of conditions and the following disclaimer in the documentation and/or other materials - provided with the distribution. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDER ``AS IS'' AND ANY EXPRESS OR IMPLIED -WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND -FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR -CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR -CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR -SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON -ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING -NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF -ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - -The views and conclusions contained in the software and documentation are those of the -authors and should not be interpreted as representing official policies, either expressed -or implied, of the copyright holder. -""" - -import argparse -import csv -import os - -from collections import defaultdict - -import numpy - -from .common import unmarshal_iter, MapPool, Enumerator, write_weights, read_features - - -def entropy(v, axis=0): - """ - Optimized implementation of entropy. This version is faster than that in - scipy.stats.distributions, particularly over long vectors. - """ - v = numpy.array(v, dtype='float') - s = numpy.sum(v, axis=axis) - with numpy.errstate(divide='ignore', invalid='ignore'): - rhs = numpy.nansum(v * numpy.log(v), axis=axis) / s - r = numpy.log(s) - rhs - # Where dealing with binarized events, it is possible that an event always - # occurs and thus has 0 information. In this case, the negative class - # will have frequency 0, resulting in log(0) being computed as nan. - # We replace these nans with 0 - nan_index = numpy.isnan(rhs) - if nan_index.any(): - r[nan_index] = 0 - return r - -def setup_pass_IG(features, dist, binarize, suffix): - """ - @param features the list of features to compute IG for - @param dist the background distribution - @param binarize (boolean) compute IG binarized per-class if True - @param suffix of files in bucketdir to process - """ - global __features, __dist, __binarize, __suffix - __features = features - __dist = dist - __binarize = binarize - __suffix = suffix - -def pass_IG(bucket): - """ - In this pass we compute the information gain for each feature, binarized - with respect to each language as well as unified over the set of all - classes. - - @global __features the list of features to compute IG for - @global __dist the background distribution - @global __binarize (boolean) compute IG binarized per-class if True - @global __suffix of files in bucketdir to process - @param bucket the bucket file to process. It is assumed to contain marshalled (term, event_id, count) triplets. - """ - global __features, __dist, __binarize, __suffix - - # We first tally the per-event frequency of each - # term in our selected feature set. - term_freq = defaultdict(lambda: defaultdict(int)) - term_index = defaultdict(Enumerator()) - - for path in os.listdir(bucket): - if path.endswith(__suffix): - for key, event_id, count in unmarshal_iter(os.path.join(bucket,path)): - # Select only our listed features - if key in __features: - term_index[key] - term_freq[key][event_id] += count - - num_term = len(term_index) - num_event = len(__dist) - - cm_pos = numpy.zeros((num_term, num_event), dtype='int') - - for term,term_id in term_index.iteritems(): - # update event matrix - freq = term_freq[term] - for event_id, count in freq.iteritems(): - cm_pos[term_id, event_id] = count - cm_neg = __dist - cm_pos - cm = numpy.dstack((cm_neg, cm_pos)) - - if not __binarize: - # non-binarized event space - x = cm.sum(axis=1) - term_w = x / x.sum(axis=1)[:, None].astype(float) - - # Entropy of the term-present/term-absent events - e = entropy(cm, axis=1) - - # Information Gain with respect to the set of events - ig = entropy(__dist) - (term_w * e).sum(axis=1) - - else: - # binarized event space - # Compute IG binarized with respect to each event - ig = list() - for event_id in range(num_event): - num_doc = __dist.sum() - prior = numpy.array((num_doc - __dist[event_id], __dist[event_id]), dtype=float) / num_doc - - cm_bin = numpy.zeros((num_term, 2, 2), dtype=int) # (term, p(term), p(lang|term)) - cm_bin[:,0,:] = cm.sum(axis=1) - cm[:,event_id,:] - cm_bin[:,1,:] = cm[:,event_id,:] - - e = entropy(cm_bin, axis=1) - x = cm_bin.sum(axis=1) - term_w = x / x.sum(axis=1)[:, None].astype(float) - - ig.append( entropy(prior) - (term_w * e).sum(axis=1) ) - ig = numpy.vstack(ig) - - terms = sorted(term_index, key=term_index.get) - return terms, ig - - -def compute_IG(bucketlist, features, dist, binarize, suffix, job_count=None): - pass_IG_args = (features, dist, binarize, suffix) - - num_chunk = len(bucketlist) - weights = [] - terms = [] - - with MapPool(job_count, setup_pass_IG, pass_IG_args) as f: - pass_IG_out = f(pass_IG, bucketlist) - - for i, (t, w) in enumerate(pass_IG_out): - weights.append(w) - terms.extend(t) - print("processed chunk (%d/%d) [%d terms]" % (i+1, num_chunk, len(t))) - - if binarize: - weights = numpy.hstack(weights).transpose() - else: - weights = numpy.concatenate(weights) - terms = ["".join(t) for t in terms] - - return zip(terms, weights) - -def read_dist(path): - """ - Read the distribution from a file containing item, count pairs. - @param path path to read form - """ - with open(path) as f: - reader = csv.reader(f) - return numpy.array(zip(*reader)[1], dtype=int) - -if __name__ == "__main__": - parser = argparse.ArgumentParser() - parser.add_argument("-j","--jobs", type=int, metavar='N', help="spawn N processes (set to 1 for no paralleization)") - parser.add_argument("-f","--features", metavar='FEATURE_FILE', help="read features from FEATURE_FILE") - parser.add_argument("-w","--weights", metavar='WEIGHTS', help="output weights to WEIGHTS") - parser.add_argument("model", metavar='MODEL_DIR', help="read index and produce output in MODEL_DIR") - parser.add_argument("-d","--domain", action="store_true", default=False, help="compute IG with respect to domain") - parser.add_argument("-b","--binarize", action="store_true", default=False, help="binarize the event space in the IG computation") - parser.add_argument("-l","--lang", action="store_true", default=False, help="compute IG with respect to language") - - args = parser.parse_args() - if not(args.domain or args.lang) or (args.domain and args.lang): - parser.error("exactly one of domain(-d) or language (-l) must be specified") - - if args.features: - feature_path = args.features - else: - feature_path = os.path.join(args.model, 'DFfeats') - - bucketlist_path = os.path.join(args.model, 'bucketlist') - - if not os.path.exists(feature_path): - parser.error('{0} does not exist'.format(feature_path)) - - bucketlist = map(str.strip, open(bucketlist_path)) - features = read_features(feature_path) - - if args.domain: - index_path = os.path.join(args.model,'domain_index') - suffix = '.domain' - elif args.lang: - index_path = os.path.join(args.model,'lang_index') - suffix = '.lang' - else: - raise ValueError("no event specified") - - if args.weights: - weights_path = args.weights - else: - weights_path = os.path.join(args.model, 'IGweights' + suffix + ('.bin' if args.binarize else '')) - - # display paths - print("model path:", args.model ) - print("buckets path:", bucketlist_path) - print("features path:", feature_path) - print("weights path:", weights_path) - print("index path:", index_path) - print("suffix:", suffix) - - print("computing information gain") - - dist = read_dist(index_path) - ig = compute_IG(bucketlist, features, dist, args.binarize, suffix, args.jobs) - - write_weights(ig, weights_path) diff --git a/py3langid/train/LDfeatureselect.py b/py3langid/train/LDfeatureselect.py deleted file mode 100644 index 543c5757..00000000 --- a/py3langid/train/LDfeatureselect.py +++ /dev/null @@ -1,114 +0,0 @@ -""" -LDfeatureselect.py - -LD (Lang-Domain) feature extractor -Marco Lui November 2011 - -Based on research by Marco Lui and Tim Baldwin. - -Copyright 2011 Marco Lui . All rights reserved. - -Redistribution and use in source and binary forms, with or without modification, are -permitted provided that the following conditions are met: - - 1. Redistributions of source code must retain the above copyright notice, this list of - conditions and the following disclaimer. - - 2. Redistributions in binary form must reproduce the above copyright notice, this list - of conditions and the following disclaimer in the documentation and/or other materials - provided with the distribution. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDER ``AS IS'' AND ANY EXPRESS OR IMPLIED -WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND -FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR -CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR -CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR -SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON -ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING -NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF -ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - -The views and conclusions contained in the software and documentation are those of the -authors and should not be interpreted as representing official policies, either expressed -or implied, of the copyright holder. -""" - -###### -# Default values -# Can be overriden with command-line options -###### -FEATURES_PER_LANG = 300 # number of features to select for each language - -import argparse -import csv -import os - -from collections import defaultdict - -import numpy - -from common import read_weights, Enumerator, write_features - -def select_LD_features(ig_lang, ig_domain, feats_per_lang, ignore_domain=False): - """ - @param ignore_domain boolean to indicate whether to use domain weights - """ - assert (ig_domain is None) or (len(ig_lang) == len(ig_domain)) - num_lang = len(ig_lang.values()[0]) - num_term = len(ig_lang) - - term_index = defaultdict(Enumerator()) - - - ld = numpy.empty((num_lang, num_term), dtype=float) - - for term in ig_lang: - term_id = term_index[term] - if ignore_domain: - ld[:, term_id] = ig_lang[term] - else: - ld[:, term_id] = ig_lang[term] - ig_domain[term] - - terms = sorted(term_index, key=term_index.get) - # compile the final feature set - selected_features = {} - for lang_id, lang_w in enumerate(ld): - term_inds = numpy.argsort(lang_w)[-feats_per_lang:] - selected_features[lang_id] = [terms[t] for t in term_inds] - - return selected_features - -if __name__ == "__main__": - parser = argparse.ArgumentParser() - parser.add_argument("-o","--output", metavar="OUTPUT_PATH", help = "write selected features to OUTPUT_PATH") - parser.add_argument("--feats_per_lang", type=int, metavar='N', help="select top N features for each language", default=FEATURES_PER_LANG) - parser.add_argument("--per_lang", action="store_true", default=False, help="produce a list of features selecter per-language") - parser.add_argument("--no_domain_ig", action="store_true", default=False, help="use only per-langugage IG in LD calculation") - parser.add_argument("model", metavar='MODEL_DIR', help="read index and produce output in MODEL_DIR") - args = parser.parse_args() - - lang_w_path = os.path.join(args.model, 'IGweights.lang.bin') - domain_w_path = os.path.join(args.model, 'IGweights.domain') - feature_path = args.output if args.output else os.path.join(args.model, 'LDfeats') - - # display paths - print("model path:", args.model) - print("lang weights path:", lang_w_path) - print("domain weights path:", domain_w_path) - print("feature output path:", feature_path) - - lang_w = read_weights(lang_w_path) - domain_w = read_weights(domain_w_path) if not args.no_domain_ig else None - - features_per_lang = select_LD_features(lang_w, domain_w, args.feats_per_lang, ignore_domain=args.no_domain_ig) - if args.per_lang: - with open(feature_path + '.perlang', 'w') as f: - writer = csv.writer(f) - for i in range(len(features_per_lang)): - writer.writerow(map(repr,features_per_lang[i])) - - - final_feature_set = reduce(set.union, map(set, features_per_lang.values())) - print('selected %d features' % len(final_feature_set)) - - write_features(sorted(final_feature_set), feature_path) - print('wrote features to "%s"' % feature_path) diff --git a/py3langid/train/NBtrain.py b/py3langid/train/NBtrain.py deleted file mode 100644 index 629acdb0..00000000 --- a/py3langid/train/NBtrain.py +++ /dev/null @@ -1,295 +0,0 @@ -""" -NBtrain.py - -Model generator for langid.py - -Marco Lui, January 2013 - -Based on research by Marco Lui and Tim Baldwin. - -Copyright 2013 Marco Lui . All rights reserved. - -Redistribution and use in source and binary forms, with or without modification, are -permitted provided that the following conditions are met: - - 1. Redistributions of source code must retain the above copyright notice, this list of - conditions and the following disclaimer. - - 2. Redistributions in binary form must reproduce the above copyright notice, this list - of conditions and the following disclaimer in the documentation and/or other materials - provided with the distribution. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDER ``AS IS'' AND ANY EXPRESS OR IMPLIED -WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND -FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR -CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR -CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR -SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON -ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING -NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF -ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - -The views and conclusions contained in the software and documentation are those of the -authors and should not be interpreted as representing official policies, either expressed -or implied, of the copyright holder. -""" -MAX_CHUNK_SIZE = 100 # maximum number of files to tokenize at once -NUM_BUCKETS = 64 # number of buckets to use in k-v pair generation - -import array -import argparse -import atexit -import base64 -import bz2 -import csv -import marshal -import multiprocessing as mp -import os -import pickle -import shutil -import tempfile - -from collections import defaultdict - -import numpy as np - -from .common import chunk, unmarshal_iter, MapPool - - -def offsets(chunks): - # Work out the path chunk start offsets - chunk_offsets = [0] - for c in chunks: - chunk_offsets.append(chunk_offsets[-1] + len(c)) - return chunk_offsets - -def state_trace(path): - """ - Returns counts of how often each state was entered - """ - global __nm_arr - c = defaultdict(int) - state = 0 - - with open(path) as f: - text = f.read() - for letter in map(ord,text): - state = __nm_arr[(state << 8) + letter] - c[state] += 1 - return c - -def setup_pass_tokenize(nm_arr, output_states, tk_output, b_dirs): - """ - Set the global next-move array used by the aho-corasick scanner - """ - global __nm_arr, __output_states, __tk_output, __b_dirs - __nm_arr = nm_arr - __output_states = output_states - __tk_output = tk_output - __b_dirs = b_dirs - -def pass_tokenize(arg): - """ - Tokenize documents and do counts for each feature - Split this into buckets chunked over features rather than documents - """ - global __output_states, __tk_output, __b_dirs - chunk_offset, chunk_paths = arg - term_freq = defaultdict(int) - __procname = mp.current_process().name - __buckets = [tempfile.mkstemp(prefix=__procname, suffix='.index', dir=p)[0] for p in __b_dirs] - - # Tokenize each document and add to a count of (doc_id, f_id) frequencies - for doc_count, path in enumerate(chunk_paths): - doc_id = doc_count + chunk_offset - count = state_trace(path) - for state in (set(count) & __output_states): - for f_id in __tk_output[state]: - term_freq[doc_id, f_id] += count[state] - - # Distribute the aggregated counts into buckets - bucket_count = len(__buckets) - for doc_id, f_id in term_freq: - bucket_index = hash(f_id) % bucket_count - count = term_freq[doc_id, f_id] - item = ( f_id, doc_id, count ) - os.write(__buckets[bucket_index], marshal.dumps(item)) - - for f in __buckets: - os.close(f) - - return len(term_freq) - -def setup_pass_ptc(cm, num_instances): - global __cm, __num_instances - __cm = cm - __num_instances = num_instances - -def pass_ptc(b_dir): - """ - Take a bucket, form a feature map, compute the count of - each feature in each class. - @param b_dir path to the bucket directory - @returns (read_count, f_ids, prod) - """ - global __cm, __num_instances - - terms = defaultdict(lambda : np.zeros((__num_instances,), dtype='int')) - - read_count = 0 - for path in os.listdir(b_dir): - if path.endswith('.index'): - for f_id, doc_id, count in unmarshal_iter(os.path.join(b_dir, path)): - terms[f_id][doc_id] = count - read_count += 1 - - f_ids, f_vs = zip(*terms.items()) - fm = np.vstack(f_vs) - prod = np.dot(fm, __cm) - return read_count, f_ids, prod - - -def learn_pc(cm): - """ - @param cm class map - @returns nb_pc: log(P(C)) - """ - pc = np.log(cm.sum(0)) - nb_pc = array.array('d', pc) - return nb_pc - -def generate_cm(items, num_classes): - """ - @param items (class id, path) pairs - @param num_classes The number of classes present - """ - num_instances = len(items) - - # Generate the class map - cm = np.zeros((num_instances, num_classes), dtype='bool') - for docid, (lang_id, path) in enumerate(items): - cm[docid, lang_id] = True - - return cm - -def learn_ptc(paths, tk_nextmove, tk_output, cm, temp_path, args): - global b_dirs - num_instances = len(paths) - num_features = max( i for v in tk_output.values() for i in v) + 1 - - # Generate the feature map - nm_arr = mp.Array('i', tk_nextmove, lock=False) - - if args.jobs: - chunksize = min(len(paths) / (args.jobs*2), args.chunksize) - else: - chunksize = min(len(paths) / (mp.cpu_count()*2), args.chunksize) - - # TODO: Set the output dir - b_dirs = [ tempfile.mkdtemp(prefix="train-",suffix='-bucket', dir=temp_path) for i in range(args.buckets) ] - - output_states = set(tk_output) - - path_chunks = list(chunk(paths, chunksize)) - pass_tokenize_arg = zip(offsets(path_chunks), path_chunks) - - pass_tokenize_params = (nm_arr, output_states, tk_output, b_dirs) - with MapPool(args.jobs, setup_pass_tokenize, pass_tokenize_params) as f: - pass_tokenize_out = f(pass_tokenize, pass_tokenize_arg) - - write_count = sum(pass_tokenize_out) - print("wrote a total of %d keys" % write_count) - - pass_ptc_params = (cm, num_instances) - with MapPool(args.jobs, setup_pass_ptc, pass_ptc_params) as f: - pass_ptc_out = f(pass_ptc, b_dirs) - - reads, ids, prods = zip(*pass_ptc_out) - read_count = sum(reads) - print("read a total of %d keys (%d short)" % (read_count, write_count - read_count)) - - prod = np.zeros((num_features, cm.shape[1]), dtype=int) - prod[np.concatenate(ids)] = np.vstack(prods) - - ptc = np.log(1 + prod) - np.log(num_features + prod.sum(0)) - - nb_ptc = array.array('d') - for term_dist in ptc.tolist(): - nb_ptc.extend(term_dist) - - return nb_ptc - -@atexit.register -def cleanup(): - global b_dirs - try: - for d in b_dirs: - shutil.rmtree(d) - except NameError: - # Failed before b_dirs is defined, nothing to clean - pass - -if __name__ == "__main__": - parser = argparse.ArgumentParser() - parser.add_argument("-j","--jobs", type=int, metavar='N', help="spawn N processes (set to 1 for no paralleization)") - parser.add_argument("-t", "--temp", metavar='TEMP_DIR', help="store buckets in TEMP_DIR instead of in MODEL_DIR/buckets") - parser.add_argument("-s", "--scanner", metavar='SCANNER', help="use SCANNER for feature counting") - parser.add_argument("-o", "--output", metavar='OUTPUT', help="output langid.py-compatible model to OUTPUT") - #parser.add_argument("-i","--index",metavar='INDEX',help="read list of training document paths from INDEX") - parser.add_argument("model", metavar='MODEL_DIR', help="read index and produce output in MODEL_DIR") - parser.add_argument("--chunksize", type=int, help='maximum chunk size (number of files)', default=MAX_CHUNK_SIZE) - parser.add_argument("--buckets", type=int, metavar='N', help="distribute features into N buckets", default=NUM_BUCKETS) - args = parser.parse_args() - - if args.temp: - temp_path = args.temp - else: - temp_path = os.path.join(args.model, 'buckets') - - if args.scanner: - scanner_path = args.scanner - else: - scanner_path = os.path.join(args.model, 'LDfeats.scanner') - - if args.output: - output_path = args.output - else: - output_path = os.path.join(args.model, 'model') - - index_path = os.path.join(args.model, 'paths') - lang_path = os.path.join(args.model, 'lang_index') - - # display paths - print("model path:", args.model) - print("temp path:", temp_path) - print("scanner path:", scanner_path) - #print "index path:", index_path - print("output path:", output_path) - - # read list of training files - with open(index_path) as f: - reader = csv.reader(f) - items = [ (l,p) for _,l,p in reader ] - - # read scanner - with open(scanner_path) as f: - tk_nextmove, tk_output, _ = pickle.load(f) - - # read list of languages in order - with open(lang_path) as f: - reader = csv.reader(f) - langs = zip(*reader)[0] - - cm = generate_cm(items, len(langs)) - paths = zip(*items)[1] - - nb_classes = langs - nb_pc = learn_pc(cm) - nb_ptc = learn_ptc(paths, tk_nextmove, tk_output, cm, temp_path, args) - - # output the model - model = nb_ptc, nb_pc, nb_classes, tk_nextmove, tk_output - string = base64.b64encode(bz2.compress(pickle.dumps(model))) - with open(output_path, 'w') as f: - f.write(string) - - print("wrote model to %s (%d bytes)" % (output_path, len(string))) diff --git a/py3langid/train/README b/py3langid/train/README index 5dde6d29..bd546fae 100644 --- a/py3langid/train/README +++ b/py3langid/train/README @@ -1,16 +1,23 @@ -Refactoring of the langid.py training tools, to allow for -more flexibility and easier experimentation. +Training pipeline for py3langid models. -Planned tools: -1) index.py - index a corpus. Produce a list of file, corpus, language pairs. -2) tokenize.py - take an index and tokenize the corresponding files -3) DFfeatureselect.py - choose features by document frequency -3) IGweight.py - compute the IG weights for language and for domain -4) LDfeatureselect.py - take the IG weights and use them to select a feature set -5) scanner.py - build a scanner on the basis of a feature set -6) NBtrain.py - learn NB parameters using an indexed corpus and a scanner +Everything runs through two commands: -Optional: -A single tool that integrates all steps, calling on each submodule as required. +1) gather_data.py - download a multi-domain corpus (tatoeba, cc100, wiki, + leipzig; language code tables in sources.py); raw downloads are kept + in raw_downloads/ and reused. topup.py adds generic sources (GlotCC, + Glot500, GlotSparse, UDHR) for thin classes via --domains topup. + python -m py3langid.train.gather_data --output corpusN +2) train.py - single integrated trainer: index -> per-(domain,lang) n-gram + count shards (cached) -> DF feature selection -> IG weights (lang+domain) + -> LD feature selection -> scanner -> NB parameters -> model.npz.xz. + python -m py3langid.train.train -m MODEL_DIR corpusN -Marco Lui, January 2013 +Supporting modules (used as libraries by train.py): + stages.py (indexing, feature selection, IG, NB parameters), + shards.py, scanner.py, common.py + +verify.py - corpus cleaning: classify training docs with a trusted model, +move mismatched docs (or strip mismatched paragraphs with --paragraphs). + python -m py3langid.train.verify --model MODEL_FILE corpusN + +Originally a refactoring of the langid.py training tools by Marco Lui (2013). diff --git a/py3langid/train/common.py b/py3langid/train/common.py index 2fbafe9e..4ff54de9 100644 --- a/py3langid/train/common.py +++ b/py3langid/train/common.py @@ -1,143 +1,162 @@ -""" -Common functions +"""Training pipeline constants and helpers.""" -Marco Lui, January 2013 -""" - -import csv -import errno -import marshal import multiprocessing as mp -import os - -from contextlib import contextmanager, closing -from itertools import imap, islice - -import numpy - - -class Enumerator(object): - """ - Enumerator object. Returns a larger number each call. - Can be used with defaultdict to enumerate a sequence of items. - """ - def __init__(self, start=0): - self.n = start - - def __call__(self): - retval = self.n - self.n += 1 - return retval - -def chunk(seq, chunksize): - """ - Break a sequence into chunks not exceeeding a predetermined size - """ - seq_iter = iter(seq) - while True: - chunk = tuple(islice(seq_iter, chunksize)) - if not chunk: break - yield chunk - -def unmarshal_iter(path): - """ - Open a given path and yield an iterator over items unmarshalled from it. - """ - with open(path, 'rb') as f: - while True: - try: - yield marshal.load(f) - except EOFError: - break - -def makedir(path): +import sys +import unicodedata +from collections.abc import Callable +from contextlib import contextmanager +from pathlib import Path +from typing import NamedTuple + +from ..langid import decode_trimmed + +MAX_NGRAM_ORDER = 5 +MIN_NGRAM_ORDER = 2 +DF_TOKENS = 60000 # candidate pool per order +FEATURES_PER_LANG = 1050 # per-language, not global (keeps script-novel langs viable) +DOC_CAP = 3000 # byte budget: gathering, tokenization, verifier, zxx +MIN_DOC = 500 +MIN_DOMAINS = 2 # feature selection and topup both target this + +CLUSTERS = (("ms", "id"), ("bs", "hr"), ("no", "nn", "da"), + ("zh", "yue", "wuu")) +CLUSTER_K = 150 # extra features per cluster + +TOKENIZE_ORDER = 6 # CJK codepoint bigrams (3+3 bytes) +SELECT_ORDERS = frozenset(range(MIN_NGRAM_ORDER, MAX_NGRAM_ORDER + 1)) \ + | {TOKENIZE_ORDER} + + +def is_cjk_bigram(term): + """True if term is exactly two CJK codepoints (6 bytes).""" + if len(term) != TOKENIZE_ORDER: + return False try: - os.makedirs(path) - except OSError as e: - if e.errno != errno.EEXIST: - raise - - -def write_weights(weights, path): - w = dict(weights) - with open(path, 'w') as f: - writer = csv.writer(f) - try: - key_order = sorted(w, key=w.get, reverse=True) - except ValueError: - # Could not order keys by value, value is probably a vector. - # Order keys alphabetically in this case. - key_order = sorted(w) - - for k in key_order: - row = [repr(k)] - try: - row.extend(w[k]) - except TypeError: - row.append(w[k]) - writer.writerow(row) - - -def read_weights(path): - with open(path) as f: - reader = csv.reader(f) - retval = {} - for row in reader: - key = eval(row[0]) - #val = numpy.array( map(float,row[1:]) ) - val = numpy.array( [float(v) if v != 'nan' else 0. for v in row[1:]] ) - retval[key] = val - return retval - -def read_features(path): - """ - Read a list of features in feature-per-line format, where each - feature is a repr and needs to be evaled. - @param path path to read from - """ - with open(path) as f: - return map(eval, f) - -def write_features(features, path): - """ - Write a list of features to a file at `path`. The repr of each - feature is written on a new line. - @param features list of features to write - @param path path to write to - """ - with open(path,'w') as f: - for feat in features: - print(repr(feat),file=f) - - -def index(seq): - """ - Build an index for a sequence of items. Assumes - that the items in the sequence are unique. - @param seq the sequence to index - @returns a dictionary from item to position in the sequence - """ - return {(k,v) for (v,k) in enumerate(seq)} + s = term.decode("utf-8") + except UnicodeDecodeError: + return False + return len(s) == 2 and all(ord(ch) >= 0x2E80 for ch in s) + + +def latin_majority(doc): + """True if doc has more Latin than Cyrillic letters.""" + text = doc.decode("utf-8", errors="surrogateescape") + cyr = sum(1 for ch in text if "Ѐ" <= ch <= "ӿ") + lat = sum(1 for ch in text if ch.isalpha() and ch < "ɐ") + return lat > cyr + + +class SplitScript(NamedTuple): + """A language trained as two script-specific classes.""" + alt: str + script: str + alt_script: str + routes_to_alt: Callable + + +SPLIT_SCRIPT = { + "sr": SplitScript("srl", "Cyrl", "Latn", latin_majority), + "uz": SplitScript("uzc", "Latn", "Cyrl", lambda doc: not latin_majority(doc)), +} + +ALT_CLASS = {s.alt: lang for lang, s in SPLIT_SCRIPT.items()} +LABEL_ALIAS = {**ALT_CLASS, "nb": "no"} +CLASS_SCRIPT = {lang: s.script for lang, s in SPLIT_SCRIPT.items()} +CLASS_SCRIPT.update({s.alt: s.alt_script for s in SPLIT_SCRIPT.values()}) + + +def route_script(out_dir, doc): + """Route doc to its alt class dir if split-script.""" + spec = SPLIT_SCRIPT.get(out_dir.name) + if spec and spec.routes_to_alt(doc): + return out_dir.with_name(spec.alt) + return out_dir + + +def script_filter(cls): + "Doc predicate for `cls`'s script, or None if `cls` is not split-script." + if cls in ALT_CLASS: + return SPLIT_SCRIPT[ALT_CLASS[cls]].routes_to_alt + if cls in SPLIT_SCRIPT: + routes_to_alt = SPLIT_SCRIPT[cls].routes_to_alt + return lambda doc: not routes_to_alt(doc) + return None + + +def walk_corpus(root, skip_langs=(), pattern="*.txt"): + """Yield (domain, lang, path) for docs three levels down, sorted.""" + for domain in sorted(p for p in Path(root).iterdir() if p.is_dir()): + for lang_dir in sorted(p for p in domain.iterdir() if p.is_dir()): + if lang_dir.name in skip_langs: + continue + for doc in sorted(lang_dir.glob(pattern)): + if doc.is_file(): + yield domain.name, lang_dir.name, str(doc) + + +def nfc_bytes(data): + """NFC-normalize UTF-8 bytes, trimming partial trailing codepoints.""" + text = decode_trimmed(data) + if text is None: + return data + return unicodedata.normalize("NFC", text).encode("utf-8") + + +def read_doc(path, cap=0): + """Read NFC-normalized doc bytes, truncated to cap (0 = no cap).""" + with open(path, "rb") as f: + return nfc_bytes(f.read(cap) if cap else f.read()) + + +def drop(corpus, paths): + """Move docs to a sibling _dropped tree, keeping relative paths.""" + dropped_root = Path(str(corpus).rstrip("/") + "_dropped") + for p in paths: + src = Path(p) + dst = dropped_root / src.relative_to(corpus) + dst.parent.mkdir(parents=True, exist_ok=True) + src.rename(dst) + + +def job_count(processes=None): + """Resolve to concrete worker count (None = all cores).""" + return mp.cpu_count() if processes is None else max(1, processes) + + +def chunks(seq, size): + """Split into chunks of at most size items.""" + size = max(1, size) + return [seq[i:i + size] for i in range(0, len(seq), size)] + + +def job_chunks(seq, jobs): + """One contiguous chunk per job.""" + return chunks(seq, -(-len(seq) // job_count(jobs))) + + +_SHARED = () + + +def set_shared(*args): + """MapPool initializer: stash per-worker constants.""" + global _SHARED + _SHARED = args + + +def shared(): + return _SHARED @contextmanager -def MapPool(processes=None, initializer=None, initargs=None, maxtasksperchild=None, chunksize=1): - """ - Contextmanager to express the common pattern of not using multiprocessing if - only 1 job is allocated (for example for debugging reasons) - """ - if processes is None: - processes = mp.cpu_count() + 4 +def MapPool(processes=None, initializer=None, initargs=None, chunksize=1): + """Process pool that falls back to serial map when processes=1.""" + processes = job_count(processes) if processes > 1: - with closing( mp.Pool(processes, initializer, initargs, maxtasksperchild)) as pool: - f = lambda fn, chunks: pool.imap_unordered(fn, chunks, chunksize=chunksize) - yield f + ctx = mp.get_context('fork') if sys.platform == 'darwin' else mp + with ctx.Pool(processes, initializer, initargs) as pool: + yield lambda fn, chunks: pool.imap_unordered(fn, chunks, chunksize) else: if initializer is not None: initializer(*initargs) - f = imap - yield f - - if processes > 1: - pool.join() + yield map diff --git a/py3langid/train/dedup.py b/py3langid/train/dedup.py new file mode 100644 index 00000000..6b74f41d --- /dev/null +++ b/py3langid/train/dedup.py @@ -0,0 +1,46 @@ +"""Cross-domain exact line dedup per language (first occurrence kept).""" +import sys +from collections import defaultdict +from pathlib import Path + +from .common import MIN_DOC, drop, walk_corpus + +MIN_LINE = 60 + + +def dedup(corpus): + """Returns (lines_removed, docs_rewritten, docs_dropped).""" + removed = touched = 0 + dropped = [] + docs_by_lang = defaultdict(list) + for _domain, lang, path in walk_corpus(corpus, skip_langs=("zxx",)): + docs_by_lang[lang].append(Path(path)) + for docs in docs_by_lang.values(): + s = set() + for doc in docs: + lines = doc.read_bytes().split(b"\n") + new = set() + kept = [] + for ln in lines: + if len(ln) >= MIN_LINE: + if ln in s or ln in new: + removed += 1 + continue + new.add(ln) + kept.append(ln) + if len(kept) != len(lines): + out = b"\n".join(kept) + if len(out.strip()) < MIN_DOC: # too little left to train on + dropped.append(doc) + continue # a dropped doc's lines must stay usable elsewhere + doc.write_bytes(out) + touched += 1 + s |= new + drop(corpus, dropped) + return removed, touched, len(dropped) + + +if __name__ == "__main__": + removed, touched, dropped = dedup(sys.argv[1]) + print(f"dedup: removed {removed} duplicate lines across {touched} docs, " + f"dropped {dropped} docs left under {MIN_DOC} bytes") diff --git a/py3langid/train/gather_data.py b/py3langid/train/gather_data.py new file mode 100644 index 00000000..cde0bdec --- /dev/null +++ b/py3langid/train/gather_data.py @@ -0,0 +1,325 @@ +"""Gather a multi-domain training corpus (see TRAINING.md).""" + +import argparse +import bz2 +import json +import lzma +import re +import shutil +import tarfile +import time +import urllib.request +from collections import defaultdict +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +from ..langid import MODEL_DIR, MODEL_FILE +from ..modelio import load_model +from .common import DOC_CAP, MIN_DOC, SPLIT_SCRIPT, chunks, route_script +from .sources import CC100_CODE, ISO3, LEIPZIG_NAME, WIKI_CODE + +TATOEBA_URL = "https://downloads.tatoeba.org/exports/sentences.tar.bz2" +CC100_URL = "https://data.statmt.org/cc-100/{code}.txt.xz" +CIRRUS_INDEX = "https://dumps.wikimedia.org/other/cirrus_search_index/" +CIRRUS_URL = CIRRUS_INDEX + "{date}/index_name%3D{code}wiki_content/{code}wiki_content-{date}-00000.json.bz2" +LEIPZIG_URL = "https://downloads.wortschatz-leipzig.de/corpora/{name}.tar.gz" + +CC100_RANGE = 2 * 1024 * 1024 +RAW_CACHE = Path("raw_downloads") # downloads kept on disk, reused on re-gather + +USER_AGENT = "py3langid-gather/0.1 (https://github.com/adbar/py3langid)" + + +def fetch(url, headers=None, retries=3): + req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT, **(headers or {})}) + for attempt in range(retries): + try: + return urllib.request.urlopen(req, timeout=120) + except urllib.error.HTTPError as e: + if e.code == 404 or attempt == retries - 1: + raise + time.sleep(60 if e.code == 429 else 5) + except Exception: + if attempt == retries - 1: + raise + time.sleep(5) + + +def fetch_cached(url, cache_path, headers=None): + """Download to cache_path once; return open binary handle.""" + if not cache_path.exists(): + cache_path.parent.mkdir(parents=True, exist_ok=True) + tmp = cache_path.with_name(cache_path.name + ".tmp") + with fetch(url, headers) as resp, open(tmp, "wb") as f: + shutil.copyfileobj(resp, f) + tmp.replace(cache_path) + return open(cache_path, "rb") + + +class _TeeReader: + """Tee response bytes to a cache file; kept on finalize().""" + + def __init__(self, resp, cache_path): + self.resp = resp + self.path = cache_path + cache_path.parent.mkdir(parents=True, exist_ok=True) + self.tmp = cache_path.with_name(cache_path.name + ".tmp") + self.f = open(self.tmp, "wb") # noqa: SIM115 + + def read(self, n=-1): + chunk = self.resp.read(n) + self.f.write(chunk) + return chunk + + def finalize(self): + self.f.close() + self.resp.close() + self.tmp.replace(self.path) + + def discard(self): + self.f.close() + self.resp.close() + self.tmp.unlink(missing_ok=True) + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc, tb): + self.finalize() if exc_type is None else self.discard() + + +def model_langs(): + return list(dict.fromkeys(load_model(MODEL_DIR / MODEL_FILE)[2])) # alias columns collapse + + +class DocWriter: + """Route docs by script, number files, enforce per-dir caps.""" + + def __init__(self, out_dir, max_docs): + self.out_dir = out_dir + self.lang = out_dir.name + self.max_docs = max_docs + self.spec = SPLIT_SCRIPT.get(self.lang) + self.counts = defaultdict(int) + + def write(self, doc): + d = route_script(self.out_dir, doc) + if self.counts[d.name] < self.max_docs: + d.mkdir(parents=True, exist_ok=True) + (d / f"doc{self.counts[d.name]:04d}.txt").write_bytes(doc) + self.counts[d.name] += 1 + + @property + def done(self): + if self.spec: + return min(self.counts[self.lang], + self.counts[self.spec.alt]) >= self.max_docs + return self.counts[self.lang] >= self.max_docs + + @property + def total(self): + return sum(self.counts.values()) + + +def valid_doc(doc): + """Strip, truncate to DOC_CAP, drop stubs. Returns bytes or None.""" + doc = doc.strip()[:DOC_CAP] + return doc if len(doc) >= MIN_DOC else None + + +def valid_docs(docs): + return (d for d in map(valid_doc, docs) if d is not None) + + +def write_docs(out_dir, docs, max_docs): + w = DocWriter(out_dir, max_docs) + for doc in valid_docs(docs): + w.write(doc) + if w.done: + break + return w.total + + +def gather_cc100(out_root, lang, max_docs): + code = CC100_CODE.get(lang, lang) + with fetch_cached(CC100_URL.format(code=code), + RAW_CACHE / "cc100" / f"{code}.txt.xz.head{CC100_RANGE}", + {"Range": f"bytes=0-{CC100_RANGE - 1}"}) as resp: + data = resp.read() + out = [] + dec = lzma.LZMADecompressor() + try: + for i in range(0, len(data), 1 << 16): + out.append(dec.decompress(data[i:i + (1 << 16)])) + except lzma.LZMAError: + pass + docs = b"".join(out).split(b"\n\n")[:-1] + return write_docs(out_root / "cc100" / lang, docs, max_docs) + + +def gather_wiki(out_root, lang, max_docs, date): + code = WIKI_CODE.get(lang, lang) + stem = f"{code}wiki-{date}.json.bz2.head" + cache = RAW_CACHE / "wiki" / f"{stem}{max_docs}" + usable = [p for p in sorted(cache.parent.glob(f"{stem}*")) + if p.name[len(stem):].isdigit() + and int(p.name[len(stem):]) >= max_docs] + if usable: + resp = open(usable[0], "rb") # noqa: SIM115 + else: + resp = _TeeReader(fetch(CIRRUS_URL.format(code=code, date=date)), cache) + docs = [] + dec = bz2.BZ2Decompressor() + buf = b"" + read = 0 + with resp: + while len(docs) < max_docs and read < (1 << 28): # 256 MiB safety cap + chunk = resp.read(1 << 18) + if not chunk: + break + read += len(chunk) + buf += dec.decompress(chunk) + *lines, buf = buf.split(b"\n") + for line in lines: + if b'"text"' not in line: + continue + text = json.loads(line).get("text") + if text: + doc = text.encode("utf-8").strip() + if len(doc) >= MIN_DOC: + docs.append(doc) + if isinstance(resp, _TeeReader) and len(docs) < max_docs: + resp.path = cache.with_name(f"{stem}{len(docs)}") # truncated stream + return write_docs(out_root / "wiki" / lang, docs, max_docs) + + +def _writer_counts(writers): + return {name: n for w in writers.values() for name, n in w.counts.items() if n} + + +def gather_tatoeba(out_root, langs, max_docs, per_doc): + by_iso3 = {ISO3[lang]: lang for lang in langs if lang in ISO3} + remaining = set(by_iso3.values()) + buf = defaultdict(list) + writers = {} + with fetch_cached(TATOEBA_URL, RAW_CACHE / "tatoeba" / "sentences.tar.bz2") as resp, \ + tarfile.open(fileobj=resp, mode="r|bz2") as tar: + for member in tar: + if not member.name.endswith("sentences.csv"): + continue + for raw in tar.extractfile(member): + parts = raw.decode("utf-8").rstrip("\n").split("\t") + if len(parts) != 3: + continue + lang = by_iso3.get(parts[1]) + if lang is None or lang not in remaining: + continue + buf[lang].append(parts[2]) + if len(buf[lang]) == per_doc: + doc = valid_doc("\n".join(buf[lang]).encode("utf-8")) + buf[lang] = [] + if doc is None: + continue + if lang not in writers: + writers[lang] = DocWriter(out_root / "tatoeba" / lang, max_docs) + writers[lang].write(doc) + if writers[lang].done: + remaining.discard(lang) + if not remaining: + return _writer_counts(writers) + return _writer_counts(writers) + + + +def gather_leipzig(out_root, lang, max_docs, per_doc): + name = LEIPZIG_NAME.get(lang) + if not name: + return 0 + sentences = [] + with fetch_cached(LEIPZIG_URL.format(name=name), + RAW_CACHE / "leipzig" / f"{name}.tar.gz") as resp, \ + tarfile.open(fileobj=resp, mode="r|gz") as tar: + for member in tar: + if not member.name.endswith("-sentences.txt"): + continue + for raw in tar.extractfile(member): + parts = raw.decode("utf-8", errors="replace").rstrip("\n").split("\t", 1) + if len(parts) == 2: + sentences.append(parts[1]) + docs = ("\n".join(c).encode("utf-8") for c in chunks(sentences, per_doc)) + return write_docs(out_root / "leipzig" / lang, docs, max_docs) + + +def latest_cirrus_date(): + html = fetch(CIRRUS_INDEX).read().decode() + dates = sorted(set(re.findall(r'href="(\d{8})/"', html))) + if not dates: + raise RuntimeError(f"no dumps listed at {CIRRUS_INDEX}; pass --wiki-date") + return dates[-2] if len(dates) > 1 else dates[-1] + + +def _doc_count(domain_dir, lang): + # only the primary dir gates completion: minority-script dirs may never + # fill from mono-script sources (topup covers them) + return sum(1 for _ in (domain_dir / lang).glob("*.txt")) + + +def per_lang_domain(name, func, langs, jobs, out_root, max_docs): + todo = [lang for lang in langs + if _doc_count(out_root / name, lang) < max_docs] + if len(todo) < len(langs): + print(f"{name}: {len(langs) - len(todo)} langs already complete") + counts = {} + + def one(lang): + try: + counts[lang] = func(lang) + except Exception as e: + print(f"{name}/{lang}: SKIP ({e})") + + with ThreadPoolExecutor(jobs) as pool: + list(pool.map(one, todo)) + for lang in sorted(counts): + print(f"{name}/{lang}: {counts[lang]} docs") + return counts + + +def main(argv=None): + parser = argparse.ArgumentParser() + parser.add_argument("--output", required=True, help="corpus output directory") + parser.add_argument("--langs", help="comma-separated language codes (default: the shipped model's labels)") + parser.add_argument("--domains", default="tatoeba,cc100,wiki,leipzig", help="comma-separated subset of domains") + parser.add_argument("--max-docs-per-lang", type=int, default=300, help="cap per language per domain (default: 300)") + parser.add_argument("--sentences-per-doc", type=int, default=50, help="sentences per doc (default: 50)") + parser.add_argument("--jobs", type=int, default=4, help="parallel downloads (cc100/wiki)") + parser.add_argument("--wiki-date", help="cirrus dump date YYYYMMDD (default: latest complete)") + args = parser.parse_args(argv) + + langs = args.langs.split(",") if args.langs else model_langs() + domains = args.domains.split(",") + out_root = Path(args.output) + out_root.mkdir(parents=True, exist_ok=True) + max_docs = args.max_docs_per_lang + + if "cc100" in domains: + per_lang_domain("cc100", lambda lang: gather_cc100(out_root, lang, max_docs), langs, args.jobs, out_root, max_docs) + if "wiki" in domains: + date = args.wiki_date or latest_cirrus_date() + print(f"wiki: cirrus dump {date}") + per_lang_domain("wiki", lambda lang: gather_wiki(out_root, lang, max_docs, date), langs, args.jobs, out_root, max_docs) + if "tatoeba" in domains: + counts = gather_tatoeba(out_root, langs, max_docs, args.sentences_per_doc) + print(f"tatoeba: {len(counts)} langs") + for lang in sorted(set(langs) - set(counts)): + print(f"tatoeba/{lang}: 0 docs") + if "leipzig" in domains: + per_lang_domain("leipzig", + lambda lang: gather_leipzig(out_root, lang, max_docs, args.sentences_per_doc), + langs, args.jobs, out_root, max_docs) + if "topup" in domains: + from .topup import gather_topup + gather_topup(out_root, langs, max_docs, args.jobs) + + +if __name__ == "__main__": + main() diff --git a/py3langid/train/index.py b/py3langid/train/index.py deleted file mode 100644 index 0ac78480..00000000 --- a/py3langid/train/index.py +++ /dev/null @@ -1,267 +0,0 @@ -""" -index.py - -Index a corpus that is stored in a directory hierarchy as follows: - -- corpus - - domain1 - - language1 - - file1 - - file2 - - ... - - language2 - - ... - - domain2 - - language1 - - file1 - - file2 - - ... - - language2 - - ... - - ... - -This produces 3 files: -* index: a list of paths, together with the langid and domainid as integers -* lang_index: a list of languages in ascending order of id, with the count for each -* domain_index: a list of domains in ascending order of id, with the count for each - -Marco Lui, January 2013 - -Copyright 2013 Marco Lui . All rights reserved. - -Redistribution and use in source and binary forms, with or without modification, are -permitted provided that the following conditions are met: - - 1. Redistributions of source code must retain the above copyright notice, this list of - conditions and the following disclaimer. - - 2. Redistributions in binary form must reproduce the above copyright notice, this list - of conditions and the following disclaimer in the documentation and/or other materials - provided with the distribution. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDER ``AS IS'' AND ANY EXPRESS OR IMPLIED -WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND -FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR -CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR -CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR -SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON -ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING -NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF -ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - -The views and conclusions contained in the software and documentation are those of the -authors and should not be interpreted as representing official policies, either expressed -or implied, of the copyright holder. -""" - -###### -# Default values -# Can be overriden with command-line options -###### -TRAIN_PROP = 1.0 # probability than any given document is selected -MIN_DOMAIN = 1 # minimum number of domains a language must be present in to be included - -import argparse -import csv -import os -import random - -from collections import defaultdict - -import numpy - -from .common import Enumerator, makedir - - -class CorpusIndexer(object): - """ - Class to index the contents of a corpus - """ - def __init__(self, root, min_domain=MIN_DOMAIN, proportion=TRAIN_PROP, langs=None, domains=None): - self.root = root - self.min_domain = min_domain - self.proportion = proportion - - if langs is None: - self.lang_index = defaultdict(Enumerator()) - else: - # pre-specified lang set - self.lang_index = {(k,v) for v,k in enumerate(langs)} - - if domains is None: - self.domain_index = defaultdict(Enumerator()) - else: - # pre-specified domain set - self.domain_index = dict((k,v) for v,k in enumerate(domains)) - - self.coverage_index = defaultdict(set) - self.items = list() - - self.index(root) - self.prune_min_domain(self.min_domain) - - def index(self, root): - # build a list of paths - paths = [] - for dirpath, dirnames, filenames in os.walk(root, followlinks=True): - for docname in filenames: - if random.random() < self.proportion: - # Each file has 'proportion' chance of being selected. - path = os.path.join(dirpath, docname) - - # split the dirpath into identifying components - d, lang = os.path.split(dirpath) - d, domain = os.path.split(d) - - # index the language and the domain - try: - # TODO: If lang is pre-specified but not domain, we can end up - # enumerating empty domains. - domain_id = self.domain_index[domain] - lang_id = self.lang_index[lang] - except KeyError: - # lang or domain outside a pre-specified set so - # skip this document. - continue - - # add the domain-lang relation to the coverage index - self.coverage_index[domain].add(lang) - - # add the item to our list - self.items.append((domain_id,lang_id,docname,path)) - - - def prune_min_domain(self, min_domain): - # prune files for all languages that do not occur in at least min_domain - - # Work out which languages to reject as they are not present in at least - # the required number of domains - lang_domain_count = defaultdict(int) - for langs in self.coverage_index.values(): - for lang in langs: - lang_domain_count[lang] += 1 - reject_langs = set( l for l in lang_domain_count if lang_domain_count[l] < min_domain) - - # Remove the languages from the indexer - if reject_langs: - #print "reject (<{0} domains): {1}".format(min_domain, sorted(reject_langs)) - reject_ids = set(self.lang_index[l] for l in reject_langs) - - new_lang_index = defaultdict(Enumerator()) - lm = dict() - for k,v in self.lang_index.items(): - if v not in reject_ids: - new_id = new_lang_index[k] - lm[v] = new_id - - # Eliminate all entries for the languages - self.items = [ (d, lm[l], n, p) for (d, l, n, p) in self.items if l in lm] - - self.lang_index = new_lang_index - - - @property - def dist_lang(self): - """ - @returns A vector over frequency counts for each language - """ - retval = numpy.zeros((len(self.lang_index),), dtype='int') - for d, l, n, p in self.items: - retval[l] += 1 - return retval - - @property - def dist_domain(self): - """ - @returns A vector over frequency counts for each domain - """ - retval = numpy.zeros((len(self.domain_index),), dtype='int') - for d, l, n, p in self.items: - retval[d] += 1 - return retval - - # TODO: Remove this as it should no longer be needed - @property - def classmaps(self): - num_instances = len(self.items) - if num_instances == 0: - raise ValueError("no items indexed!") - cm_domain = numpy.zeros((num_instances, len(self.domain_index)), dtype='bool') - cm_lang = numpy.zeros((num_instances, len(self.lang_index)), dtype='bool') - - # Populate the class maps - for docid, (domain_id, lang_id, docname, path) in enumerate(self.items): - cm_domain[docid, domain_id] = True - cm_lang[docid, lang_id] = True - return cm_domain, cm_lang - - @property - def paths(self): - return [ p for (d,l,n,p) in self.items ] - - -if __name__ == "__main__": - parser = argparse.ArgumentParser() - parser.add_argument("-p","--proportion", type=float, default=TRAIN_PROP, - help="proportion of training data to use" ) - parser.add_argument("-m","--model", help="save output to MODEL_DIR", metavar="MODEL_DIR") - parser.add_argument("-d","--domain", metavar="DOMAIN", action='append', - help="use DOMAIN - can be specified multiple times (uses all domains found if not specified)") - parser.add_argument("-l","--lang", metavar="LANG", action='append', - help="use LANG - can be specified multiple times (uses all langs found if not specified)") - parser.add_argument("--min_domain", type=int, default=MIN_DOMAIN, - help="minimum number of domains a language must be present in" ) - parser.add_argument("corpus", help="read corpus from CORPUS_DIR", metavar="CORPUS_DIR") - - args = parser.parse_args() - - corpus_name = os.path.basename(args.corpus) - if args.model: - model_dir = args.model - else: - model_dir = os.path.join('.', corpus_name+'.model') - - makedir(model_dir) - - langs_path = os.path.join(model_dir, 'lang_index') - domains_path = os.path.join(model_dir, 'domain_index') - index_path = os.path.join(model_dir, 'paths') - - # display paths - print("corpus path:", args.corpus) - print("model path:", model_dir) - print("writing langs to:", langs_path) - print("writing domains to:", domains_path) - print("writing index to:", index_path) - - indexer = CorpusIndexer(args.corpus, min_domain=args.min_domain, proportion=args.proportion, - langs = args.lang, domains = args.domain) - - # Compute mappings between files, languages and domains - lang_dist = indexer.dist_lang - lang_index = indexer.lang_index - lang_info = ' '.join(("{0}({1})".format(k, lang_dist[v]) for k,v in lang_index.items())) - print("langs({0}): {1}".format(len(lang_dist), lang_info)) - - domain_dist = indexer.dist_domain - domain_index = indexer.domain_index - domain_info = ' '.join(("{0}({1})".format(k, domain_dist[v]) for k,v in domain_index.items())) - print("domains({0}): {1}".format(len(domain_dist), domain_info)) - - print("identified {0} files".format(len(indexer.items))) - - # output the language index - with open(langs_path,'w') as f: - writer = csv.writer(f) - writer.writerows((l, lang_dist[lang_index[l]]) - for l in sorted(lang_index.keys(), key=lang_index.get)) - - # output the domain index - with open(domains_path,'w') as f: - writer = csv.writer(f) - writer.writerows((d, domain_dist[domain_index[d]]) - for d in sorted(domain_index.keys(), key=domain_index.get)) - - # output items found - with open(index_path,'w') as f: - writer = csv.writer(f) - writer.writerows( (d,l,p) for (d,l,n,p) in indexer.items ) diff --git a/py3langid/train/scanner.py b/py3langid/train/scanner.py index d8ad09a3..248877ad 100644 --- a/py3langid/train/scanner.py +++ b/py3langid/train/scanner.py @@ -1,240 +1,55 @@ -""" -scanner.py - -Assemble a "feature scanner" using Aho-Corasick string matching. -This takes a list of features (byte sequences) and builds a DFA -that when run on a byte stream can identify how often each of -the features is present in a single pass over the stream. +"""Aho-Corasick DFA: longest-match feature scanner.""" -Marco Lui, January 2013 - -Copyright 2013 Marco Lui . All rights reserved. - -Redistribution and use in source and binary forms, with or without modification, are -permitted provided that the following conditions are met: - - 1. Redistributions of source code must retain the above copyright notice, this list of - conditions and the following disclaimer. - - 2. Redistributions in binary form must reproduce the above copyright notice, this list - of conditions and the following disclaimer in the documentation and/or other materials - provided with the distribution. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDER ``AS IS'' AND ANY EXPRESS OR IMPLIED -WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND -FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR -CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR -CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR -SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON -ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING -NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF -ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - -The views and conclusions contained in the software and documentation are those of the -authors and should not be interpreted as representing official policies, either expressed -or implied, of the copyright holder. -""" - -import argparse import array -import os -import pickle -from collections import deque, defaultdict -from .common import read_features - -class Scanner(object): - alphabet = map(chr, range(1<<8)) - """ - Implementation of Aho-Corasick string matching. - This class should be instantiated with a set of keywords, which - will then be the only tokens generated by the class's search method, - """ - @classmethod - def from_file(cls, path): - with open(path) as f: - tk_nextmove, tk_output, feats = pickle.load(f) - if isinstance(feats, dict): - # The old scanner format had two identical dictionaries as the last - # two items in the tuple. This format can still be used by langid.py, - # but it does not carry the feature list, and so cannot be unpacked - # back into a Scanner object. - raise ValueError("old format scanner - please retrain. see code for details.") - # tk_output is a mapping from state to a list of feature indices. - # because of the way the scanner class is written, it needs a mapping - # from state to the feature itself. We rebuild this here. - tk_output_f = dict( (k,[feats[i] for i in v]) for k,v in tk_output.iteritems() ) - scanner = cls.__new__(cls) - scanner.__setstate__((tk_nextmove, tk_output_f)) - return scanner - - def __init__(self, keywords): - self.build(keywords) - - def __call__(self, value): - return self.search(value) - - def build(self, keywords): - goto = dict() - fail = dict() - output = defaultdict(set) - - # Algorithm 2 - newstate = 0 - for a in keywords: - state = 0 - j = 0 - while (j < len(a)) and (state, a[j]) in goto: - state = goto[(state, a[j])] - j += 1 - for p in range(j, len(a)): - newstate += 1 - goto[(state, a[p])] = newstate - #print "(%d, %s) -> %d" % (state, a[p], newstate) - state = newstate - output[state].add(a) - for a in self.alphabet: - if (0,a) not in goto: - goto[(0,a)] = 0 - - # Algorithm 3 - queue = deque() - for a in self.alphabet: - if goto[(0,a)] != 0: - s = goto[(0,a)] - queue.append(s) - fail[s] = 0 - while queue: - r = queue.popleft() - for a in self.alphabet: - if (r,a) in goto: - s = goto[(r,a)] - queue.append(s) - state = fail[r] - while (state,a) not in goto: - state = fail[state] - fail[s] = goto[(state,a)] - #print "f(%d) -> %d" % (s, goto[(state,a)]), output[fail[s]] - if output[fail[s]]: - output[s].update(output[fail[s]]) - - # Algorithm 4 - self.nextmove = {} - for a in self.alphabet: - self.nextmove[(0,a)] = goto[(0,a)] - if goto[(0,a)] != 0: - queue.append(goto[(0,a)]) - while queue: - r = queue.popleft() - for a in self.alphabet: - if (r,a) in goto: - s = goto[(r,a)] - queue.append(s) - self.nextmove[(r,a)] = s - else: - self.nextmove[(r,a)] = self.nextmove[(fail[r],a)] - - # convert the output to tuples, as tuple iteration is faster - # than set iteration - self.output = dict((k, tuple(output[k])) for k in output) +from collections import defaultdict, deque - # Next move encoded as a single array. The index of the next state - # is located at current state * alphabet size + ord(c). - # The choice of 'H' array typecode limits us to 64k states. - def generate_nm_arr(typecode): - def nextstate_iter(): - # State count starts at 0, so the number of states is the number of i - # the last state (newstate) + 1 - for state in range(newstate+1): - for letter in self.alphabet: - yield self.nextmove[(state, letter)] - return array.array(typecode, nextstate_iter()) - try: - self.nm_arr = generate_nm_arr('H') - except OverflowError: - # Could not fit in an unsigned short array, let's try an unsigned long array. - self.nm_arr = generate_nm_arr('L') - - def __getstate__(self): - """ - Compiled nextmove and output. - """ - return (self.nm_arr, self.output) - - def __setstate__(self, value): - nm_array, output = value - self.nm_arr = nm_array - self.output = output - self.nextmove = {} - for i, next_state in enumerate(nm_array): - state = i / 256 - letter = chr(i % 256) - self.nextmove[(state, letter)] = next_state - - def search(self, string): - state = 0 - for letter in map(ord,string): - state = self.nm_arr[(state << 8) + letter] - for key in self.output.get(state, []): - yield key def build_scanner(features): - """ - In difference to the Scanner class, this function unwraps a layer of indirection in - the detection of features. It translates the string output of the scanner's output - mapping into the index values (positions in the list) of the features in the supplied - feature set. This is very useful where we are only interested in the relative frequencies - of features. - - @param features a list of features (byte sequences) - @returns a compiled scanner model - """ - feat_index = index(features) - - # Build the actual scanner - print("building scanner") - scanner = Scanner(features) - tk_nextmove, raw_output = scanner.__getstate__() - - # tk_output is the output function of the scanner. It should generate indices into - # the feature space directly, as this saves a lookup - tk_output = {} - for k,v in raw_output.items(): - tk_output[k] = tuple(feat_index[f] for f in v) - return tk_nextmove, tk_output - + """Compile features into a DFA. Returns (rows, row_index, tk_output).""" + feat_index = {f: i for i, f in enumerate(features)} -def index(seq): - """ - Build an index for a sequence of items. Assumes - that the items in the sequence are unique. - @param seq the sequence to index - @returns a dictionary from item to position in the sequence - """ - return dict((k,v) for (v,k) in enumerate(seq)) - -if __name__ == "__main__": - parser = argparse.ArgumentParser() - parser.add_argument("input", metavar="INPUT", help="build a scanner for INPUT. If input is a directory, read INPUT/LDfeats") - parser.add_argument("-o","--output", help="output scanner to OUTFILE", metavar="OUTFILE") - args = parser.parse_args() - - if os.path.isdir(args.input): - input_path = os.path.join(args.input, 'LDfeats') - else: - input_path = args.input - - if args.output: - output_path = args.output - else: - output_path = input_path + '.scanner' - - # display paths - print("input path:", input_path) - print("output path:", output_path) - - nb_features = read_features(input_path) - tk_nextmove, tk_output = build_scanner(nb_features) - scanner = tk_nextmove, tk_output, nb_features - - with open(output_path, 'w') as f: - pickle.dump(scanner, f) - print("wrote scanner to {0}".format(output_path)) + children = defaultdict(dict) + newstate = 0 + ends = {} + for a in features: + state = 0 + j = 0 + while j < len(a) and a[j] in children[state]: + state = children[state][a[j]] + j += 1 + for p in range(j, len(a)): + newstate += 1 + children[state][a[p]] = newstate + state = newstate + ends[state] = feat_index[a] + + # fail links + DFA fill (row allocated only when a state has its own edges) + nstates = newstate + 1 + typecode = 'H' if nstates <= 1 << 16 else 'L' + rows = array.array(typecode, [0]) * 256 # state 0's row + row_index = array.array('L', [0]) * nstates + fail = array.array('L', [0]) * nstates + tk_output = array.array('l', [-1]) * nstates + for state, feat in ends.items(): + tk_output[state] = feat + queue = deque() + for a, s in children[0].items(): + rows[a] = s + queue.append(s) # fail[s] = 0 already + while queue: + r = queue.popleft() + fbase = row_index[fail[r]] << 8 + edges = children[r] + if edges: + row_index[r] = len(rows) >> 8 + rows.extend(rows[fbase:fbase + 256]) + else: + row_index[r] = fbase >> 8 + base = row_index[r] << 8 + for a, s in edges.items(): + fail[s] = rows[fbase + a] # = nextmove(fail[r], a), pre-overwrite + rows[base + a] = s + if tk_output[s] < 0: + tk_output[s] = tk_output[fail[s]] + queue.append(s) + return rows, row_index, tk_output diff --git a/py3langid/train/shards.py b/py3langid/train/shards.py new file mode 100644 index 00000000..ee8bf4e7 --- /dev/null +++ b/py3langid/train/shards.py @@ -0,0 +1,172 @@ +"""Per-(domain, lang) n-gram document-frequency shards with content-based caching.""" + +import hashlib +import marshal +import os +from collections import Counter, defaultdict + +import numpy as np + +from .common import ( + DOC_CAP, + MAX_NGRAM_ORDER, + MIN_NGRAM_ORDER, + TOKENIZE_ORDER, + MapPool, + chunks, + is_cjk_bigram, + job_chunks, + read_doc, + set_shared, + shared, +) + +MERGE_SHARDS_PER_CHUNK = 8 +COUNT_DTYPE = np.int32 + + +def doc_ngrams(data, max_order): + """Distinct byte n-grams in a doc, plus CJK codepoint bigrams.""" + terms = set() + n = len(data) + for i in range(n): + for k in range(MIN_NGRAM_ORDER, min(max_order, n - i) + 1): + terms.add(data[i:i + k]) + if i + TOKENIZE_ORDER <= n \ + and data[i] >= 0xE2 and data[i + 3] >= 0xE2: + term = data[i:i + TOKENIZE_ORDER] + if is_cjk_bigram(term): + terms.add(term) + return terms + + +def group_items(items): + """Group by (domain, lang), sorted.""" + groups = defaultdict(list) + for domain, lang, path in items: + groups[(domain, lang)].append(path) + return sorted(groups.items()) + + +def _group_key(paths): + """Cache key from doc metadata + tokenization constants.""" + h = hashlib.sha256() + h.update(f"{MIN_NGRAM_ORDER}\0{MAX_NGRAM_ORDER}\0{TOKENIZE_ORDER}\0" + f"{DOC_CAP}\0".encode()) + for p in sorted(paths): + st = os.stat(p) + h.update(f"{os.path.basename(p)}\0{st.st_size}\0{st.st_mtime_ns}\0".encode()) + return h.hexdigest() + + +def _build_shard(arg): + """Build one shard or reuse cached. Returns (shard_path, built).""" + shard_path, key, paths = arg + try: + with open(shard_path, 'rb') as f: + if marshal.load(f) == key: + return shard_path, False + except (OSError, EOFError, ValueError, TypeError): + pass + + docfreq = Counter() + for path in paths: + docfreq.update(doc_ngrams(read_doc(path, DOC_CAP), MAX_NGRAM_ORDER)) + + tmp_path = shard_path + '.tmp' + with open(tmp_path, 'wb') as f: + marshal.dump(key, f) + marshal.dump(dict(docfreq), f) + os.replace(tmp_path, shard_path) + return shard_path, True + + +def build_shards(items, shard_dir, jobs=None): + """Build/reuse cached shards. Returns [(domain, lang, shard_path), ...].""" + os.makedirs(shard_dir, exist_ok=True) + tasks = [] + shard_items = [] + for (domain, lang), paths in group_items(items): + shard_path = os.path.join(shard_dir, f"{domain}__{lang}") + tasks.append((shard_path, _group_key(paths), paths)) + shard_items.append((domain, lang, shard_path)) + + with MapPool(jobs) as f: + built = sum(new for _, new in f(_build_shard, tasks)) + print(f"shards: {built} built, {len(tasks) - built} cached") + return shard_items + + +def load_shard(shard_path): + """Load term → document frequency dict.""" + with open(shard_path, 'rb') as f: + marshal.load(f) + return marshal.load(f) + + +def _merge_chunk(chunk): + merged = Counter() + for _, _, shard_path in chunk: + merged.update(load_shard(shard_path)) + return merged + + +def merge_docfreq(shard_items, jobs=None): + """Global term → document frequency.""" + doc_count = Counter() + with MapPool(jobs) as f: + for partial in f(_merge_chunk, + chunks(shard_items, MERGE_SHARDS_PER_CHUNK)): + doc_count.update(partial) + return doc_count + + +def _select_counts(counts, feat_index): + """Intersect shard counts with feature index. Returns (indices, counts).""" + idx, vals = [], [] + if len(counts) < len(feat_index): + for feat, count in counts.items(): + i = feat_index.get(feat) + if i is not None: + idx.append(i) + vals.append(count) + else: + for feat, i in feat_index.items(): + count = counts.get(feat) + if count: + idx.append(i) + vals.append(count) + return np.asarray(idx, dtype=np.intp), np.asarray(vals, dtype=COUNT_DTYPE) + + +def _zero_matrices(nf, nl, nd): + return (np.zeros((nf, nl), dtype=COUNT_DTYPE), + np.zeros((nf, nd), dtype=COUNT_DTYPE), + np.zeros((nf, nl), dtype=COUNT_DTYPE)) + + +def _matrices_chunk(chunk): + feat_index, lang_index, domain_index = shared() + cm_lang, cm_domain, domcount = _zero_matrices( + len(feat_index), len(lang_index), len(domain_index)) + for domain, lang, shard_path in chunk: + idx, vals = _select_counts(load_shard(shard_path), feat_index) + j = lang_index[lang] + cm_lang[idx, j] += vals + cm_domain[idx, domain_index[domain]] += vals + domcount[idx, j] += 1 + return cm_lang, cm_domain, domcount + + +def count_matrices(shard_items, features, lang_index, domain_index, jobs=None): + """Returns (lang counts, domain counts, domain-presence) matrices.""" + feat_index = {f: i for i, f in enumerate(features)} + cm_lang, cm_domain, domcount = _zero_matrices( + len(features), len(lang_index), len(domain_index)) + with MapPool(jobs, set_shared, (feat_index, lang_index, domain_index)) as f: + for part_lang, part_domain, part_dom in f( + _matrices_chunk, job_chunks(shard_items, jobs)): + cm_lang += part_lang + cm_domain += part_domain + domcount += part_dom + return cm_lang, cm_domain, domcount diff --git a/py3langid/train/sources.py b/py3langid/train/sources.py new file mode 100644 index 00000000..5aa5f101 --- /dev/null +++ b/py3langid/train/sources.py @@ -0,0 +1,108 @@ +"""Per-source language code tables for the corpus gatherers.""" + +# Tatoeba ISO 639-3 codes per target lang; missing/uncertain codes yield 0 docs and are logged. +ISO3 = { + "af": "afr", "am": "amh", "an": "arg", "ar": "ara", "as": "asm", "az": "aze", + "arz": "arz", "ary": "ary", + "ba": "bak", "be": "bel", "bg": "bul", "bn": "ben", "br": "bre", "bs": "bos", + "ca": "cat", "crh": "crh", "cs": "ces", "cy": "cym", + "da": "dan", "de": "deu", "dz": "dzo", + "el": "ell", "en": "eng", "eo": "epo", "es": "spa", "et": "est", "eu": "eus", + "fa": "pes", "fi": "fin", "fo": "fao", "fr": "fra", "fy": "fry", + "ga": "gle", "gcf": "gcf", "gd": "gla", "gl": "glg", "gu": "guj", + "ha": "hau", "he": "heb", "hi": "hin", "hr": "hrv", "ht": "hat", + "hu": "hun", "hy": "hye", + "id": "ind", "ig": "ibo", "is": "isl", "it": "ita", + "ja": "jpn", "jv": "jav", "ka": "kat", + "kab": "kab", "kk": "kaz", "km": "khm", "kn": "kan", "ko": "kor", + "ku": "kmr", "ky": "kir", + "la": "lat", "lb": "ltz", "lg": "lug", "lij": "lij", "ln": "lin", + "lo": "lao", "lt": "lit", "ltg": "ltg", "lv": "lvs", + "mg": "mlg", "mk": "mkd", "ml": "mal", "mn": "mon", "mr": "mar", + "ms": "zsm", "mt": "mlt", "my": "mya", + "ne": "npi", "nl": "nld", "nn": "nno", "no": "nob", "nso": "nso", + "oc": "oci", "om": "orm", "or": "ori", + "pa": "pan", "pcm": "pcm", "pl": "pol", "ps": "pus", "pt": "por", + "qu": "que", "ro": "ron", + "ru": "rus", "rw": "kin", "sa": "san", "se": "sme", "si": "sin", "sk": "slk", + "sl": "slv", "sn": "sna", "so": "som", "sq": "sqi", + "sr": "srp", "st": "sot", "sv": "swe", "sw": "swh", "tg": "tgk", + "ta": "tam", "te": "tel", "th": "tha", "tk": "tuk", "tl": "tgl", "tr": "tur", + "tt": "tat", "ug": "uig", "uk": "ukr", "ur": "urd", "uz": "uzb", + "vec": "vec", "vi": "vie", "vo": "vol", "wa": "wln", + "wuu": "wuu", + "xh": "xho", "yo": "yor", "yue": "yue", "zh": "cmn", "zu": "zul", + # batch 4: CommonLID coverage + "ace": "ace", "bcl": "bcl", "ext": "ext", + # batch 5 (fro/ars/aeb rejected -- verdicts in sweep25 ledger) + "uzs": "uzs", + "fuv": "fuv", "gcr": "gcr", "gom": "gom", "grc": "grc", + "gug": "gug", "guw": "guw", "hbo": "hbo", "kik": "kik", +} +CC100_CODE = {"zh": "zh-Hans"} +# cc100 has no data for: tt, ba, vec, tk, sn, st, nso, kab, crh +# nb is not gathered: it duplicated 'no' (Bokmål) and the two classes split +# the same language arbitrarily. Tatoeba nob feeds 'no' (decision 2026-08-26). +WIKI_CODE = { + "yue": "zh_yue", # Cantonese wiki is at zh-yue.wikipedia.org + "gug": "gn", "kik": "ki", "fuv": "ff", # batch 4 +} + +# Leipzig Corpora Collection: lang -> archive name (news preferred, wiki fallback). +# Scanned 2026-08-25; 7 langs have no Leipzig corpus (dz, km, ku, lo, rw, xh, zu). +LEIPZIG_NAME = { + "af": "afr_news_2020_30K", "am": "amh_wikipedia_2021_30K", + "arz": "arz_wikipedia_2021_100K", + "an": "arg_wikipedia_2021_30K", "ar": "ara_news_2022_1M", + "as": "asm_wikipedia_2021_100K", "az": "aze_news_2020_30K", + "ba": "bak_wikipedia_2021_30K", + "be": "bel_news_2020_100K", "bg": "bul_news_2022_1M", + "bn": "ben_news_2020_300K", "br": "bre_wikipedia_2021_100K", + "bs": "bos_news_2020_300K", "ca": "cat_news_2022_300K", + "cs": "ces_news_2023_1M", "cy": "cym_wikipedia_2021_100K", + "da": "dan_news_2022_300K", "de": "deu_news_2023_1M", + "el": "ell_news_2023_1M", "en": "eng_news_2023_1M", + "eo": "epo_wikipedia_2021_300K", "es": "spa_news_2023_1M", + "et": "est_news_2022_300K", "eu": "eus_news_2020_30K", + "fa": "fas_news_2023_1M", "fi": "fin_news_2022_1M", + "fo": "fao_news_2020_30K", "fr": "fra_news_2023_1M", + "fy": "fry_wikipedia_2021_100K", + "ga": "gle_wikipedia_2021_30K", "gl": "glg_wikipedia_2021_300K", + "gu": "guj_news_2020_30K", "ha": "hau_wikipedia_2021_30K", + "he": "heb_news_2020_1M", + "hi": "hin_news_2022_1M", "hr": "hrv_news_2020_1M", + "ht": "hat_wikipedia_2021_30K", "hu": "hun_news_2023_1M", + "hy": "hye_news_2021_30K", "id": "ind_news_2023_1M", + "is": "isl_news_2020_30K", "it": "ita_news_2023_1M", + "ja": "jpn_news_2023_100K", "jv": "jav_wikipedia_2021_100K", + "ka": "kat_news_2020_100K", "kk": "kaz_news_2020_30K", + "kn": "kan_wikipedia_2021_300K", "ko": "kor_news_2022_1M", + "ky": "kir_wikipedia_2021_300K", "la": "lat_wikipedia_2021_100K", + "lb": "ltz_wikipedia_2021_100K", "lt": "lit_news_2020_1M", + "lv": "lav_news_2020_300K", "mg": "plt_wikipedia_2021_100K", + "mk": "mkd_news_2020_100K", "ml": "mal_wikipedia_2021_300K", + "mn": "mon_news_2020_100K", "mr": "mar_news_2020_300K", + "ms": "msa_news_2019_100K", "mt": "mlt_news_2020_30K", + "ne": "nep_news_2020_300K", "nl": "nld_news_2023_1M", + "nn": "nno_wikipedia_2021_300K", "no": "nob_newscrawl_2019_1M", + "oc": "oci_wikipedia_2021_100K", "or": "ori_wikipedia_2021_100K", + "pa": "pan_wikipedia_2021_300K", "pl": "pol_news_2023_1M", + "ps": "pus_news_2020_100K", "pt": "por_news_2023_1M", + "qu": "que_wikipedia_2021_10K", "ro": "ron_news_2022_1M", + "ru": "rus_news_2023_1M", "sa": "san_wikipedia_2021_100K", + "se": "sme_wikipedia_2021_10K", + "si": "sin_wikipedia_2021_100K", "sk": "slk_news_2020_100K", + "sl": "slv_news_2020_1M", "sq": "sqi_news_2020_1M", + "sr": "srp_news_2023_30K", "sv": "swe_news_2023_1M", + "sw": "swa_news_2020_30K", "ta": "tam_news_2020_100K", + "te": "tel_news_2020_100K", "tg": "tgk_wikipedia_2021_100K", + "th": "tha_news_2020_30K", + "tk": "tuk_wikipedia_2021_30K", "tl": "tgl_news_2020_30K", + "tr": "tur_news_2023_1M", "tt": "tat_wikipedia_2021_100K", + "ug": "uig_wikipedia_2021_30K", "uk": "ukr_news_2023_1M", + "ur": "urd_news_2020_30K", "uz": "uzb_news_2020_30K", + "vec": "vec_wikipedia_2021_30K", "vi": "vie_news_2022_1M", + "vo": "vol_wikipedia_2021_100K", "wa": "wln_wikipedia_2021_30K", + "wuu": "wuu_wikipedia_2021_30K", + "zh": "zho_news_2020_300K", +} diff --git a/py3langid/train/stages.py b/py3langid/train/stages.py new file mode 100644 index 00000000..bc088f23 --- /dev/null +++ b/py3langid/train/stages.py @@ -0,0 +1,157 @@ +"""Corpus indexing, feature selection (DF/LD/IG), NB parameter estimation.""" + +import heapq +from collections import defaultdict + +import numpy as np + +from ..langid import visit_counts +from .common import ( + DF_TOKENS, + DOC_CAP, + SELECT_ORDERS, + MapPool, + job_chunks, + read_doc, + set_shared, + shared, + walk_corpus, +) + + +def index_corpus(root): + """Returns (items, langs, domains) in first-appearance walk order.""" + items = [] + langs, domains = {}, {} + for domain, lang, path in walk_corpus(root): + langs.setdefault(lang, None) + domains.setdefault(domain, None) + items.append((domain, lang, path)) + return items, list(langs), list(domains) + + +def ngram_select(doc_count, tokens_per_order=DF_TOKENS, orders=SELECT_ORDERS): + """Top tokens_per_order terms by DF at each admissible order.""" + buckets = defaultdict(list) + for term, count in doc_count.items(): + order = len(term) + if order in orders: + buckets[order].append((count, term)) + features = set() + for bucket in buckets.values(): + top = heapq.nsmallest(tokens_per_order, bucket, + key=lambda x: (-x[0], x[1])) + features.update(term for _, term in top) + return sorted(features) + + +def _xlogx(v): + """v * log(v), with 0*log(0) = 0.""" + log = np.zeros(v.shape, dtype=float) + np.log(v, where=v > 0, out=log) + return v * log + + +def entropy(v, axis=-1): + """Entropy (nats) of count vectors; all-zero → 0.""" + v = np.asarray(v, dtype=float) + total = v.sum(axis) + nonzero = total > 0 + safe = np.where(nonzero, total, 1.0) + return np.where(nonzero, np.log(safe) - _xlogx(v).sum(axis) / safe, 0.0) + + +def _binary_entropy(a, b): + # inlined two-column entropy: ~2x faster than entropy(np.stack([a, b])) + total = a + b + nonzero = total > 0 + safe = np.where(nonzero, total, 1.0) + return np.where(nonzero, + np.log(safe) - (_xlogx(a) + _xlogx(b)) / safe, 0.0) + + +def compute_IG(cm_pos, dist): + """Information gain per term. Returns (num_term,) array.""" + present = np.asarray(cm_pos, dtype=float) + dist = np.asarray(dist, dtype=float) + n = dist.sum() + t = present.sum(1) + return entropy(dist) - (t * entropy(present) + + (n - t) * entropy(dist - present)) / n + + +def ld_weights(cm_lang, lang_dist, domain_ig): + """Yield per-language LD weight arrays (IG_lang − IG_domain).""" + dist = np.asarray(lang_dist, dtype=float) + n = dist.sum() + prior = _binary_entropy(dist, n - dist) + t = cm_lang.sum(1, dtype=np.int64).astype(float) + rest = n - t + for j, dist_j in enumerate(dist): + pos = np.asarray(cm_lang[:, j], dtype=float) + neg = dist_j - pos + yield prior[j] - (t * _binary_entropy(pos, t - pos) + + rest * _binary_entropy(neg, rest - neg)) / n \ + - domain_ig + + +def select_LD_features(ld_columns, feats_per_lang, present): + """Top feats_per_lang per language by LD weight. Returns union of row indices.""" + if feats_per_lang < 1: + raise ValueError("feats_per_lang must be >= 1") + selected = set() + for j, lang_w in enumerate(ld_columns): + cand = np.flatnonzero(present[:, j]) + selected.update(cand[np.argsort(lang_w[cand])[-feats_per_lang:]].tolist()) + return selected + + +_JUNK_BYTES = frozenset( + b"0123456789 \t\n\r\x0b\x0c" + b"!\"#$%&'()*+,-./:;<=>?@[\\]^_`{|}~") + + +def cluster_features(cm_lang, lang_dist, lang_index, feats, base, clusters, k): + """Top-k new features per confusable cluster by cluster-restricted IG.""" + base = set(base) + selected = set(base) + for cluster in clusters: + if any(lang not in lang_index for lang in cluster): + continue + cols = [lang_index[lang] for lang in cluster] + ig = compute_IG(cm_lang[:, cols], lang_dist[cols]) + taken = 0 + for t in np.argsort(ig)[::-1]: + t = int(t) + if t in selected or all(b in _JUNK_BYTES for b in feats[t]): + continue + selected.add(t) + taken += 1 + if taken >= k: + break + return selected - base + + +def _feature_counts_chunk(chunk): + nm, rowbase, out, n_feats, num_langs = shared() + counts = np.zeros((n_feats, num_langs), dtype=np.int64) + for col, path in chunk: + visits = visit_counts(nm, rowbase, out, read_doc(path, DOC_CAP)) + if visits: + counts[list(visits), col] += np.fromiter( + visits.values(), dtype=np.int64, count=len(visits)) + return counts + + +def feature_counts(items, tk_nextmove, tk_row, tk_output, n_feats, lang_index, + jobs=None): + """Per-(feature, lang) longest-match counts via DFA walk.""" + tasks = [(lang_index[lang], path) for _, lang, path in items] + counts = np.zeros((n_feats, len(lang_index)), dtype=np.int64) + rowbase = [r << 8 for r in tk_row] + with MapPool(jobs, set_shared, + (tk_nextmove, rowbase, tk_output, n_feats, + len(lang_index))) as f: + for partial in f(_feature_counts_chunk, job_chunks(tasks, jobs)): + counts += partial + return counts diff --git a/py3langid/train/tokenize.py b/py3langid/train/tokenize.py deleted file mode 100644 index 4e1ad857..00000000 --- a/py3langid/train/tokenize.py +++ /dev/null @@ -1,253 +0,0 @@ -""" -tokenize.py - -Tokenizer for langid.py training system. This takes a list of files and tokenizes them -in parallel. - -Marco Lui, January 2013 - -Copyright 2013 Marco Lui . All rights reserved. - -Redistribution and use in source and binary forms, with or without modification, are -permitted provided that the following conditions are met: - - 1. Redistributions of source code must retain the above copyright notice, this list of - conditions and the following disclaimer. - - 2. Redistributions in binary form must reproduce the above copyright notice, this list - of conditions and the following disclaimer in the documentation and/or other materials - provided with the distribution. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDER ``AS IS'' AND ANY EXPRESS OR IMPLIED -WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND -FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR -CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR -CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR -SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON -ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING -NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF -ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - -The views and conclusions contained in the software and documentation are those of the -authors and should not be interpreted as representing official policies, either expressed -or implied, of the copyright holder. -""" - -###### -# Default values -# Can be overriden with command-line options -###### - -MIN_NGRAM_ORDER = 1 # smallest order of n-grams to consider -MAX_NGRAM_ORDER = 4 # largest order of n-grams to consider -TOP_DOC_FREQ = 15000 # number of tokens to consider for each order -NUM_BUCKETS = 64 # number of buckets to use in k-v pair generation -CHUNKSIZE = 50 # maximum size of chunk (number of files tokenized - less = less memory use) - -import argparse -import atexit -import csv -import marshal -import multiprocessing as mp -import os -import random -import shutil -import tempfile - -from itertools import tee -from collections import defaultdict - -from .common import makedir, chunk, MapPool - -class NGramTokenizer(object): - def __init__(self, min_order=1, max_order=3): - self.min_order = min_order - self.max_order = max_order - - def __call__(self, seq): - min_order = self.min_order - max_order = self.max_order - t = tee(seq, max_order) - for i in range(max_order): - for _ in range(i): - # advance iterators, ignoring result - t[i].next() - while True: - token = ''.join(tn.next() for tn in t) - if len(token) < max_order: break - for n in range(min_order-1, max_order): - yield token[:n+1] - for a in range(max_order-1): - for b in range(min_order, max_order-a): - yield token[a:a+b] - -@atexit.register -def cleanup(): - global b_dirs, complete - try: - if not complete: - for d in b_dirs: - shutil.rmtree(d) - except NameError: - # Failed before globals defined, nothing to clean - pass - -def setup_pass_tokenize(tokenizer, b_dirs, sample_count, sample_size): - global __tokenizer, __b_dirs, __sample_count, __sample_size - __tokenizer = tokenizer - __b_dirs = b_dirs - __sample_count = sample_count - __sample_size = sample_size - -def pass_tokenize(chunk_items): - """ - Chunk files into a doc->term mapping, - and simultaneously build a term->df count. - The term->df counts are redistributed to - buckets via python's in-built hash function. - This is basically an inversion step, so that - now we are chunked on the term axis rather - than the document axis. - """ - global __maxorder, __b_dirs, __extractor, __sample_count, __sample_size - __procname = mp.current_process().name - b_freq_lang = [tempfile.mkstemp(prefix=__procname+'-', suffix='.lang', dir=p)[0] for p in __b_dirs] - b_freq_domain = [tempfile.mkstemp(prefix=__procname+'-', suffix='.domain', dir=p)[0] for p in __b_dirs] - - extractor = __tokenizer - term_lng_freq = defaultdict(lambda: defaultdict(int)) - term_dom_freq = defaultdict(lambda: defaultdict(int)) - - for domain_id, lang_id, path in chunk_items: - with open(path) as f: - if __sample_count: - # sampling tokenization - text = f.read() - poss = max(1,len(text) - __sample_size) # possibe start locations - count = min(poss, __sample_count) # reduce number of samples if document is too short - offsets = random.sample(range(poss), count) - for offset in offsets: - tokenset = set(extractor(text[offset: offset+__sample_size])) - for token in tokenset: - term_lng_freq[token][lang_id] += 1 - term_dom_freq[token][domain_id] += 1 - - else: - # whole-document tokenization - tokenset = set(extractor(f.read())) - for token in tokenset: - term_lng_freq[token][lang_id] += 1 - term_dom_freq[token][domain_id] += 1 - - for term in term_lng_freq: - bucket_index = hash(term) % len(b_freq_lang) - for lang, count in term_lng_freq[term].iteritems(): - os.write(b_freq_lang[bucket_index], marshal.dumps((term, lang, count))) - for domain, count in term_dom_freq[term].iteritems(): - os.write(b_freq_domain[bucket_index], marshal.dumps((term, domain, count))) - - # Close all the open files - for f in b_freq_lang + b_freq_domain: - os.close(f) - - return len(term_lng_freq) - -def build_index(items, tokenizer, outdir, buckets=NUM_BUCKETS, jobs=None, chunksize=CHUNKSIZE, sample_count=None, sample_size=None): - """ - @param items a list of (domain, language, path) tuples - """ - global b_dirs, complete - - # Our exitfunc uses this to know whether to delete the tokenized files - complete = False - - if jobs is None: - jobs = mp.cpu_count() + 4 - - b_dirs = [ tempfile.mkdtemp(prefix="tokenize-",suffix='-{0}'.format(tokenizer.__class__.__name__), dir=outdir) for i in range(buckets) ] - - # PASS 1: Tokenize documents into sets of terms - - # If there are few items, make the chunk size such that each job - # will have 2 chunks - chunk_size = max(1,min(len(items) / (jobs * 2), chunksize)) - item_chunks = list(chunk(items, chunk_size)) - pass_tokenize_globals = (tokenizer, b_dirs, sample_count, sample_size) - - with MapPool(jobs, setup_pass_tokenize, pass_tokenize_globals) as f: - pass_tokenize_out = f(pass_tokenize, item_chunks) - - - doc_count = defaultdict(int) - chunk_count = len(item_chunks) - print("chunk size: {0} ({1} chunks)".format(chunk_size, chunk_count)) - print("job count: {0}".format(jobs)) - - if sample_count: - print("sampling-based tokenization: size {0} count {1}".format(sample_size, sample_count)) - else: - print("whole-document tokenization") - - for i, keycount in enumerate(pass_tokenize_out): - print("tokenized chunk (%d/%d) [%d keys]" % (i+1,chunk_count, keycount)) - - complete = True - - return b_dirs - -if __name__ == "__main__": - parser = argparse.ArgumentParser() - parser.add_argument("-j","--jobs", type=int, metavar='N', help="spawn N processes (set to 1 for no paralleization)") - parser.add_argument("-s", "--scanner", metavar='SCANNER', help="use SCANNER for tokenizing") - parser.add_argument("--buckets", type=int, metavar='N', help="distribute features into N buckets", default=NUM_BUCKETS) - parser.add_argument("--max_order", type=int, help="highest n-gram order to use") - parser.add_argument("--word", action='store_true', default=False, help="use 'word' tokenization (currently str.split)") - parser.add_argument("--chunksize", type=int, help="max chunk size (number of files to tokenize at a time - smaller should reduce memory use)", default=CHUNKSIZE) - parser.add_argument("-t", "--temp", metavar='TEMP_DIR', help="store buckets in TEMP_DIR instead of in MODEL_DIR/buckets") - parser.add_argument("model", metavar='MODEL_DIR', help="read index and produce output in MODEL_DIR") - - group = parser.add_argument_group('sampling') - group.add_argument("--sample_size", type=int, help="size of sample for sampling-based tokenization", default=140) - group.add_argument("--sample_count", type=int, help="number of samples for sampling-based tokenization", default=None) - - args = parser.parse_args() - - if args.temp: - buckets_dir = args.temp - else: - buckets_dir = os.path.join(args.model, 'buckets') - makedir(buckets_dir) - - bucketlist_path = os.path.join(args.model, 'bucketlist') - index_path = os.path.join(args.model, 'paths') - - # display paths - print("index path:", index_path) - print("bucketlist path:", bucketlist_path) - print("buckets path:", buckets_dir) - - with open(index_path) as f: - reader = csv.reader(f) - items = list(reader) - - if sum(map(bool,(args.scanner, args.max_order, args.word))) > 1: - parser.error('can only specify one of --word, --scanner and --max_order') - - # Tokenize - print("will tokenize %d files" % len(items)) - if args.scanner: - from .scanner import Scanner - tokenizer = Scanner.from_file(args.scanner) - print("using provided scanner: ", args.scanner) - elif args.word: - tokenizer = str.split - print("using str.split to tokenize") - else: - max_order = args.max_order if args.max_order else MAX_NGRAM_ORDER - tokenizer = NGramTokenizer(1,max_order) - print("using n-gram tokenizer: max_order({0})".format(max_order)) - b_dirs = build_index(items, tokenizer, buckets_dir, args.buckets, args.jobs, args.chunksize, args.sample_count, args.sample_size) - - # output the paths to the buckets - with open(bucketlist_path,'w') as f: - for d in b_dirs: - f.write(d+'\n') diff --git a/py3langid/train/topup.py b/py3langid/train/topup.py new file mode 100644 index 00000000..983aa16e --- /dev/null +++ b/py3langid/train/topup.py @@ -0,0 +1,175 @@ +"""Top-up thin classes from GlotCC / Glot500 / UDHR (see TRAINING.md).""" + +import itertools +import re +import shutil +from collections import defaultdict +from concurrent.futures import ThreadPoolExecutor +from functools import partial +from pathlib import Path + +from .common import ( + ALT_CLASS, + CLASS_SCRIPT, + MIN_DOC, + MIN_DOMAINS, + SPLIT_SCRIPT, + script_filter, + walk_corpus, +) +from .gather_data import valid_docs, write_docs +from .sources import ISO3 + +GLOTCC_REPO = "cis-lmu/GlotCC-V1" +GLOT500_REPO = "cis-lmu/Glot500" +GLOTSPARSE_REPO = "cis-lmu/GlotSparse" +UDHR_CSV = Path("raw_downloads/udhr/udhr-lid.csv") +TOPUP_MIN_DOCS = 600 +GLOT_SCRIPT = {**CLASS_SCRIPT, "crh": "Latn", "gom": "Deva"} +GLOT_ISO3 = {"uz": "uzn", "sdh": "sdh"} + +_CONFIG_RE = re.compile(r"^[a-z]{3}([-_])[A-Z][a-z]{3}$") + + +def _class_iso3(cls): + base = ALT_CLASS.get(cls, cls) + return GLOT_ISO3.get(base) or ISO3.get(base) + + +def _grouped(rows, target=2000): + """Pack small rows into ~target-byte docs; large rows pass through.""" + buf, size = [], 0 + for row in rows: + raw = row.encode("utf-8") if isinstance(row, str) else row + if len(raw) >= MIN_DOC: + yield raw + else: + buf.append(raw) + size += len(raw) + 1 + if size >= target: + yield b"\n".join(buf) + buf, size = [], 0 + if size >= MIN_DOC: + yield b"\n".join(buf) + + +def _write_topup(out_dir, rows, max_docs, cls, extra_dir=None): + keep = script_filter(cls) + docs = (d for d in valid_docs(_grouped(rows)) if keep is None or keep(d)) + docs = itertools.islice(docs, max_docs * 5) + if out_dir.is_dir() and any(out_dir.glob("*.txt")): + n = sum(1 for _ in itertools.islice(docs, max_docs)) # skip past primary + else: + n = write_docs(out_dir, docs, max_docs) + if extra_dir is not None: + # extra_dir is the resume marker: write to a tmp dir and rename on + # success so a mid-stream failure never marks the class as done + tmp = extra_dir.with_name(extra_dir.name + ".tmp") + if tmp.is_dir(): + shutil.rmtree(tmp) + tmp.mkdir(parents=True) + for i, d in enumerate(docs): + (tmp / f"doc{i:04d}.txt").write_bytes(d) + tmp.rename(extra_dir) + return n + + +def _repo_configs(repo): + from huggingface_hub import list_repo_files + configs = defaultdict(set) + for f in list_repo_files(repo, repo_type="dataset"): + for comp in f.split("/"): + if _CONFIG_RE.match(comp): + configs[comp[:3]].add(comp) + break + return configs + + +def _glot_config(configs, cls): + cands = sorted(configs.get(_class_iso3(cls) or "", ())) + if len(cands) > 1: + want = GLOT_SCRIPT.get(cls) + cands = [c for c in cands if want and c.endswith(want)] + return cands[0] if len(cands) == 1 else None + + +def _stream_hf(repo, config, field): + from datasets import load_dataset + ds = load_dataset(repo, config, split="train", streaming=True) + return (row[field] for row in ds) + + +def _topup_class(out_root, extra_root, repo, source, field, configs, max_docs, cls): + out_dir = out_root / source / cls + extra_dir = extra_root / source / cls + if extra_dir.is_dir(): + return # resume: fetched with extras kept + config = _glot_config(configs, cls) + if not config: + print(f"{source}/{cls}: no config") + return + try: + n = _write_topup(out_dir, _stream_hf(repo, config, field), + max_docs, cls, extra_dir) + extra = sum(1 for _ in extra_dir.glob("*.txt")) + print(f"{source}/{cls}: {n} docs (+{extra} extra) ({config})") + except Exception as e: + print(f"{source}/{cls}: SKIP ({e})") + + +def class_counts(out_root): + counts = defaultdict(lambda: defaultdict(int)) + for domain, cls, _path in walk_corpus(out_root): + counts[cls][domain] += 1 + return counts + + +def gather_udhr(out_root, classes, max_docs): + """UDHR sentences (legal register) for classes the CSV covers.""" + if not UDHR_CSV.exists(): + print(f"udhr: {UDHR_CSV} missing, skipped") + return + import csv + rows = defaultdict(list) + with open(UDHR_CSV, encoding="utf-8") as f: + for row in csv.DictReader(f): + rows[row["iso639-3"]].append(row["sentence"]) + for cls in classes: + sents = rows.get(_class_iso3(cls) or "", ()) + if sents: + n = _write_topup(out_root / "udhr" / cls, sents, max_docs, cls) + if n: + print(f"udhr/{cls}: {n} docs") + + +def gather_topup(out_root, langs, max_docs, jobs=4): + classes = [] + for lang in langs: + classes.append(lang) + if lang in SPLIT_SCRIPT: + classes.append(SPLIT_SCRIPT[lang].alt) + + def needy(): + """Classes below TOPUP_MIN_DOCS or MIN_DOMAINS.""" + counts = class_counts(out_root) + return [c for c in classes + if sum(counts[c].values()) < TOPUP_MIN_DOCS + or len(counts[c]) < MIN_DOMAINS] + + extra_root = out_root.with_name(out_root.name + "_extra") + thin = needy() + print(f"topup: {len(thin)} thin classes: {thin}") + for repo, source, field in ((GLOTCC_REPO, "glotcc", "content"), + (GLOT500_REPO, "glot500", "text"), + (GLOTSPARSE_REPO, "glotsparse", "Content")): + configs = _repo_configs(repo) + one = partial(_topup_class, out_root, extra_root, repo, source, field, + configs, max_docs) + gathered = ([p.name for p in (out_root / source).iterdir() if p.is_dir()] + if (out_root / source).is_dir() else []) + with ThreadPoolExecutor(jobs) as pool: + list(pool.map(one, sorted(set(thin) | set(gathered)))) + thin = needy() # this repo just wrote docs + if thin: + gather_udhr(out_root, thin, max_docs) + print(f"topup done; still thin: {thin}") diff --git a/py3langid/train/train.py b/py3langid/train/train.py index a4fb75fc..f62d755e 100644 --- a/py3langid/train/train.py +++ b/py3langid/train/train.py @@ -1,306 +1,123 @@ -""" -train.py - -All-in-one tool for easy training of a model for langid.py. This depends on the -training tools for individual steps, which can be run separately. - -Marco Lui, January 2013 - -Copyright 2013 Marco Lui . All rights reserved. - -Redistribution and use in source and binary forms, with or without modification, are -permitted provided that the following conditions are met: - - 1. Redistributions of source code must retain the above copyright notice, this list of - conditions and the following disclaimer. - - 2. Redistributions in binary form must reproduce the above copyright notice, this list - of conditions and the following disclaimer in the documentation and/or other materials - provided with the distribution. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDER ``AS IS'' AND ANY EXPRESS OR IMPLIED -WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND -FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR -CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR -CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR -SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON -ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING -NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF -ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - -The views and conclusions contained in the software and documentation are those of the -authors and should not be interpreted as representing official policies, either expressed -or implied, of the copyright holder. -""" - -TRAIN_PROP = 1.0 # probability than any given document is selected -MIN_DOMAIN = 1 # minimum number of domains a language must be present in to be included -MAX_NGRAM_ORDER = 4 # largest order of n-grams to consider -TOP_DOC_FREQ = 15000 # number of tokens to consider for each order -NUM_BUCKETS = 64 # number of buckets to use in k-v pair generation -CHUNKSIZE = 50 # maximum size of chunk (number of files tokenized - less = less memory use) -FEATURES_PER_LANG = 300 # number of features to select for each language +"""Train a langid model from a prepared corpus.""" import argparse -import base64 -import bz2 -import csv +import multiprocessing as mp import os -import pickle -import shutil - -import numpy - -from .common import makedir, write_weights, write_features, read_features -from .index import CorpusIndexer -from .tokenize import build_index, NGramTokenizer -from .DFfeatureselect import tally, ngram_select -from .IGweight import compute_IG -from .LDfeatureselect import select_LD_features -from .scanner import build_scanner, Scanner - -from .NBtrain import generate_cm, learn_pc, learn_ptc - - -if __name__ == "__main__": +from collections import Counter + +import numpy as np + +from ..modelio import save_model +from .common import ( + CLUSTER_K, + CLUSTERS, + FEATURES_PER_LANG, + LABEL_ALIAS, + MIN_DOMAINS, +) +from .scanner import build_scanner +from .shards import build_shards, count_matrices, merge_docfreq +from .stages import ( + cluster_features, + compute_IG, + feature_counts, + index_corpus, + ld_weights, + ngram_select, + select_LD_features, +) + + +def _axis(names, values): + """Returns (count array, name→column index) in names order.""" + counts = Counter(values) + return (np.array([counts[name] for name in names]), + {name: i for i, name in enumerate(names)}) + + +def main(argv=None): parser = argparse.ArgumentParser() - parser.add_argument("-p","--proportion", type=float, help="proportion of training data to use", default=TRAIN_PROP) parser.add_argument("-m","--model", help="save output to MODEL_DIR", metavar="MODEL_DIR") - parser.add_argument("-j","--jobs", type=int, metavar='N', help="spawn N processes (set to 1 for no paralleization)") - parser.add_argument("-t", "--temp", metavar='TEMP_DIR', help="store buckets in TEMP_DIR instead of in MODEL_DIR/buckets") - parser.add_argument("-d","--domain", metavar="DOMAIN", action='append', - help="use DOMAIN - can be specified multiple times (uses all domains found if not specified)") - parser.add_argument("-l","--lang", metavar="LANG", action='append', - help="use LANG - can be specified multiple times (uses all langs found if not specified)") - parser.add_argument("--min_domain", type=int, help="minimum number of domains a language must be present in", default=MIN_DOMAIN) - parser.add_argument("--buckets", type=int, metavar='N', help="distribute features into N buckets", default=NUM_BUCKETS) - parser.add_argument("--max_order", type=int, help="highest n-gram order to use", default=MAX_NGRAM_ORDER) - parser.add_argument("--chunksize", type=int, help="max chunk size (number of files to tokenize at a time - smaller should reduce memory use)", default=CHUNKSIZE) - parser.add_argument("--df_tokens", type=int, help="number of tokens to consider for each n-gram order", default=TOP_DOC_FREQ) - parser.add_argument("--word", action='store_true', default=False, help="use 'word' tokenization (currently str.split)") - parser.add_argument("--df_feats", metavar="FEATS", help="Instead of DF feature selection, use a list of features from FEATS") - parser.add_argument("--ld_feats", metavar="FEATS", help="Instead of LD feature selection, use a list of features from FEATS") + parser.add_argument("-j","--jobs", type=int, metavar='N', help="spawn N processes (set to 1 for no parallelization)") parser.add_argument("--feats_per_lang", type=int, metavar='N', help="select top N features for each language", default=FEATURES_PER_LANG) - parser.add_argument("--no_domain_ig", action="store_true", default=False, help="use only per-langugage IG in LD calculation") - parser.add_argument("--debug", action="store_true", default=False, help="produce debug output (all intermediates)") - - group = parser.add_argument_group('sampling') - group.add_argument("--sample_size", type=int, help="size of sample for sampling-based tokenization", default=140) - group.add_argument("--sample_count", type=int, help="number of samples for sampling-based tokenization", default=None) - + parser.add_argument("--shards", metavar="SHARD_DIR", help="n-gram count shard cache (default: CORPUS_DIR.shards)") parser.add_argument("corpus", help="read corpus from CORPUS_DIR", metavar="CORPUS_DIR") - args = parser.parse_args() + args = parser.parse_args(argv) - if args.df_feats and args.ld_feats: - parser.error("--df_feats and --ld_feats are mutually exclusive") + if args.jobs is None: + args.jobs = min(10, mp.cpu_count()) - corpus_name = os.path.basename(args.corpus) - if args.model: - model_dir = args.model - else: - model_dir = os.path.join('.', corpus_name+'.model') + model_dir = args.model or os.path.join('.', os.path.basename(args.corpus) + '.model') - makedir(model_dir) + os.makedirs(model_dir, exist_ok=True) - langs_path = os.path.join(model_dir, 'lang_index') - domains_path = os.path.join(model_dir, 'domain_index') - index_path = os.path.join(model_dir, 'paths') - - # display paths print("corpus path:", args.corpus) print("model path:", model_dir) - indexer = CorpusIndexer(args.corpus, min_domain=args.min_domain, proportion=args.proportion, - langs = args.lang, domains = args.domain) - - # Compute mappings between files, languages and domains - lang_dist = indexer.dist_lang - lang_index = indexer.lang_index - lang_info = ' '.join(("{0}({1})".format(k, lang_dist[v]) for k,v in lang_index.items())) - print("langs({0}): {1}".format(len(lang_dist), lang_info)) - - domain_dist = indexer.dist_domain - domain_index = indexer.domain_index - domain_info = ' '.join(("{0}({1})".format(k, domain_dist[v]) for k,v in domain_index.items())) - print("domains({0}): {1}".format(len(domain_dist), domain_info)) - - print("identified {0} files".format(len(indexer.items))) - - items = [ (d,l,p) for (d,l,n,p) in indexer.items ] - if args.debug: - # output the language index - with open(langs_path,'w') as f: - writer = csv.writer(f) - writer.writerows((l, lang_dist[lang_index[l]]) - for l in sorted(lang_index, key=lang_index.get)) - - # output the domain index - with open(domains_path,'w') as f: - writer = csv.writer(f) - writer.writerows((d, domain_dist[domain_index[d]]) - for d in sorted(domain_index, key=domain_index.get)) - - # output items found - with open(index_path,'w') as f: - writer = csv.writer(f) - writer.writerows(items) - - if args.temp: - buckets_dir = args.temp - else: - buckets_dir = os.path.join(model_dir, 'buckets') - makedir(buckets_dir) - - bucketlist_path = os.path.join(model_dir, 'bucketlist') - index_path = os.path.join(model_dir, 'paths') + items, langs, domains = index_corpus(args.corpus) + lang_dist, lang_index = _axis(langs, (lang for _, lang, _ in items)) + domain_dist, domain_index = _axis(domains, (d for d, _, _ in items)) + + def _summary(names, dist): + return f"({len(names)}): " + ' '.join( + f"{n}({c})" for n, c in zip(names, dist)) + + print("langs" + _summary(langs, lang_dist)) + print("domains" + _summary(domains, domain_dist)) + print(f"identified {len(items)} files") + + shard_dir = args.shards or os.path.normpath(args.corpus) + '.shards' + shard_items = build_shards(items, shard_dir, args.jobs) + + doc_count = merge_docfreq(shard_items, args.jobs) + print(f"tallied document frequency of {len(doc_count)} terms") + + features = ngram_select(doc_count) + del doc_count + print(f"selected {len(features)} DF features") + + cm_lang, cm_domain, domcount = count_matrices( + shard_items, features, lang_index, domain_index, args.jobs) + + nonempty = cm_lang.any(0) + empty = sorted(lang for lang, j in lang_index.items() if not nonempty[j]) + if empty: + parser.error(f"no n-grams for {len(empty)} class(es): {empty} " + "-- check for empty or unreadable docs") + + print("computing information gain") + domain_ig = compute_IG(cm_domain, domain_dist) + shards_per_lang = Counter(lang for _, lang, _ in shard_items) + need = np.array([min(MIN_DOMAINS, shards_per_lang[lang]) for lang in langs]) + present = domcount >= need[None, :] + LDidx = select_LD_features(ld_weights(cm_lang, lang_dist, domain_ig), + args.feats_per_lang, present) + extra = cluster_features(cm_lang, lang_dist, lang_index, features, LDidx, + CLUSTERS, CLUSTER_K) + print(f"added {len(extra)} cluster features") + LDidx |= extra + LDfeats = sorted(features[i] for i in LDidx) + print(f'selected {len(LDfeats)} features') + + tk_nextmove, tk_row, tk_output = build_scanner(LDfeats) + emitting = sum(f >= 0 for f in tk_output) + print(f"scanner: {len(tk_output)} states, {emitting} emitting, " + f"{len(tk_nextmove) // 256} distinct transition rows") + + nb_classes = [LABEL_ALIAS.get(lang, lang) for lang in langs] + nb_pc = np.log(lang_dist) + + print("counting longest-match feature occurrences") + prod = feature_counts(items, tk_nextmove, tk_row, tk_output, len(LDfeats), + lang_index, args.jobs) + nb_ptc = np.log(1.0 + prod) - np.log(len(LDfeats) + prod.sum(0)) # add-one smoothed + + model = nb_ptc, nb_pc, nb_classes, tk_nextmove, tk_row, tk_output + npz_path = os.path.join(model_dir, 'model.npz.xz') + save_model(npz_path, model) + print(f"wrote model to {npz_path} ({os.path.getsize(npz_path)} bytes)") - if args.ld_feats: - # LD features are pre-specified. We are basically just building the NB model. - LDfeats = read_features(args.ld_feats) - else: - # LD features not pre-specified, so we compute them. - - # Tokenize - DFfeats = None - print("will tokenize %d files" % len(items)) - # TODO: Custom tokenizer if doing custom first-pass features - if args.df_feats: - print("reading custom features from:", args.df_feats) - DFfeats = read_features(args.df_feats) - print("building tokenizer for custom list of {0} features".format(len(DFfeats))) - tk = Scanner(DFfeats) - elif args.word: - print("using word tokenizer") - tk = str.split - else: - print("using byte NGram tokenizer, max_order: {0}".format(args.max_order)) - tk = NGramTokenizer(1, args.max_order) - - # First-pass tokenization, used to determine DF of features - b_dirs = build_index(items, tk, buckets_dir, args.buckets, args.jobs, args.chunksize, args.sample_count, args.sample_size) - - if args.debug: - # output the paths to the buckets - with open(bucketlist_path,'w') as f: - for d in b_dirs: - f.write(d+'\n') - - # We need to compute a tally if we are selecting features by DF, but also if - # we want full debug output. - if DFfeats is None or args.debug: - # Compute DF per-feature - doc_count = tally(b_dirs, args.jobs) - if args.debug: - doc_count_path = os.path.join(model_dir, 'DF_all') - write_weights(doc_count, doc_count_path) - print("wrote DF counts for all features to:", doc_count_path) - - if DFfeats is None: - # Choose the first-stage features - DFfeats = ngram_select(doc_count, args.max_order, args.df_tokens) - doc_count = None - - if args.debug: - feature_path = os.path.join(model_dir, 'DFfeats') - write_features(DFfeats, feature_path) - print('wrote features to "%s"' % feature_path ) - - # Dispose of the first-pass tokenize output as it is no longer - # needed. - if not args.debug: - for b in b_dirs: - shutil.rmtree(b) - - # Second-pass tokenization to only obtain counts for the selected features. - # As the first-pass set is typically much larger than the second pass, it often - # works out to be faster to retokenize the raw documents rather than iterate - # over the first-pass counts. - DF_scanner = Scanner(DFfeats) - b_dirs = build_index(items, DF_scanner, buckets_dir, args.buckets, args.jobs, args.chunksize) - DF_scanner = None - - # Build vectors of domain and language distributions for use in IG calculation - domain_dist_vec = numpy.array([ domain_dist[domain_index[d]] - for d in sorted(domain_index, key=domain_index.get)], dtype=int) - domain_dist = None - lang_dist_vec = numpy.array([ lang_dist[lang_index[l]] - for l in sorted(lang_index.keys(), key=lang_index.get)], dtype=int) - lang_dist = None - - # Compute IG - ig_params = [ - ('lang', lang_dist_vec, '.lang', True), - ] - if not args.no_domain_ig: - ig_params.append( ('domain', domain_dist_vec, '.domain', False) ) - - ig_vals = {} - for label, dist, suffix, binarize in ig_params: - print("Computing information gain for {0}".format(label)) - ig = compute_IG(b_dirs, DFfeats, dist, binarize, suffix, args.jobs) - if args.debug: - weights_path = os.path.join(model_dir, 'IGweights' + suffix + ('.bin' if binarize else '')) - write_weights(ig, weights_path) - ig_vals[label] = dict((row[0], numpy.array(row[1].flat)) for row in ig) - - ig = None - DFfeats = None - # Select features according to the LD criteria - features_per_lang = select_LD_features(ig_vals['lang'], ig_vals.get('domain'), args.feats_per_lang, ignore_domain = args.no_domain_ig) - ig_vals = None - LDfeats = reduce(set.union, map(set, features_per_lang.values())) - print('selected %d features' % len(LDfeats)) - - if args.debug: - feature_path = os.path.join(model_dir, 'LDfeats') - write_features(sorted(LDfeats), feature_path) - print('wrote LD features to "%s"' % feature_path ) - - with open(feature_path + '.perlang', 'w') as f: - writer = csv.writer(f) - for i in range(len(features_per_lang)): - writer.writerow(map(repr,features_per_lang[i])) - - print('wrote LD.perlang features to "%s"' % feature_path + '.perlang') - features_per_lang = None - - # Compile a scanner for the LDfeats - tk_nextmove, tk_output = build_scanner(LDfeats) - if args.debug: - scanner_path = feature_path + '.scanner' - with open(scanner_path, 'w') as f: - pickle.dump((tk_nextmove, tk_output, LDfeats), f) - - print("wrote scanner to {0}".format(scanner_path)) - - LDfeats = None - - # Assemble the NB model - langs = sorted(lang_index, key=lang_index.get) - lang_index = None - - cm = generate_cm([ (l,p) for d,l,p in items], len(langs)) - paths = zip(*items)[2] - - nb_classes = langs - nb_pc = learn_pc(cm) - nb_ptc = learn_ptc(paths, tk_nextmove, tk_output, cm, buckets_dir, args) - - # output the model - output_path = os.path.join(model_dir, 'model') - model = nb_ptc, nb_pc, nb_classes, tk_nextmove, tk_output - string = base64.b64encode(bz2.compress(pickle.dumps(model))) - with open(output_path, 'w') as f: - f.write(string) - - print("wrote model to %s (%d bytes)" % (output_path, len(string))) - - # remove buckets if debug is off. We don't generate buckets if ldfeats is supplied. - if not args.debug and not args.ld_feats: - for b in b_dirs: - shutil.rmtree(b) - if not args.temp: - # Do not remove the buckets dir if temp was supplied as we don't know - # if we created it. - shutil.rmtree(buckets_dir) +if __name__ == "__main__": + main() diff --git a/py3langid/train/verify.py b/py3langid/train/verify.py new file mode 100644 index 00000000..94670854 --- /dev/null +++ b/py3langid/train/verify.py @@ -0,0 +1,127 @@ +"""Corpus verifier: drop docs predicted as a different non-confusable language.""" + +import argparse +from collections import Counter +from itertools import combinations +from pathlib import Path + +from ..langid import LanguageIdentifier +from .common import ( + DOC_CAP, + LABEL_ALIAS, + MIN_DOC, + MapPool, + drop, + read_doc, + walk_corpus, +) + +CONFUSABLE_GROUPS = [ + {"bs", "hr", "sr"}, {"sr", "mk"}, + {"no", "nn", "da"}, + {"ms", "id", "ace"}, {"ace", "tl"}, {"bcl", "tl"}, + {"xh", "zu", "sn", "st", "nso"}, {"lg", "sw"}, {"lg", "sn"}, + {"kik", "sn"}, {"kik", "sw"}, {"kik", "rw"}, + {"hi", "mr", "sa"}, {"hi", "ne", "sa"}, + {"gom", "mr"}, {"gom", "hi"}, {"gom", "ne"}, + {"tt", "ba", "kk"}, {"tt", "ba", "ky"}, + {"uz", "tk", "az"}, {"uz", "tk", "tr"}, + {"crh", "tr"}, {"crh", "az"}, {"crh", "tt"}, + {"ar", "arz", "ary"}, {"fa", "ps"}, {"fa", "ar"}, + {"uzs", "fa"}, {"uzs", "ps"}, {"uzs", "ur"}, {"uzs", "ug"}, + {"zh", "yue", "wuu"}, + {"it", "lij", "vec"}, {"gcf", "gcr", "ht"}, {"gcf", "fr"}, + {"ext", "an"}, {"ext", "es"}, {"ext", "pt"}, + {"gd", "ga"}, {"fy", "nl"}, {"fy", "af"}, {"ltg", "lv"}, + {"grc", "el"}, {"hbo", "he"}, + {"pcm", "en"}, {"fuv", "ha"}, {"fuv", "om"}, +] +CONFUSABLE = {frozenset(p) for g in CONFUSABLE_GROUPS + for p in combinations(sorted(g), 2)} +MIN_PARA = 150 + +_ident = None + + +def _init(model_path): + global _ident + _ident = LanguageIdentifier.from_modelpath(model_path) + + +def is_foreign(label, pred): + label = LABEL_ALIAS.get(label, label) + pred = LABEL_ALIAS.get(pred, pred) + return pred != label and frozenset((label, pred)) not in CONFUSABLE + + +def _check_doc(arg): + lang, path = arg + pred, _ = _ident.classify(read_doc(path, DOC_CAP)) + return path if is_foreign(lang, pred) else None + + +def _filter_paragraphs(arg): + lang, path = arg + data = Path(path).read_bytes() + kept, stripped = [], 0 + for para in data.split(b"\n"): + if len(para) >= MIN_PARA and is_foreign(lang, _ident.classify(para)[0]): + stripped += len(para) + continue + kept.append(para) + if not stripped: + return lang, path, 0, False + out = b"\n".join(kept) + if len(out) < MIN_DOC: + return lang, path, stripped, True + Path(path).write_bytes(out) + return lang, path, stripped, False + + +def corpus_items(corpus, verifier_langs=None): + items = [] + skipped_langs = set() + for _domain, lang, path in walk_corpus(corpus, skip_langs=("zxx",)): + label = LABEL_ALIAS.get(lang, lang) + if verifier_langs is not None and label not in verifier_langs: + skipped_langs.add(lang) + continue + items.append((lang, path)) + if skipped_langs: + print(f"verify: skipped {len(skipped_langs)} unknown lang(s): {sorted(skipped_langs)}") + return items + + +def main(argv=None): + parser = argparse.ArgumentParser() + parser.add_argument("--model", required=True, help="verifier model file (npz.xz)") + parser.add_argument("--paragraphs", action="store_true", + help="filter foreign paragraphs instead of whole docs") + parser.add_argument("-j", "--jobs", type=int, default=8, help="parallel workers (default: 8)") + parser.add_argument("corpus", metavar="CORPUS_DIR") + args = parser.parse_args(argv) + + verifier_langs = set(LanguageIdentifier.from_modelpath(args.model).nb_classes) + items = corpus_items(args.corpus, verifier_langs) + if args.paragraphs: + stripped = Counter() + junk = [] + with MapPool(args.jobs, _init, (args.model,), chunksize=200) as f: + for lang, path, n, dead in f(_filter_paragraphs, items): + if n: + stripped[lang] += n + if dead: + junk.append(path) + drop(args.corpus, junk) + print("stripped bytes by lang:", stripped.most_common(15)) + print(f"docs dropped entirely: {len(junk)}") + else: + with MapPool(args.jobs, _init, (args.model,), chunksize=500) as f: + drops = [p for p in f(_check_doc, items) if p] + drop(args.corpus, drops) + by_lang = Counter(Path(p).parent.name for p in drops) + print(f"dropped {len(drops)} of {len(items)} docs:", by_lang.most_common(15)) + + +if __name__ == "__main__": + main() diff --git a/py3langid/train/zxx.py b/py3langid/train/zxx.py new file mode 100644 index 00000000..3343bcb1 --- /dev/null +++ b/py3langid/train/zxx.py @@ -0,0 +1,59 @@ +"""Synthetic not-a-language (zxx) training docs. Deterministic per-domain seeds.""" +import base64 +import json +import random +import string +import sys +from pathlib import Path + +from .common import DOC_CAP + +DOCS_PER_DOMAIN = 300 +DOMAIN_SEEDS = {"wiki": 1, "cc100": 2} + +def _doc(rng): + genre = rng.randrange(10) + lines = [] + for _ in range(rng.randint(30, 60)): + n = rng.randint(20, 70) + if genre == 0: + lines.append(" ".join(str(rng.randint(0, 10**rng.randint(1, 9))) for _ in range(8))) + elif genre == 1: + lines.append("".join(rng.choice("!@#$%^&*()_+-=[]{};:,.<>/?|~`\"'\\") for _ in range(n))) + elif genre == 2: + lines.append("".join(rng.choice(string.ascii_letters + " ") for _ in range(n))) + elif genre == 3: + lines.append("".join(chr(rng.choice([rng.randint(0x2200, 0x23FF), rng.randint(0x2500, 0x27BF), rng.randint(0x1F300, 0x1F5FF)])) for _ in range(n // 2))) + elif genre == 4: + lines.append(base64.b64encode(rng.randbytes(n)).decode()) + elif genre == 5: + lines.append(rng.randbytes(n // 2).hex()) + elif genre == 6: + lines.append(" ".join(f"https://ex{rng.randint(1,999)}.com/{rng.randbytes(4).hex()}?id={rng.randint(1,9999)}" for _ in range(3))) + elif genre == 7: + lines.append("".join(f"" for _ in range(6))) + elif genre == 8: + lines.append(json.dumps({f"k{rng.randint(1,99)}": rng.randint(0, 9999) for _ in range(5)})) + else: + tok = "".join(rng.choice(string.ascii_lowercase) for _ in range(rng.randint(2, 6))) + lines.append(" ".join([tok] * rng.randint(5, 15))) + return "\n".join(lines).encode()[:DOC_CAP] + + +def ensure_zxx(corpus): + """Write zxx dirs if absent. Returns docs written.""" + written = 0 + for domain, seed in DOMAIN_SEEDS.items(): + out = Path(corpus) / domain / "zxx" + if out.is_dir() and len(list(out.glob("*.txt"))) >= DOCS_PER_DOMAIN: + continue + out.mkdir(parents=True, exist_ok=True) + rng = random.Random(seed) + for i in range(DOCS_PER_DOMAIN): + (out / f"doc{i:04d}.txt").write_bytes(_doc(rng)) + written += 1 + return written + + +if __name__ == "__main__": + print("zxx docs written:", ensure_zxx(sys.argv[1])) diff --git a/pyproject.toml b/pyproject.toml index 882fc6ed..203a4fd8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ # https://pip.pypa.io/en/stable/reference/build-system/pyproject-toml/ [build-system] -requires = ["setuptools>=61.0"] +requires = ["setuptools>=77.0"] build-backend = "setuptools.build_meta" [project] @@ -52,7 +52,7 @@ packages = ["py3langid"] version = {attr = "py3langid.__version__"} [tool.setuptools.package-data] -py3langid = ["data/model.plzma"] +py3langid = ["data/model.npz.xz"] [project.scripts] langid = "py3langid.langid:main" diff --git a/tests/__init__.py b/tests/__init__.py index 200fcec9..e69de29b 100644 --- a/tests/__init__.py +++ b/tests/__init__.py @@ -1,2 +0,0 @@ - -import pytest diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 00000000..ad012d55 --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,10 @@ +import pytest + + +@pytest.fixture +def tokenize_order2(monkeypatch): + """Tokenize byte order 2 only, so a shard payload is small enough to + assert on exactly. Moves the tokenizer and the cache key together, but + NOT selection: SELECT_ORDERS is derived at import, so tests needing a + different selection pass ngram_select's `orders`.""" + monkeypatch.setattr("py3langid.train.shards.MAX_NGRAM_ORDER", 2) diff --git a/tests/test_gather_data.py b/tests/test_gather_data.py new file mode 100644 index 00000000..833f0dd9 --- /dev/null +++ b/tests/test_gather_data.py @@ -0,0 +1,135 @@ +"""Offline unit tests for gather_data (network downloaders are tested by use).""" +from pathlib import Path + +import pytest + +from py3langid.train.common import DOC_CAP, MIN_DOC +from py3langid.train.gather_data import ( + CC100_CODE, + ISO3, + WIKI_CODE, + write_docs, +) + + +def test_write_docs(tmp_path): + docs = [ + b"x" * (MIN_DOC - 1), # stub, skipped + b"a" * (DOC_CAP + 5000), # truncated + b"b" * 600, + b"c" * 600, + ] + n = write_docs(tmp_path / "out", iter(docs), max_docs=2) + assert n == 2 + files = sorted((tmp_path / "out").iterdir()) + assert [f.name for f in files] == ["doc0000.txt", "doc0001.txt"] + assert files[0].stat().st_size == DOC_CAP + assert files[1].read_bytes() == b"b" * 600 + + +def test_write_docs_all_stubs(tmp_path): + assert write_docs(tmp_path / "out", [b"tiny"], max_docs=5) == 0 + assert not (tmp_path / "out").exists() + + +def test_tatoeba_uses_the_validity_gate(tmp_path, monkeypatch): + """tatoeba routes docs through valid_doc like every other source""" + import io + import tarfile + + from py3langid.train import gather_data + + # two "languages": eng packs 1 long sentence per doc, deu 1 stub per doc + rows = [("1", "eng", "L" * (DOC_CAP + 500))] * 3 + [("2", "deu", "tiny")] * 3 + csv = "".join("\t".join(r) + "\n" for r in rows).encode() + buf = io.BytesIO() + with tarfile.open(fileobj=buf, mode="w:bz2") as tar: + info = tarfile.TarInfo("sentences.csv") + info.size = len(csv) + tar.addfile(info, io.BytesIO(csv)) + monkeypatch.setattr(gather_data, "fetch_cached", + lambda *a, **k: io.BytesIO(buf.getvalue())) + monkeypatch.setattr(gather_data, "ISO3", {"en": "eng", "de": "deu"}) + + counts = gather_data.gather_tatoeba(tmp_path, ["en", "de"], max_docs=2, + per_doc=1) + assert counts == {"en": 2} # de produced only stubs -> nothing written + for f in (tmp_path / "tatoeba" / "en").iterdir(): + assert f.stat().st_size == DOC_CAP # truncated, not written raw + assert not (tmp_path / "tatoeba" / "de").exists() + + +@pytest.mark.parametrize("mapping", [ISO3, CC100_CODE, WIKI_CODE]) +def test_mapping_keys_are_iso_639(mapping): + """keys are ISO 639-1, or 639-3 for langs without a 639-1 code""" + assert all(2 <= len(k) <= 3 for k in mapping) + + +def test_iso3_values_are_distinct_639_3(): + assert all(len(v) == 3 for v in ISO3.values()) + assert len(set(ISO3.values())) == len(ISO3) + + +def test_fetch_cached(tmp_path, monkeypatch): + import io + + from py3langid.train import gather_data + + calls = [] + + def fake_fetch(url, headers=None, retries=3): + calls.append(url) + return io.BytesIO(b"payload") + + monkeypatch.setattr(gather_data, "fetch", fake_fetch) + path = tmp_path / "cache" / "f.bin" + for _ in range(2): + with gather_data.fetch_cached("http://x", path) as resp: + assert resp.read() == b"payload" + assert len(calls) == 1 # second call served from disk + assert not path.with_name(path.name + ".tmp").exists() + + +def test_teereader(tmp_path): + import io + + from py3langid.train.gather_data import _TeeReader + + cache = tmp_path / "sub" / "prefix.bz2" + tee = _TeeReader(io.BytesIO(b"abcdef"), cache) + assert tee.read(4) == b"abcd" + tee.finalize() + assert cache.read_bytes() == b"abcd" + assert not tee.tmp.exists() + + tee = _TeeReader(io.BytesIO(b"xyz"), tmp_path / "sub" / "other") + tee.read() + tee.discard() + assert not tee.tmp.exists() and not tee.path.exists() + + +def test_dedup(tmp_path): + from py3langid.train.dedup import dedup + + line = b"x" * 80 + d1 = tmp_path / "wiki" / "aa" + d2 = tmp_path / "cc100" / "aa" + other = tmp_path / "wiki" / "bb" + zxx = tmp_path / "wiki" / "zxx" + for d in (d1, d2, other, zxx): + d.mkdir(parents=True) + (d1 / "doc0000.txt").write_bytes(line + b"\nshort\n" + b"y" * 70) + (d2 / "doc0000.txt").write_bytes(line + b"\nunique here " + b"z" * 60) + (other / "doc0000.txt").write_bytes(line) # same line, different lang: kept + (zxx / "doc0000.txt").write_bytes(line + b"\n" + line) # zxx untouched + + removed, touched, dropped = dedup(tmp_path) + # d1 falls under MIN_DOC once the duplicate line goes, so it is dropped + assert (removed, touched, dropped) == (1, 0, 1) + # sorted traversal: cc100 before wiki, so d2 keeps the first occurrence + assert (d2 / "doc0000.txt").read_bytes().startswith(line) + assert not (d1 / "doc0000.txt").exists() + assert (Path(str(tmp_path) + "_dropped") / "wiki" / "aa" / "doc0000.txt").exists() + assert (other / "doc0000.txt").read_bytes() == line + assert (zxx / "doc0000.txt").read_bytes().count(line) == 2 + assert dedup(tmp_path) == (0, 0, 0) # idempotent diff --git a/tests/test_langid.py b/tests/test_langid.py index d471200f..d1dedadb 100644 --- a/tests/test_langid.py +++ b/tests/test_langid.py @@ -1,23 +1,49 @@ import csv +import json +import lzma +import math +import shutil import subprocess import sys +from collections import Counter from pathlib import Path +import numpy as np import pytest import py3langid as langid -from py3langid.langid import MODEL_FILE, LanguageIdentifier +from py3langid.langid import ( + MODEL_DIR, + MODEL_FILE, + RAW_FLOOR, + LanguageIdentifier, + _load_identifier, +) + + +# a model load costs ~0.25s: share one per variant, undoing set_languages, +# the only mutable state, between tests +@pytest.fixture(scope='module') +def _shared_identifier(): + return LanguageIdentifier.from_model_file(MODEL_FILE) + + +@pytest.fixture(scope='module') +def _shared_norm_identifier(): + return LanguageIdentifier.from_model_file(MODEL_FILE, norm_probs=True) @pytest.fixture -def identifier(): - return LanguageIdentifier.from_pickled_model(MODEL_FILE) +def identifier(_shared_identifier): + yield _shared_identifier + _shared_identifier.set_languages(None) @pytest.fixture -def norm_identifier(): - return LanguageIdentifier.from_pickled_model(MODEL_FILE, norm_probs=True) +def norm_identifier(_shared_norm_identifier): + yield _shared_norm_identifier + _shared_norm_identifier.set_languages(None) @pytest.mark.parametrize('text,expected', [ @@ -35,6 +61,83 @@ def test_norm_probs(norm_identifier): assert 0 <= prob <= 1 +def test_calibration_sqrt(norm_identifier): + """sqrt-of-bytes temperature: confidence tracks ambiguity, not saturated""" + _, hi = norm_identifier.classify('This is clearly an English sentence with plenty of text.') + _, lo = norm_identifier.classify('ovo je') # short, bs/hr/sr-ambiguous + assert lo < hi <= 1.0 + assert lo < 0.9 + + +def test_featureless_input(identifier, norm_identifier): + """no features -> a flat floor and zero confidence, in both modes""" + raw = identifier.classify('hi') + # finite (stays JSON-serializable) but below any real log-probability, + # so a caller thresholding raw scores never mistakes it for a hit + assert raw[1] == RAW_FLOOR + assert math.isfinite(raw[1]) + assert raw[1] < identifier.classify('This is an English sentence.')[1] + # the flat score makes every class column equally likely under norm_probs; + # merging leaves an aliased label (sr, uz) with two columns' worth + label, prob = norm_identifier.classify('hi') + per_col = 1 / len(identifier.nb_classes) + probs = [p for _, p in norm_identifier.rank('hi')] + assert min(probs) == pytest.approx(per_col, rel=1e-6) + assert max(probs) == pytest.approx(2 * per_col, rel=1e-6) + assert sum(probs) == pytest.approx(1.0, rel=1e-6) + assert prob == pytest.approx(max(probs), rel=1e-6) + assert label in ('sr', 'uz') + + +def test_unique_labels(identifier): + """LABEL_ALIAS gives sr/uz two columns each; the public view has one""" + assert len(identifier.nb_classes) > len(identifier.labels) + assert len(identifier.labels) == len(set(identifier.labels)) + for dup in ('sr', 'uz'): + assert identifier.nb_classes.count(dup) == 2 + assert identifier.labels.count(dup) == 1 + # a restricted set collapses too, however many columns it kept + identifier.set_languages(['sr']) + assert identifier.labels == ['sr'] + assert identifier.rank('ovo je tekst za probu') == [('sr', pytest.approx( + identifier.classify('ovo je tekst za probu')[1]))] + + +def test_rank_takes_max_over_aliased_columns(identifier): + """each label appears once in rank(), scored by its best column""" + text = 'ovo je tekst za probu, ne znam sto to znaci' + scores = identifier._decide(text) + ranked = identifier.rank(text) + + assert len(ranked) == len(identifier.labels) + assert {lang for lang, _ in ranked} == set(identifier.labels) + for lang, score in ranked: + cols = [i for i, c in enumerate(identifier.nb_classes) if c == lang] + assert score == pytest.approx(max(float(scores[i]) for i in cols)) + # sr/uz really do exercise the multi-column path + assert any(identifier.nb_classes.count(c) == 2 for c in identifier.labels) + + +def test_rank_agrees_with_classify(identifier): + """rank()[0] is classify()""" + texts = ['ne znam sto to znaci', 'ovo je test', 'dobar dan', 'kaj', + 'Test Unicode sur du texte en français', 'hi', 'a'] + for text in texts: + lang, conf = identifier.classify(text) + assert identifier.rank(text)[0] == (lang, pytest.approx(conf)), text + + +def test_language_restriction(identifier): + """a restriction narrows the class set, still classifies, and reverts""" + full = len(identifier.nb_classes) + identifier.set_languages(['en', 'de']) + assert set(identifier.labels) == {'en', 'de'} + assert len(identifier.nb_classes) == 2 + assert identifier.classify('This should be enough text.')[0] == 'en' + identifier.set_languages(None) + assert len(identifier.nb_classes) == full + + def test_unnormalized(identifier): _, prob = identifier.classify('Test Unicode sur du texte en français') assert prob < 0 @@ -62,20 +165,49 @@ def test_bytes_str_parity(): assert langid.rank(text) == langid.rank(text.encode('utf8')) +def test_truncated_bytes_reach_the_str_branch(): + '''bytes cut mid-codepoint still get lowered and NFC-normalized''' + full = 'ЭТО РУССКИЙ ТЕКСТ ДЛЯ ТЕСТА ОПРЕДЕЛЕНИЯ ЯЗЫКА'.encode() + cut = full[:-1] # drops one byte of a 2-byte codepoint + with pytest.raises(UnicodeDecodeError): + cut.decode('utf8') + # the partial tail is dropped, so all-caps lowering still applies + assert langid.classify(cut)[0] == 'ru' + assert LanguageIdentifier._encode(cut) == full[:-2].decode().lower().encode() + # genuinely undecodable bytes are still passed through untouched + assert LanguageIdentifier._encode(b'\xff\xfe\xff\xfe') == b'\xff\xfe\xff\xfe' + + def test_empty_and_short(): - '''Feature-less input scores -inf, short input does not crash''' - for empty in ('', b'', '12345'): + '''Feature-less input scores a finite floor, short input does not crash''' + for empty in ('', b''): lang, score = langid.classify(empty) assert isinstance(lang, str) - assert score == float('-inf') + assert score == RAW_FLOOR + # finite, so the server's JSON stays valid for strict parsers + json.dumps({'confidence': score}, allow_nan=False) + # digit-only input has features since the fpl700 budget: routed to zxx + assert langid.classify('12345')[0] == 'zxx' lang, score = langid.classify('a') assert isinstance(lang, str) def test_norm_probs_empty(norm_identifier): - '''norm_probs=True on empty input returns uniform distribution''' - _, prob = norm_identifier.classify('') - assert abs(prob - 1.0 / len(norm_identifier.nb_classes)) < 1e-6 + '''norm_probs=True on empty input returns a flat per-column distribution''' + probs = [p for _, p in norm_identifier.rank('')] + per_col = 1.0 / len(norm_identifier.nb_classes) + assert abs(min(probs) - per_col) < 1e-6 + assert abs(max(probs) - 2 * per_col) < 1e-6 # aliased labels merge two columns + assert abs(sum(probs) - 1.0) < 1e-6 + + +def test_classify_matches_rank(norm_identifier): + """aliased columns merge the same way in both APIs (regression)""" + for text in ('Prema Jungovoj teoriji, m', 'Serbia II Регионална лига', + 'Toshkent shahri markazida', 'This is an English sentence.'): + lang, conf = norm_identifier.classify(text) + top_lang, top_conf = norm_identifier.rank(text)[0] + assert (lang, conf) == (top_lang, pytest.approx(top_conf, rel=1e-6)) def test_rank_sorted(identifier): @@ -85,7 +217,9 @@ def test_rank_sorted(identifier): scores = [s for _, s in ranking] assert all(isinstance(s, float) for s in scores) assert scores == sorted(scores, reverse=True) - assert len(ranking) == len(identifier.nb_classes) + # one entry per output label: aliased columns (sr/uz) share one + assert len(ranking) == len(identifier.labels) + assert len(ranking) == len({lang for lang, _ in ranking}) def test_set_languages_error(identifier): @@ -94,24 +228,15 @@ def test_set_languages_error(identifier): identifier.set_languages(['xx_invalid']) -def test_set_languages_reset(identifier): - '''set_languages(None) restores the full model''' - full = len(identifier.nb_classes) - identifier.set_languages(['en', 'fr']) - assert len(identifier.nb_classes) == 2 - identifier.set_languages(None) - assert len(identifier.nb_classes) == full - - def test_redirection(): '''Test if STDIN redirection works''' thisdir = Path(__file__).parent - langid_path = str(thisdir.parent / 'py3langid' / 'langid.py') readme_path = str(thisdir.parent / 'README.rst') with open(readme_path, 'rb') as f: readme = f.read() - result = subprocess.check_output([sys.executable, langid_path, '-n'], input=readme) - assert b'en' in result and b'1.0' in result + result = subprocess.check_output([sys.executable, '-m', 'py3langid.langid', '-n'], + input=readme, cwd=thisdir.parent) + assert b'en' in result and 0.5 < float(result.split()[-1].rstrip(b')')) <= 1.0 def test_cli_batch(tmp_path): @@ -127,24 +252,73 @@ def test_cli_batch(tmp_path): def test_cli_external_model(tmp_path): - '''-m loads a model in the modelstring format (b64 + bz2 pickle)''' - import bz2 - import lzma - from base64 import b64encode - - from py3langid.langid import MODEL_DIR - with lzma.open(MODEL_DIR / MODEL_FILE) as f: - raw = f.read() - model_path = tmp_path / 'external.model' - model_path.write_bytes(b64encode(bz2.compress(raw, compresslevel=1))) + '''-m loads a model from a path outside the package''' + model_path = tmp_path / 'external.npz.xz' + shutil.copy(MODEL_DIR / MODEL_FILE, model_path) result = subprocess.check_output(['langid', '-n', '-m', str(model_path)], input=b'This should be enough text.') - assert b'en' in result and b'1.0' in result + assert b'en' in result and 0.5 < float(result.split()[-1].rstrip(b')')) <= 1.0 + # the path is honored, not silently replaced by the bundled model + missing = subprocess.run(['langid', '-n', '-m', str(tmp_path / 'nope.npz.xz')], + input=b'This should be enough text.', + capture_output=True, check=False) + assert missing.returncode != 0 def test_cli(): '''Test console scripts entry point''' result = subprocess.check_output(['langid', '-n'], input=b'This should be enough text.') - assert b'en' in result and b'1.0' in result + assert b'en' in result and 0.5 < float(result.split()[-1].rstrip(b')')) <= 1.0 result = subprocess.check_output(['langid', '-n', '-l', 'bg,en,uk'], input=b'This should be enough text.') - assert b'en' in result and b'1.0' in result + assert b'en' in result and 0.5 < float(result.split()[-1].rstrip(b')')) <= 1.0 + + +def _variant(ident, **kwargs): + """another identifier over the same arrays, no second model load""" + return LanguageIdentifier(ident.nb_ptc, ident.nb_pc, ident.nb_classes, + ident.tk_nextmove, ident.tk_output, + tk_row=ident.tk_row, **kwargs) + + +def test_min_confidence(identifier): + """abstention: low calibrated confidence returns 'und'""" + ident = _variant(identifier, norm_probs=True, min_confidence=0.5) + lang, conf = ident.classify('This should be enough text.') + assert lang == 'en' and conf >= 0.5 + lang, conf = ident.classify('Hi') # too short to attribute + assert lang == 'und' and conf < 0.5 + with pytest.raises(ValueError): + _variant(identifier, min_confidence=0.5) + + +def test_from_modelpath(): + """from_modelpath loads the npz+LZMA layout from an arbitrary path""" + ident = LanguageIdentifier.from_modelpath(MODEL_DIR / MODEL_FILE) + assert ident.classify('This should be enough text.')[0] == 'en' + + +def test_external_model_failure_raises(tmp_path): + """an unusable -m path raises instead of silently falling back""" + bad = tmp_path / 'not-a-model.npz.xz' + bad.write_bytes(b'definitely not an xz stream') + with pytest.raises((OSError, lzma.LZMAError, ValueError)): + _load_identifier(str(bad)) + + +def test_score_log1p(identifier): + """scoring applies sublinear TF (log1p) plus class priors""" + text = b'This should be enough text.' + state, idxs = 0, [] + for letter in text: + state = identifier.tk_nextmove[(identifier.tk_row[state] << 8) + letter] + feat = identifier.tk_output[state] # one longest match per position + if feat >= 0: + idxs.append(feat) + fc = Counter(idxs) + idx = np.fromiter(fc.keys(), dtype=np.intp, count=len(fc)) + counts = np.fromiter(fc.values(), dtype=np.float32, count=len(fc)) + expected = np.log1p(counts) @ np.asarray(identifier.nb_ptc, dtype=np.float32)[idx] \ + + identifier.nb_pc + # _raw_score is per column, before aliased columns are merged + assert np.allclose(identifier._raw_score(identifier._encode(text)), expected, + rtol=1e-4) diff --git a/tests/test_modelio.py b/tests/test_modelio.py new file mode 100644 index 00000000..48dece55 --- /dev/null +++ b/tests/test_modelio.py @@ -0,0 +1,115 @@ +import io +import lzma +import tempfile +from array import array + +import numpy as np +import pytest + +from py3langid.modelio import expand_nextmove, load_model, save_model + + +def _model(rows, row_index, output, classes=("en", "fr"), ptc_rows=1): + """save_model's tuple, with filler for the NB arrays""" + return (np.zeros((ptc_rows, len(classes)), dtype=np.float32), + np.full(len(classes), 0.5, dtype=np.float32), list(classes), + rows, row_index, output) + + +def test_roundtrip(tmp_path): + ptc = np.arange(12, dtype=np.float32).reshape(4, 3) + pc = np.array([0.1, 0.2, 0.3], dtype=np.float32) + classes = ["en", "fr", "zh"] + rows = array("H", range(512)) + row_index = array("L", [0, 1]) + output = [3, -1] # one longest-match feature per state, -1 = none + + path = tmp_path / "model.npz.xz" + save_model(path, (ptc, pc, classes, rows, row_index, output)) + ptc2, pc2, classes2, rows2, row2, output2 = load_model(path) + + assert np.array_equal(ptc2, ptc) and np.array_equal(pc2, pc) + assert classes2 == classes + assert rows2 == rows and list(row2) == [0, 1] + assert output2 == [3, -1] + + # an array of the same values produces the same file + save_model(path.with_suffix(".b"), + (ptc, pc, classes, rows, row_index, array("l", output))) + assert path.with_suffix(".b").read_bytes() == path.read_bytes() + + +def test_empty_tk_output(tmp_path): + '''model with no emitting states survives the roundtrip''' + rows = array("H", range(256)) + path = tmp_path / "model.npz.xz" + save_model(path, _model(rows, array("L", [0]), [-1], ptc_rows=0)) + _ptc2, _pc2, classes2, rows2, _row2, output2 = load_model(path) + assert output2 == [-1] and classes2 == ["en", "fr"] and rows2 == rows + + +def test_uint32_widening(tmp_path): + '''a DFA beyond the uint16 state ceiling round-trips via uint32''' + rows = array("L", [1 << 16] * 256) # state id overflows uint16 + save_model(tmp_path / "m.npz.xz", _model(rows, array("L", [0]), [0])) + _, _, _, loaded, _, _ = load_model(tmp_path / "m.npz.xz") + assert loaded.itemsize == 4 + assert list(loaded) == list(rows) + + +def test_rows_canonicalized(tmp_path): + """rows are stored sorted and distinct, with the index remapped onto them""" + # rows given in descending order, the second one used by two states + rows = array("H", [2] * 256 + [1] * 256) + path = tmp_path / "m.npz.xz" + save_model(path, _model(rows, array("L", [0, 1, 1]), [0, -1, -1])) + _, _, _, rows2, row_index, output = load_model(path) + assert list(rows2) == [1] * 256 + [2] * 256 + assert list(row_index) == [1, 0, 0] + # a duplicate row passed in anyway is still stored once + save_model(path, _model(array("H", [1] * 512), array("L", [0, 1]), [0, -1])) + _, _, _, rows3, row_index3, _ = load_model(path) + assert len(rows3) == 256 and list(row_index3) == [0, 0] + assert output == [0, -1, -1] + + +def test_unsupported_legacy_layout_rejected(tmp_path): + """a pre-row-dedup / pre-longest-match model is refused by name, not + with a bare KeyError""" + arrays = { + "ptc": np.zeros((4, 2), dtype=np.float32), + "pc": np.array([0.5, 0.5], dtype=np.float32), + "classes": np.array(["en", "fr"]), + "nextmove": np.array([1] * 256 + [2] * 256 + [0] * 256, dtype=np.uint16), + "out_offsets": np.array([0, 2, 2, 3], dtype=np.uint32), + "out_flat": np.array([1, 3, 0], dtype=np.uint32), + } + buf = io.BytesIO() + np.savez(buf, **arrays) + path = tmp_path / "legacy.npz.xz" + path.write_bytes(lzma.compress(buf.getvalue(), preset=1)) + with pytest.raises(ValueError, match="nextmove_row.*out_feat"): + load_model(path) + + +def test_expand_nextmove(): + """the inverse of the row sharing (benchmarks/fast_eval.py's flat walk)""" + rows = array("H", [7] * 256 + [9] * 256) + flat = expand_nextmove(rows, array("L", [1, 0, 1])) + assert flat.typecode == "H" + assert list(flat) == [9] * 256 + [7] * 256 + [9] * 256 + # numpy input keeps a matching typecode + assert expand_nextmove(np.array(rows, dtype=np.uint32), + np.array([0, 1])).typecode == "I" + + +def test_load_leaves_no_temp_file(tmp_path, monkeypatch): + """the loader cleans up its scratch file (dropping the name of one it still + holds open is a PermissionError on Windows)""" + scratch = tmp_path / "scratch" + scratch.mkdir() + path = tmp_path / "m.npz.xz" + save_model(path, _model(array("H", range(256)), array("L", [0]), [0])) + monkeypatch.setattr(tempfile, "tempdir", str(scratch)) + load_model(path) + assert list(scratch.iterdir()) == [] diff --git a/tests/test_server.py b/tests/test_server.py index 6e20c9a6..f74d8e81 100644 --- a/tests/test_server.py +++ b/tests/test_server.py @@ -4,7 +4,7 @@ import pytest -from py3langid.langid import application +from py3langid.server import application def _request(path, method='GET', body=None, query=None, content_length='auto'): diff --git a/tests/test_train_options.py b/tests/test_train_options.py new file mode 100644 index 00000000..a463bcfd --- /dev/null +++ b/tests/test_train_options.py @@ -0,0 +1,63 @@ +"""Unit tests for ngram_select's order set and the DOC_CAP tokenization cap. + +The n-gram orders, the DF pool size and the doc cap are constants, not +flags: tests that need a small term set patch MAX_NGRAM_ORDER or pass +`orders` explicitly, and tests that need a different cap patch DOC_CAP. +""" +import marshal + +import pytest + +from py3langid.train.shards import _build_shard, _group_key +from py3langid.train.stages import ngram_select + + +def test_ngram_select_orders(): + """`orders` alone decides which lengths are admissible""" + doc_count = {b"a": 9, b"b": 8, b"ab": 7, b"abc": 6} + assert ngram_select(doc_count, 10, orders={1, 2, 3}) == \ + [b"a", b"ab", b"abc", b"b"] + assert ngram_select(doc_count, 10, orders={2, 3}) == [b"ab", b"abc"] + assert ngram_select(doc_count, 10, orders={2}) == [b"ab"] + assert ngram_select(doc_count, 10, orders=set()) == [] + + +def test_ngram_select_default_orders(): + """the shipped order set: byte orders 2..5 plus the CJK bigram order""" + from py3langid.train.common import ( + MAX_NGRAM_ORDER, + MIN_NGRAM_ORDER, + SELECT_ORDERS, + TOKENIZE_ORDER, + ) + assert SELECT_ORDERS == set(range(MIN_NGRAM_ORDER, MAX_NGRAM_ORDER + 1)) \ + | {TOKENIZE_ORDER} + # a single byte is never selectable, so a 1-gram cannot become a feature + assert ngram_select({b"a": 9, b"ab": 1}) == [b"ab"] + + +@pytest.mark.parametrize("doc_cap,expected", [ + (3, [b"ab", b"bc"]), # cap 3 sees only b"abc": b"cd" on is never tokenized + (0, [b"ab", b"bc", b"cd", b"de", b"ef"]), # 0 = no cap +]) +def test_doc_cap_truncates(tmp_path, monkeypatch, tokenize_order2, doc_cap, + expected): + """DOC_CAP bounds how much of a document reaches the tokenizer""" + monkeypatch.setattr("py3langid.train.shards.DOC_CAP", doc_cap) + doc = tmp_path / "doc0000.txt" + doc.write_bytes(b"abcdef") + shard = tmp_path / "shard" + _build_shard((str(shard), _group_key([str(doc)]), [str(doc)])) + with open(shard, "rb") as f: + marshal.load(f) # the cache key + docfreq = marshal.load(f) + # the bigrams of b"abcdef" are distinct: one doc each + assert docfreq == dict.fromkeys(expected, 1) + + +def test_doc_cap_is_part_of_the_cache_key(tmp_path, monkeypatch): + """editing the cap must invalidate cached shards""" + monkeypatch.setattr("py3langid.train.shards.DOC_CAP", 3) + key3 = _group_key([str(tmp_path)]) + monkeypatch.setattr("py3langid.train.shards.DOC_CAP", 0) + assert _group_key([str(tmp_path)]) != key3 diff --git a/tests/test_training_pipeline.py b/tests/test_training_pipeline.py new file mode 100644 index 00000000..49ae4bc3 --- /dev/null +++ b/tests/test_training_pipeline.py @@ -0,0 +1,90 @@ +"""Smoke test for the training pipeline (end-to-end with synthetic data).""" +import pytest + + +@pytest.fixture +def corpus_dir(tmp_path): + """Create a tiny corpus: 3 langs × 1 domain × 2 docs.""" + texts = { + "en": [ + b"The quick brown fox jumps over the lazy dog and the cat sleeps", + b"London is the capital of England and a very large city indeed", + ], + "de": [ + b"Der schnelle braune Fuchs springt ueber den faulen Hund heute", + b"Berlin ist die Hauptstadt von Deutschland und eine grosse Stadt", + ], + "fr": [ + b"Le renard brun rapide saute par dessus le chien paresseux ici", + b"Paris est la capitale de la France et une tres grande ville bien", + ], + } + for lang, docs in texts.items(): + lang_dir = tmp_path / "corpus" / "web" / lang + lang_dir.mkdir(parents=True) + for i, doc in enumerate(docs): + (lang_dir / f"doc{i}.txt").write_bytes(doc) + return tmp_path / "corpus" + + +def test_training_pipeline(corpus_dir, tmp_path): + from py3langid.modelio import load_model, save_model + from py3langid.train.train import main + + model_dir = tmp_path / "model" + + common_args = [ + "-j", "1", + "--feats_per_lang", "20", + str(corpus_dir), + ] + main(["-m", str(model_dir)] + common_args) + + model_path = model_dir / "model.npz.xz" + assert model_path.exists() and model_path.stat().st_size > 0 + + # The shard cache was created next to the corpus + shard_dir = corpus_dir.parent / (corpus_dir.name + ".shards") + assert list(shard_dir.iterdir()) + + # Load with the runtime + from py3langid.langid import LanguageIdentifier + + lid = LanguageIdentifier.from_modelpath(str(model_path)) + lang, _ = lid.classify("This is a test") + assert isinstance(lang, str) + + # save_model and load_model are inverses: re-saving what was loaded + # reproduces the file byte for byte + resaved = tmp_path / "model_resaved.npz.xz" + save_model(resaved, load_model(model_path)) + assert resaved.read_bytes() == model_path.read_bytes() + lang2, _ = LanguageIdentifier.from_modelpath(str(resaved)).classify("This is a test") + assert lang2 == lang + + # Determinism: a rerun (served from cached shards) produces the same model + rerun_dir = tmp_path / "model_rerun" + main(["-m", str(rerun_dir)] + common_args) + assert (rerun_dir / "model.npz.xz").read_bytes() == model_path.read_bytes() + + +def test_relabel(corpus_dir, tmp_path): + """srl dirs fold into the sr label at model assembly.""" + from py3langid.modelio import load_model + from py3langid.train.train import main + + sr_doc = ("ово је српски текст за пробу овде").encode() + srl_doc = b"ovo je srpski tekst za probu ovde i jos malo teksta dodato" + for lang, doc in (("sr", sr_doc), ("srl", srl_doc)): + d = corpus_dir / "web" / lang + d.mkdir() + (d / "doc0.txt").write_bytes(doc) + (d / "doc1.txt").write_bytes(doc + b" jos") + + model_dir = tmp_path / "model" + main(["-m", str(model_dir), "-j", "1", "--feats_per_lang", "20", + str(corpus_dir)]) + classes = load_model(model_dir / "model.npz.xz")[2] + + assert "srl" not in classes + assert classes.count("sr") == 2 diff --git a/tests/test_training_units.py b/tests/test_training_units.py new file mode 100644 index 00000000..1b74355c --- /dev/null +++ b/tests/test_training_units.py @@ -0,0 +1,414 @@ +"""Unit tests for the numeric core of the training pipeline.""" +import math +import os +from pathlib import Path + +import numpy as np +import pytest + +from py3langid.train.common import ( + MAX_NGRAM_ORDER, + TOKENIZE_ORDER, + chunks, + is_cjk_bigram, + job_chunks, +) +from py3langid.train.shards import ( + COUNT_DTYPE, + build_shards, + count_matrices, + doc_ngrams, + load_shard, + merge_docfreq, +) +from py3langid.train.stages import ( + compute_IG, + entropy, + ld_weights, + ngram_select, + select_LD_features, +) + + +def make_corpus(tmp_path, docs): + """Write tmp_path/corpus///docN.txt from (domain, lang, data) + triples. @returns (build_shards items, shard dir)""" + items = [] + for domain, lang, data in docs: + lang_dir = tmp_path / "corpus" / domain / lang + lang_dir.mkdir(parents=True, exist_ok=True) + seen = sum(1 for item in items if item[:2] == (domain, lang)) + path = lang_dir / f"doc{seen}.txt" + path.write_bytes(data) + items.append((domain, lang, str(path))) + return items, str(tmp_path / "shards") + + +def test_entropy(): + assert entropy([1, 1]) == np.log(2) + assert entropy([2, 0]) == 0.0 + assert entropy([1, 1, 1, 1]) == np.log(4) + + +def test_doc_ngrams(): + # orders below MIN_NGRAM_ORDER are not emitted: ngram_select filters on + # length, so single bytes could never be selected as features + assert doc_ngrams(b"abab", 2) == {b"ab", b"ba"} + assert doc_ngrams(b"abab", 3) == {b"ab", b"ba", b"aba", b"bab"} + assert doc_ngrams(b"", 2) == set() + assert doc_ngrams(b"a", 2) == set() + + +def test_doc_ngrams_cjk_only_at_tokenize_order(): + """below TOKENIZE_ORDER, that order yields CJK codepoint bigrams only""" + cjk = "\u4e2d\u6587".encode() # two 3-byte CJK codepoints + latin = b"abcdef" + assert cjk in doc_ngrams(cjk, 5) + assert len(cjk) == TOKENIZE_ORDER + # no other 6-byte term survives + assert {t for t in doc_ngrams(cjk + latin, 5) if len(t) == TOKENIZE_ORDER} == {cjk} + assert not {t for t in doc_ngrams(latin, 5) if len(t) == TOKENIZE_ORDER} + # asking for the full order restores every 6-gram + assert latin in doc_ngrams(latin, TOKENIZE_ORDER) + + +def test_doc_ngrams_matches_unrestricted_selection(): + """the CJK restriction drops only unselectable terms""" + data = "\u4e2d\u6587abc\u3042\u3044 xyz\u00e9\u00e8".encode() + full = doc_ngrams(data, TOKENIZE_ORDER) + selectable = {t for t in full if len(t) < TOKENIZE_ORDER or is_cjk_bigram(t)} + assert doc_ngrams(data, TOKENIZE_ORDER - 1) == selectable + + +def test_compute_IG_nonbinarized(): + # 2 events with 2 docs each. b'aa' occurs only in event 0 (perfectly + # discriminative, IG = log 2); b'bb' occurs once per event (IG = 0). + cm = np.array([[2, 0], [1, 1]]) + dist = np.array([2, 2]) + ig = compute_IG(cm, dist) + assert math.isclose(ig[0], math.log(2)) + assert math.isclose(ig[1], 0.0, abs_tol=1e-12) + # degenerate terms score 0, not nan: absent everywhere, then present + # everywhere (an all-zero count vector has no entropy to report) + degenerate = compute_IG(np.array([[0, 0], [2, 2]]), dist) + assert list(degenerate) == [0.0, 0.0] + + +def _ld_matrix(cm, dist, domain_ig=None): + """ld_weights' columns stacked, for tests that want the whole matrix.""" + if domain_ig is None: + domain_ig = np.zeros(len(cm)) + return np.stack(list(ld_weights(cm, dist, domain_ig)), axis=1) + + +def test_ld_weights(): + # Binarized per language: same data and, by symmetry, the same exact + # values as the non-binarized IG above. + cm = np.array([[2, 0], [1, 1]]) + dist = np.array([2, 2]) + ig = _ld_matrix(cm, dist) + assert ig.shape == (2, 2) + for lang in range(2): + assert math.isclose(ig[0, lang], math.log(2)) + assert math.isclose(ig[1, lang], 0.0, abs_tol=1e-12) + assert not np.isnan(ig).any() + # degenerate terms and a single-language corpus both score 0, not nan + assert not _ld_matrix(np.array([[0, 0], [2, 2]]), dist).any() + assert not _ld_matrix(np.array([[2], [0]]), np.array([2])).any() + # one column per language, in column order, whatever the term count + big = np.array([[2, 0], [1, 1], [0, 2], [2, 2], [0, 0]]) + assert len(list(ld_weights(big, dist, np.zeros(5)))) == 2 + # the domain IG is a per-term offset subtracted from every column + domain_ig = np.array([0.25, 0.5]) + assert np.array_equal(_ld_matrix(cm, dist, domain_ig), + _ld_matrix(cm, dist) - domain_ig[:, None]) + + + +def test_ld_weights_matches_contingency_table(): + """the fused per-column formula against a brute-force 2x2 IG reference""" + def brute(cm, dist): + cm, dist = np.asarray(cm, float), np.asarray(dist, float) + n = dist.sum() + + def H(*ps): + return -sum(p * math.log(p) for p in ps if p > 0) + + out = np.zeros(cm.shape) + for i in range(cm.shape[0]): + t = cm[i].sum() + for j in range(cm.shape[1]): + a, b, c = cm[i, j], t - cm[i, j], dist[j] - cm[i, j] + d = (n - t) - c + cond = (t / n) * H(a / t, b / t) if t else 0.0 + if n - t: + cond += ((n - t) / n) * H(c / (n - t), d / (n - t)) + out[i, j] = H(dist[j] / n, (n - dist[j]) / n) - cond + return out + + rng = np.random.default_rng(0) + cases = [] + for _ in range(40): + nl, nt = int(rng.integers(1, 7)), int(rng.integers(1, 12)) + dist = rng.integers(1, 40, nl) + cases.append((np.array([[rng.integers(0, dist[j] + 1) for j in range(nl)] + for _ in range(nt)], dtype=np.int32), dist)) + # all-zero rows, a single language, an entirely empty matrix + cases += [(np.array([[0, 0], [2, 2]], dtype=np.int32), np.array([2, 2])), + (np.array([[2], [0]], dtype=np.int32), np.array([2])), + (np.zeros((3, 4), dtype=np.int32), np.array([5, 5, 5, 5]))] + for cm, dist in cases: + got = _ld_matrix(cm, dist) + assert not np.isnan(got).any() + assert np.allclose(got, brute(cm, dist), atol=1e-12) + +def test_select_LD_features(): + # LD = IG_lang - IG_domain: term 2 is penalized for being domain-informative + ld = np.array([ + [0.7, 0.1], + [0.1, 0.6], + [-0.1, -0.1], + ]) + present = np.ones(ld.shape, dtype=bool) + assert select_LD_features(ld.T, 1, present) == {0, 1} + assert select_LD_features(ld.T, 3, present) == {0, 1, 2} + # a language's picks are restricted to features present in it + only_first = np.array([[True, True], [False, True], [False, True]]) + assert select_LD_features(ld.T, 3, only_first) == {0, 1, 2} + assert select_LD_features(ld.T, 1, only_first) == {0, 1} + + +def test_ngram_select(): + doc_count = {b"a": 5, b"b": 3, b"ab": 10, b"cd": 1} + feats = ngram_select(doc_count, tokens_per_order=1, orders={1, 2}) + assert feats == [b"a", b"ab"] + + +def test_build_shards_cache(tmp_path, tokenize_order2): + items, shard_dir = make_corpus(tmp_path, [("web", "en", b"abab"), + ("web", "en", b"ab")]) + doc0 = Path(items[0][2]) + + [(domain, lang, shard_path)] = build_shards(items, shard_dir, jobs=1) + assert (domain, lang) == ("web", "en") + # shards carry document frequency only -- the NB numerators come from + # feature_counts, so total occurrence counts are never stored + docfreq = load_shard(shard_path) + assert docfreq == {b"ab": 2, b"ba": 1} + + # unchanged corpus: shard is reused, not rewritten + mtime = os.path.getmtime(shard_path) + build_shards(items, shard_dir, jobs=1) + assert os.path.getmtime(shard_path) == mtime + + # changed doc invalidates and rebuilds the shard + doc0.write_bytes(b"zzzz") + build_shards(items, shard_dir, jobs=1) + docfreq = load_shard(shard_path) + assert docfreq[b"ab"] == 1 # only doc1 still has it + assert docfreq[b"zz"] == 1 + + +def test_chunks_and_job_chunks(): + seq = list(range(10)) + assert chunks(seq, 4) == [[0, 1, 2, 3], [4, 5, 6, 7], [8, 9]] + assert chunks(seq, 0) == [[i] for i in seq] # size floors at 1 + assert chunks([], 4) == [] + assert len(job_chunks(seq, 3)) == 3 + assert [x for c in job_chunks(seq, 3) for x in c] == seq + assert job_chunks([], 3) == [] + + +def test_merge_docfreq_spans_chunks(tmp_path, monkeypatch, tokenize_order2): + """the merge reduces across several chunks, not just one""" + monkeypatch.setattr("py3langid.train.shards.MERGE_SHARDS_PER_CHUNK", 2) + items, shard_dir = make_corpus( + tmp_path, [("web", f"l{i}", b"abab") for i in range(5)]) + shard_items = build_shards(items, shard_dir, jobs=1) + assert len(chunks(shard_items, 2)) == 3 # the path under test + # every shard is b"abab": df 1 per shard, so 5 shards sum to 5 + assert merge_docfreq(shard_items, jobs=1) == {b"ab": 5, b"ba": 5} + + +def test_count_matrices(tmp_path, tokenize_order2): + """per-lang/per-domain docfreq and domain presence, in one shard pass""" + # en appears in two domains, fr in one + items, shard_dir = make_corpus(tmp_path, [("web", "en", b"abab"), + ("news", "en", b"abab"), + ("web", "fr", b"cdcd")]) + shard_items = build_shards(items, shard_dir, jobs=1) + + feats = [b"ab", b"cd"] + lang_index, domain_index = {"en": 0, "fr": 1}, {"news": 0, "web": 1} + cm_lang, cm_domain, domcount = count_matrices( + shard_items, feats, lang_index, domain_index, jobs=1) + + assert cm_lang.dtype == COUNT_DTYPE and cm_domain.dtype == COUNT_DTYPE + assert cm_lang.tolist() == [[2, 0], [0, 1]] # b"ab" in 2 en docs + assert cm_domain.tolist() == [[1, 1], [0, 1]] # b"ab" in news + web + assert domcount.tolist() == [[2, 0], [0, 1]] # b"ab" in 2 en domains + + +def test_shard_cache_keyed_on_tokenization(tmp_path, monkeypatch): + """Reuse is exact key equality: editing a tokenization constant rebuilds + instead of serving shards written under the old one. (Reuse used to be + order >=, which let features depend on the cache's history.)""" + items, shard_dir = make_corpus( + tmp_path, [("web", "zh", "中文abcdef".encode())]) + + [(_, _, shard_path)] = build_shards(items, shard_dir, jobs=1) + terms = load_shard(shard_path) + # order TOKENIZE_ORDER is CJK-only however long the byte orders run + assert {t for t in terms if len(t) == TOKENIZE_ORDER} == {"中文".encode()} + assert max(len(t) for t in terms if len(t) != TOKENIZE_ORDER) == MAX_NGRAM_ORDER + + monkeypatch.setattr("py3langid.train.shards.MAX_NGRAM_ORDER", 3) + build_shards(items, shard_dir, jobs=1) + terms = load_shard(shard_path) + # rebuilt at the new order: the 4- and 5-grams are gone, not inherited + assert max(len(t) for t in terms if len(t) != TOKENIZE_ORDER) == 3 + assert {t for t in terms if len(t) == TOKENIZE_ORDER} == {"中文".encode()} + + +def test_select_counts_intersects_either_way(): + """both branches (iterate the shard vs the feature index) must agree""" + from py3langid.train.shards import _select_counts + + feat_index = {b"ab": 0, b"cd": 1, b"ef": 2} + + def mapping(counts): + idx, vals = _select_counts(counts, feat_index) + assert idx.dtype == np.intp and vals.dtype == COUNT_DTYPE + return dict(zip(idx.tolist(), vals.tolist())) + + assert mapping({b"ab": 3, b"zz": 9}) == {0: 3} # shard smaller + assert mapping({b"ab": 3, b"zz": 9, b"cd": 1, b"qq": 2}) == {0: 3, 1: 1} + assert mapping({}) == {} # no overlap + + +def test_index_corpus_first_appearance_order(tmp_path): + """class column order is first appearance along the sorted walk, NOT + alphabetical -- it fixes nb_classes, so pin it""" + from py3langid.train.stages import index_corpus + + for domain, langs in (("aaa", ["en", "fr"]), ("bbb", ["de", "en"])): + for lang in langs: + d = tmp_path / domain / lang + d.mkdir(parents=True, exist_ok=True) + (d / "doc0.txt").write_bytes(b"some text here") + items, langs, domains = index_corpus(tmp_path) + assert langs == ["en", "fr", "de"] # "de" is absent from the first domain + assert domains == ["aaa", "bbb"] + assert len(items) == 4 + + +def test_cluster_features(): + """a cluster spends its budget on non-junk features the quota missed, + ranked by IG over the cluster's own languages""" + from py3langid.train.stages import cluster_features + + feats = [b"11", b"aa", b"bb", b"cc"] + # docfreq over langs (en, de, fr); IG within {en, de} descends 11 > aa > bb, + # and cc is uninformative there + cm_lang = np.array([[4, 0, 0], [3, 0, 0], [2, 0, 0], [2, 2, 0]]) + lang_dist = np.array([4, 4, 4]) + lang_index = {"en": 0, "de": 1, "fr": 2} + + def run(base, k, clusters=(("en", "de"),)): + return cluster_features(cm_lang, lang_dist, lang_index, feats, base, + clusters, k) + + assert run(set(), 1) == {1} # ranking: b"aa" beats b"bb" + # b"11" is digits-only and b"aa" is already selected, so the best + # *eligible* feature wins even though it ranks third by IG + assert run({1}, 1) == {2} + # no IG floor: once the eligible ranking is exhausted the budget takes + # uninformative features too + assert run({1}, 2) == {2, 3} + assert run(set(), 3) == {1, 2, 3} # only b"11" stays excluded + assert run({1, 2, 3}, 2) == set() # nothing eligible left + # a cluster naming a language absent from the corpus is skipped entirely + assert run(set(), 3, clusters=(("en", "xx"),)) == set() + + +def longest_endings(feats, data): + """Reference scanner, naive O(n*|feats|): index of the longest feature + ending at each byte position, -1 where none does.""" + res = [] + for end in range(1, len(data) + 1): + hits = [f for f in feats if data[:end].endswith(f)] + res.append(feats.index(max(hits, key=len)) if hits else -1) + return res + + +@pytest.fixture(scope="module") +def longest_match_dfa(): + from py3langid.train.scanner import build_scanner + + feats = [b"ab", b"abc", b"bc", b"c", b"xy", b"aab"] + return feats, build_scanner(feats) + + +@pytest.mark.parametrize("data", [b"", b"c", b"zzz", b"xy", b"xabcy", + b"abcabc", b"aabc"]) +def test_build_scanner_longest_match(longest_match_dfa, data): + """the DFA emits, at each byte position, the longest feature ending there""" + feats, (rows, row_index, out) = longest_match_dfa + state, got = 0, [] + for byte in data: + state = rows[(row_index[state] << 8) + byte] + got.append(out[state]) + assert got == longest_endings(feats, data) + + +def test_build_scanner_shares_every_duplicate_row(): + """row sharing is maximal: no two stored rows hold the same transitions, + which is what lets save_model canonicalize without deduplicating""" + from py3langid.train.scanner import build_scanner + + feats = [b"ab", b"abc", b"bc", b"c", b"xy", b"aab", b"bca", b"cab"] + rows, row_index, out = build_scanner(feats) + stored = len(rows) // 256 + assert stored < len(out) # sharing actually happened + + contents = {tuple(rows[(row_index[s] << 8):(row_index[s] << 8) + 256]) + for s in range(len(out))} + assert len(contents) == stored + + +def test_feature_counts(tmp_path, monkeypatch): + """NB numerators = one longest match per byte position, per language""" + from py3langid.train.scanner import build_scanner + from py3langid.train.stages import feature_counts + + feats = [b"ab", b"abc", b"c"] + rows, row_index, out = build_scanner(feats) + + docs = {"en": [b"abcabc", b"ab", b""], "fr": [b"cc", b"xabz"]} + items = [] + for lang, texts in docs.items(): + for i, text in enumerate(texts): + path = tmp_path / f"{lang}{i}.txt" + path.write_bytes(text) + items.append(("dom", lang, str(path))) + lang_index = {"en": 0, "fr": 1} + + expected = np.zeros((len(feats), 2), dtype=np.int64) + for lang, texts in docs.items(): + for text in texts: + for feat in longest_endings(feats, text): + if feat >= 0: + expected[feat, lang_index[lang]] += 1 + + got = feature_counts(items, rows, row_index, out, len(feats), lang_index, + jobs=1) + assert np.array_equal(got, expected) + # worker partials are integer sums: the parallel result is exact + assert np.array_equal( + feature_counts(items, rows, row_index, out, len(feats), lang_index, + jobs=2), expected) + # DOC_CAP truncates before counting + monkeypatch.setattr("py3langid.train.stages.DOC_CAP", 2) + capped = feature_counts(items, rows, row_index, out, len(feats), + lang_index, jobs=1) + assert capped.sum() < expected.sum() diff --git a/tests/test_verify.py b/tests/test_verify.py new file mode 100644 index 00000000..4a8bb10e --- /dev/null +++ b/tests/test_verify.py @@ -0,0 +1,39 @@ +"""Unit tests for split-script routing and the self-verify decision.""" +from py3langid.train.common import MIN_DOC, latin_majority, route_script +from py3langid.train.gather_data import write_docs +from py3langid.train.verify import is_foreign + + +def test_latin_majority(): + assert latin_majority("Republika Srbija je država".encode()) + assert not latin_majority("Република Србија је држава".encode()) + assert not latin_majority(b"12345 ...") # no letters -> not Latin-majority + + +def test_route_script(tmp_path): + sr = tmp_path / "sr" + assert route_script(sr, "Република".encode()) == sr + assert route_script(sr, b"Republika") == tmp_path / "srl" + other = tmp_path / "hr" + assert route_script(other, b"Republika") == other + + +def test_write_docs_splits_sr(tmp_path): + cyr = "Београд је главни град Србије. ".encode() * 30 + lat = b"Beograd je glavni grad Srbije. " * 30 + assert len(cyr) >= MIN_DOC and len(lat) >= MIN_DOC + n = write_docs(tmp_path / "sr", [cyr, lat, cyr, lat], max_docs=10) + assert n == 4 + assert len(list((tmp_path / "sr").iterdir())) == 2 + assert len(list((tmp_path / "srl").iterdir())) == 2 + assert (tmp_path / "srl" / "doc0000.txt").read_bytes() == lat.strip()[:3000] + + +def test_is_foreign(): + assert is_foreign("xh", "en") + assert not is_foreign("xh", "zu") # confusable pair protected + assert not is_foreign("sr", "mk") # protect-list regression guard + assert not is_foreign("srl", "hr") # alias srl->sr, sr/hr protected + assert not is_foreign("no", "nb") # legacy nb aliases to no + assert is_foreign("az", "la") + assert not is_foreign("de", "de")