Ë
    çÍ:j¬  ã                   óx   — d Z ddlmZ ddlmZ 	 ddlZ G d„ d«      Zd„ Z	e
dk(  r e	«        yy# e$ r dZY Œ$w xY w)	a˜  
A module for language identification using the TextCat algorithm.
An implementation of the text categorization algorithm
presented in Cavnar, W. B. and J. M. Trenkle,
"N-Gram-Based Text Categorization".

The algorithm takes advantage of Zipf's law and uses
n-gram frequencies to profile languages and text-yet to
be identified-then compares using a distance measure.

Language n-grams are provided by the "An Crubadan"
project. A corpus reader was created separately to read
those files.

For details regarding the algorithm, see:
https://www.let.rug.nl/~vannoord/TextCat/textcat.pdf

For details about An Crubadan, see:
https://borel.slu.edu/crubadan/index.html
é    )Úmaxsize)ÚtrigramsNc                   óD   — e Zd ZdZi ZdZdZi Zd„ Zd„ Z	d„ Z
d„ Zd„ Zd	„ Zy)
ÚTextCatNú<ú>c                 ó´   — t         st        d«      ‚ddlm} || _        | j                  j                  «       D ]  }| j                  j                  |«       Œ y )Nz›classify.textcat requires the regex module that supports unicode. Try '$ pip install regex' and see https://pypi.python.org/pypi/regex for further details.r   )Úcrubadan)ÚreÚOSErrorÚnltk.corpusr
   Ú_corpusÚlangsÚ	lang_freq)Úselfr
   Úlangs      új/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/classify/textcat.pyÚ__init__zTextCat.__init__7   sQ   € ÝÜð#óð õ 	)àˆŒà—L‘L×&Ñ&Ó(ò 	)ˆDØ�L‰L×"Ñ" 4Õ(ñ	)ó    c                 ó0   — t        j                  dd|«      S )z)Get rid of punctuation except apostrophesz[^\P{P}\']+Ú )r   Úsub©r   Útexts     r   Úremove_punctuationzTextCat.remove_punctuationG   s   € ä�v‰v�n b¨$Ó/Ð/r   c                 ó0  — ddl m}m} | j                  |«      } ||«      } |«       }|D ]c  }t	        | j
                  |z   | j                  z   «      }|D �	cg c]  }	dj                  |	«      ‘Œ }
}	|
D ]  }||v r||xx   dz  cc<   Œd||<   Œ Œe |S c c}	w )z'Create FreqDist of trigrams within textr   )ÚFreqDistÚword_tokenizer   é   )Únltkr   r   r   r   Ú_START_CHARÚ	_END_CHARÚjoin)r   r   r   r   Ú
clean_textÚtokensÚfingerprintÚtÚtoken_trigram_tuplesÚtriÚtoken_trigramsÚcur_trigrams               r   ÚprofilezTextCat.profileK   s®   € ç0à×,Ñ,¨TÓ2ˆ
Ù˜zÓ*ˆá“jˆØò 	1ˆAÜ#+¨D×,<Ñ,<¸qÑ,@À4Ç>Á>Ñ,QÓ#RÐ Ø6JÖK¨s˜bŸg™g c�lÐKˆNÐKà-ò 1�Ø +Ñ-Ø Ó,°Ñ1Ô,à/0�K Ò,ñ	1ð		1ð Ðùò Ls   ÁBc                 ó  — | j                   j                  |«      }d}||v r`t        |j                  «       «      j	                  |«      }t        |j                  «       «      j	                  |«      }t        ||z
  «      }|S t        }|S )zgCalculate the "out-of-place" measure between the
        text and language profile for a single trigramr   )r   r   ÚlistÚkeysÚindexÚabsr   )r   r   ÚtrigramÚtext_profileÚlang_fdÚdistÚidx_lang_profileÚidx_texts           r   Ú	calc_distzTextCat.calc_dist_   s„   € ð —,‘,×(Ñ(¨Ó.ˆØˆà�gÑÜ# G§L¡L£NÓ3×9Ñ9¸'ÓBÐÜ˜L×-Ñ-Ó/Ó0×6Ñ6°wÓ?ˆHô Ð'¨(Ñ2Ó3ˆDð ˆô ˆDàˆr   c                 óÆ   — i }| j                  |«      }| j                  j                  j                  «       D ]&  }d}|D ]  }|| j	                  |||«      z  }Œ |||<   Œ( |S )zOCalculate the "out-of-place" measure between
        the text and all languagesr   )r,   r   Ú_all_lang_freqr/   r8   )r   r   Ú	distancesr,   r   Ú	lang_distr2   s          r   Ú
lang_distszTextCat.lang_distst   s{   € ð ˆ	Ø—,‘,˜tÓ$ˆà—L‘L×/Ñ/×4Ñ4Ó6ò 	(ˆDð ˆIØ"ò D�Ø˜TŸ^™^¨D°'¸7ÓCÑC‘	ðDð (ˆI�dŠOð	(ð Ðr   c                 ó„   — | j                  |«      | _        t        | j                  | j                  j                  ¬«      S )zYFind the language with the min distance
        to the text and return its ISO 639-3 code)Úkey)r=   Úlast_distancesÚminÚgetr   s     r   Úguess_languagezTextCat.guess_language†   s4   € ð #Ÿo™o¨dÓ3ˆÔä�4×&Ñ&¨D×,?Ñ,?×,CÑ,CÔDÐDr   )Ú__name__Ú
__module__Ú__qualname__r   Úfingerprintsr!   r"   r@   r   r   r,   r8   r=   rC   © r   r   r   r   /   s:   „ Ø€GØ€LØ€KØ€Ià€Nò)ò 0òò(ò*ó$Er   r   c            
      óð  — ddl m}  g d¢}dddddd	d
dddœ	}t        «       }|D ]Ì  }| j                  |«      }t	        |«      dz
  }t        t        t        |«      «      }d}t        d|«      D ]<  }	ddj                  t        d||	   «      D �
cg c]
  }
||	   |
   ‘Œ c}
«      z   }||z  }Œ> t        d|dd z   dz   «       |j                  |«      }t        d|› d||   › d�«       t        d«       ŒÎ y c c}
w )Nr   )Úudhr)	zKurdish-UTF8zAbkhaz-UTF8zFarsi_Persian-UTF8z
Hindi-UTF8zHawaiian-UTF8zRussian-UTF8zVietnamese-UTF8zSerbian_Srpski-UTF8zEsperanto-UTF8zNorthern KurdishÚ	AbkhazianzIranian PersianÚHindiÚHawaiianÚRussianÚ
VietnameseÚSerbianÚ	Esperanto)	ÚkmrÚabkÚpesÚhinÚhawÚrusÚvieÚsrpÚepor   r   ú zLanguage snippet: éŒ   z...zLanguage detection: z (ú)zŒ############################################################################################################################################)r   rJ   r   ÚsentsÚlenr.   ÚmapÚranger#   ÚprintrC   )rJ   r   ÚfriendlyÚtcÚcur_langÚraw_sentencesÚrowsÚcolsÚsampleÚiÚjÚcur_sentÚguesss                r   Údemorn   �   s&  € Ý ò
€Eð "ØØ ØØØØØØñ
€Hô 
‹€Bàò ˆàŸ
™
 8Ó,ˆÜ�=Ó! AÑ%ˆÜ”Cœ˜]Ó+Ó,ˆàˆô �q˜$“ò 	ˆAØ˜SŸX™XÄEÈ!ÈTÐRSÉWÓDUÖ&V¸q }°QÑ'7¸Ó':Ò&VÓWÑWˆHØ�hÑ‰Fð	ô
 	Ð" V¨A¨c ]Ñ2°UÑ:Ô;Ø×!Ñ! &Ó)ˆÜÐ$ U G¨2¨h°u©oÐ->¸aÐ@ÔAÜˆiÕñ#ùò 'Ws   ÂC3Ú__main__)Ú__doc__Úsysr   Ú	nltk.utilr   Úregexr   ÚImportErrorr   rn   rD   rH   r   r   ú<module>ru      sZ   ðñõ* å ðÛ÷\Eñ \Eò@.ðb ˆzÒÙ…Fð øðq ò Ø	‚Bðús   �/ ¯9¸9