Ë
    çÍ:j=  ã            
       óŽ  — d Z ddlZddlmZ ddlZddlmZ dZdZdZ	dZ
eed	d
dddde	df
Zed   e
gedd ¢­Z ej                  d«      Z ej                  eej                  ej                   z  ej"                  z  «      Z ej                  d«      Z ej                  d«      Zdd„Zdd„Z G d„ de«      Zd„ Zd„ Z	 	 	 	 dd„Zy)a�  
Twitter-aware tokenizer, designed to be flexible and easy to adapt to new
domains and tasks. The basic logic is this:

1. The tuple REGEXPS defines a list of regular expression
   strings.

2. The REGEXPS strings are put, in order, into a compiled
   regular expression object called WORD_RE, under the TweetTokenizer
   class.

3. The tokenization is done by WORD_RE.findall(s), where s is the
   user-supplied string, inside the tokenize() method of the class
   TweetTokenizer.

4. When instantiating Tokenizer objects, there are several options:
    * preserve_case. By default, it is set to True. If it is set to
      False, then the tokenizer will downcase everything except for
      emoticons.
    * reduce_len. By default, it is set to False. It specifies whether
      to replace repeated character sequences of length 3 or greater
      with sequences of length 3.
    * strip_handles. By default, it is set to False. It specifies
      whether to remove Twitter handles of text used in the
      `tokenize` method.
    * match_phone_numbers. By default, it is set to True. It indicates
      whether the `tokenize` method should look for phone numbers.
é    N)ÚList)Ú
TokenizerIac  
    (?:
      [<>]?
      [:;=8]                     # eyes
      [\-o\*\']?                 # optional nose
      [\)\]\(\[dDpP/\:\}\{@\|\\] # mouth
      |
      [\)\]\(\[dDpP/\:\}\{@\|\\] # mouth
      [\-o\*\']?                 # optional nose
      [:;=8]                     # eyes
      [<>]?
      |
      </?3                       # heart
    )u  			# Capture 1: entire matched URL
  (?:
  https?:				# URL protocol and colon
    (?:
      /{1,3}				# 1-3 slashes
      |					#   or
      [a-z0-9%]				# Single letter or digit or '%'
                                       # (Trying not to match e.g. "URI::Escape")
    )
    |					#   or
                                       # looks like domain name followed by a slash:
    [a-z0-9.\-]+[.]
    (?:[a-z]{2,13})
    /
  )
  (?:					# One or more:
    [^\s()<>{}\[\]]+			# Run of non-space, non-()<>{}[]
    |					#   or
    \([^\s()]*?\([^\s()]+\)[^\s()]*?\) # balanced parens, one level deep: (...(...)...)
    |
    \([^\s]+?\)				# balanced parens, non-recursive: (...)
  )+
  (?:					# End with:
    \([^\s()]*?\([^\s()]+\)[^\s()]*?\) # balanced parens, one level deep: (...(...)...)
    |
    \([^\s]+?\)				# balanced parens, non-recursive: (...)
    |					#   or
    [^\s`!()\[\]{};:'".,<>?Â«Â»â€œâ€�â€˜â€™]	# not a space or one of these punct chars
  )
  |					# OR, the following to match naked domains:
  (?:
  	(?<!@)			        # not preceded by a @, avoid matching foo@_gmail.com_
    [a-z0-9]+
    (?:[.\-][a-z0-9]+)*
    [.]
    (?:[a-z]{2,13})
    \b
    /?
    (?!@)			        # not succeeded by a @,
                            # avoid matching "foo.na" in "foo.na@example.com"
  )
uÏ  
  (?:
    [\U0001F1E6-\U0001F1FF]{2}  # all enclosed letter pairs
    |
    # English flag
    \U0001F3F4\U000E0067\U000E0062\U000E0065\U000E006e\U000E0067\U000E007F
    |
    # Scottish flag
    \U0001F3F4\U000E0067\U000E0062\U000E0073\U000E0063\U000E0074\U000E007F
    |
    # For Wales? Why Richard, it profit a man nothing to give his soul for the whole world â€¦ but for Wales!
    \U0001F3F4\U000E0067\U000E0062\U000E0077\U000E006C\U000E0073\U000E007F
  )
a	  
    (?:
      (?:            # (international)
        \+?[01]
        [ *\-.\)]*
      )?
      (?:            # (area code)
        [\(]?
        \d{3}
        [ *\-.\)]*
      )?
      \d{3}          # exchange
      [ *\-.\)]*
      \d{4}          # base
    )z	<[^>\s]+>z[\-]+>|<[\-]+z(?:@[\w_]+)z(?:\#+[\w_]+[\w\'_\-]*[\w_]+)z#[\w.+-]+@[\w-]+\.(?:[\w-]\.?)+[\w-]uR   .(?:
        [ðŸ�»-ðŸ�¿]?(?:â€�.[ðŸ�»-ðŸ�¿]?)+
        |
        [ðŸ�»-ðŸ�¿]
    )a…  
    (?:[^\W\d_](?:[^\W\d_]|['\-_])+[^\W\d_]) # Words with apostrophes or dashes.
    |
    (?:[+\-]?\d+[,/.:-]\d+[+\-]?)  # Numbers, including fractions, decimals.
    |
    (?:[\w_]+)                     # Words without apostrophes or dashes.
    |
    (?:\.(?:\s*\.){1,})            # Ellipsis dots.
    |
    (?:\S)                         # Everything else that isn't whitespace.
    é   z([^a-zA-Z0-9])\1{3,}z&(#?(x?))([^&;\s]+);zZ(?<![A-Za-z0-9_!@#\$%&*])@(([A-Za-z0-9_]){15}(?!@)|([A-Za-z0-9_]){1,14}(?![A-Za-z0-9_]*@))c                 óR   — |€d}t        | t        «      r| j                  ||«      S | S )Núutf-8)Ú
isinstanceÚbytesÚdecode)ÚtextÚencodingÚerrorss      úi/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/tokenize/casual.pyÚ_str_to_unicoder   ì   s-   € ØÐØˆÜ�$œÔØ�{‰{˜8 VÓ,Ð,Ø€Kó    c                 óR   ‡‡— ˆˆfd„}t         j                  |t        | |«      «      S )u·  
    Remove entities from text by converting them to their
    corresponding unicode character.

    :param text: a unicode string or a byte string encoded in the given
    `encoding` (which defaults to 'utf-8').

    :param list keep:  list of entity names which should not be replaced.    This supports both numeric entities (``&#nnnn;`` and ``&#hhhh;``)
    and named entities (such as ``&nbsp;`` or ``&gt;``).

    :param bool remove_illegal: If `True`, entities that can't be converted are    removed. Otherwise, entities that can't be converted are kept "as
    is".

    :returns: A unicode string with the entities removed.

    See https://github.com/scrapy/w3lib/blob/master/w3lib/html.py

        >>> from nltk.tokenize.casual import _replace_html_entities
        >>> _replace_html_entities(b'Price: &pound;100')
        'Price: \xa3100'
        >>> print(_replace_html_entities(b'Price: &pound;100'))
        Price: Â£100
        >>>
    c                 ó   •— | j                  d«      }| j                  d«      rU	 | j                  d«      rt        |d«      }nt        |d«      }d|cxk  rdk  rn nt        |f«      j                  d«      S n>|‰v r| j                  d	«      S t
        j                  j                  j                  |«      }|�	 t        |«      S ‰rd
S | j                  d	«      S # t        $ r d }Y Œ0w xY w# t        t        f$ r Y Œ7w xY w)Né   r   é   é   é
   é€   éŸ   Úcp1252r   Ú )ÚgroupÚintr	   r
   Ú
ValueErrorÚhtmlÚentitiesÚname2codepointÚgetÚchrÚOverflowError)ÚmatchÚentity_bodyÚnumberÚkeepÚremove_illegals      €€r   Ú_convert_entityz/_replace_html_entities.<locals>._convert_entity  s÷   ø€ Ø—k‘k !“nˆØ�;‰;�qŒ>ðØ—;‘;˜q”>Ü  ¨bÓ1‘Fä  ¨bÓ1�Fð
 ˜6Ô) TÕ)Ü  & Ó+×2Ñ2°8Ó<Ð<øð ˜dÑ"Ø—{‘{ 1“~Ð%Ü—]‘]×1Ñ1×5Ñ5°kÓBˆFØÐðÜ˜6“{Ð"ñ $ˆrÐ7¨¯©°Q«Ð7øô ò Ø’ðûô ¤Ð.ò Ùðús$   ¥AC Â:
C+ ÃC(Ã'C(Ã+C=Ã<C=)ÚENT_REÚsubr   )r   r'   r(   r   r)   s    ``  r   Ú_replace_html_entitiesr,   ô   s"   ù€ õ88ô8 �:‰:�o¤°t¸XÓ'FÓGÐGr   c                   ób   — e Zd ZdZdZdZ	 	 	 	 d	d„Zdedee   fd„Z	e
d
d„«       Ze
d
d„«       Zy)ÚTweetTokenizeraÚ  
    Tokenizer for tweets.

        >>> from nltk.tokenize import TweetTokenizer
        >>> tknzr = TweetTokenizer()
        >>> s0 = "This is a cooool #dummysmiley: :-) :-P <3 and some arrows < > -> <--"
        >>> tknzr.tokenize(s0) # doctest: +NORMALIZE_WHITESPACE
        ['This', 'is', 'a', 'cooool', '#dummysmiley', ':', ':-)', ':-P', '<3', 'and', 'some', 'arrows', '<', '>', '->',
         '<--']

    Examples using `strip_handles` and `reduce_len parameters`:

        >>> tknzr = TweetTokenizer(strip_handles=True, reduce_len=True)
        >>> s1 = '@remy: This is waaaaayyyy too much for you!!!!!!'
        >>> tknzr.tokenize(s1)
        [':', 'This', 'is', 'waaayyy', 'too', 'much', 'for', 'you', '!', '!', '!']
    Nc                 ó<   — || _         || _        || _        || _        y)ae  
        Create a `TweetTokenizer` instance with settings for use in the `tokenize` method.

        :param preserve_case: Flag indicating whether to preserve the casing (capitalisation)
            of text used in the `tokenize` method. Defaults to True.
        :type preserve_case: bool
        :param reduce_len: Flag indicating whether to replace repeated character sequences
            of length 3 or greater with sequences of length 3. Defaults to False.
        :type reduce_len: bool
        :param strip_handles: Flag indicating whether to remove Twitter handles of text used
            in the `tokenize` method. Defaults to False.
        :type strip_handles: bool
        :param match_phone_numbers: Flag indicating whether the `tokenize` method should look
            for phone numbers. Defaults to True.
        :type match_phone_numbers: bool
        N©Úpreserve_caseÚ
reduce_lenÚstrip_handlesÚmatch_phone_numbers)Úselfr1   r2   r3   r4   s        r   Ú__init__zTweetTokenizer.__init__L  s#   € ð. +ˆÔØ$ˆŒØ*ˆÔØ#6ˆÕ r   r   Úreturnc                 ón  — t        |«      }| j                  rt        |«      }| j                  rt	        |«      }t
        j                  d|«      }| j                  r| j                  j                  |«      }n| j                  j                  |«      }| j                  st        t        d„ |«      «      }|S )zÒTokenize the input text.

        :param text: str
        :rtype: list(str)
        :return: a tokenized list of strings; joining this list returns        the original string if `preserve_case=False`.
        ú\1\1\1c                 óP   — t         j                  | «      r| S | j                  «       S )N)ÚEMOTICON_REÚsearchÚlower)Úxs    r   ú<lambda>z)TweetTokenizer.tokenize.<locals>.<lambda>‚  s   € ¤K×$6Ñ$6°qÔ$9˜q€ ¸q¿w¹w»y€ r   )r,   r3   Úremove_handlesr2   Úreduce_lengtheningÚHANG_REr+   r4   ÚPHONE_WORD_REÚfindallÚWORD_REr1   ÚlistÚmap)r5   r   Ú	safe_textÚwordss       r   ÚtokenizezTweetTokenizer.tokenizeh  sš   € ô & dÓ+ˆà×ÒÜ! $Ó'ˆDà�?Š?Ü% dÓ+ˆDä—K‘K 	¨4Ó0ˆ	à×#Ò#Ø×&Ñ&×.Ñ.¨yÓ9‰Eà—L‘L×(Ñ(¨Ó3ˆEà×!Ò!ÜÜÑHÈ5ÓQóˆEð ˆr   c                 ó,  — t        | «      j                  skt        j                  ddj	                  t
        «      › d�t        j                  t        j                  z  t        j                  z  «      t        | «      _        t        | «      j                  S )zCore TweetTokenizer regexú(ú|ú))	ÚtypeÚ_WORD_REÚregexÚcompileÚjoinÚREGEXPSÚVERBOSEÚIÚUNICODE©r5   s    r   rE   zTweetTokenizer.WORD_RE†  sg   € ô �D‹z×"Ò"Ü"'§-¡-Ø�C—H‘HœWÓ%Ð& aÐ(Ü—‘¤§¡Ñ'¬%¯-©-Ñ7ó#ŒD�‹JÔô �D‹z×"Ñ"Ð"r   c                 ó,  — t        | «      j                  skt        j                  ddj	                  t
        «      › d�t        j                  t        j                  z  t        j                  z  «      t        | «      _        t        | «      j                  S )z#Secondary core TweetTokenizer regexrL   rM   rN   )	rO   Ú_PHONE_WORD_RErQ   rR   rS   ÚREGEXPS_PHONErU   rV   rW   rX   s    r   rC   zTweetTokenizer.PHONE_WORD_RE‘  sg   € ô �D‹z×(Ò(Ü(-¯©Ø�C—H‘Hœ]Ó+Ð,¨AÐ.Ü—‘¤§¡Ñ'¬%¯-©-Ñ7ó)ŒD�‹JÔ%ô �D‹z×(Ñ(Ð(r   ©TFFT)r7   zregex.Pattern)Ú__name__Ú
__module__Ú__qualname__Ú__doc__rP   rZ   r6   ÚstrrF   rJ   ÚpropertyrE   rC   © r   r   r.   r.   2  se   „ ñð( €HØ€Nð ØØØ ó7ð8˜Sð  T¨#¡Yó ð< ò#ó ð#ð ò)ó ñ)r   r.   c                 óP   — t        j                  d«      }|j                  d| «      S )ze
    Replace repeated character sequences of length 3 or greater with sequences
    of length 3.
    z	(.)\1{2,}r9   )rQ   rR   r+   )r   Úpatterns     r   rA   rA   ¢  s#   € ô
 �m‰m˜LÓ)€GØ�;‰;�y $Ó'Ð'r   c                 ó.   — t         j                  d| «      S )z4
    Remove Twitter username handles from text.
    ú )Ú
HANDLES_REr+   )r   s    r   r@   r@   «  s   € ô
 �>‰>˜#˜tÓ$Ð$r   c                 ó>   — t        ||||¬«      j                  | «      S )z:
    Convenience function for wrapping the tokenizer.
    r0   )r.   rJ   )r   r1   r2   r3   r4   s        r   Úcasual_tokenizerj   ¸  s(   € ô Ø#ØØ#Ø/ô	÷
 �hˆtƒnðr   )NÚstrict)rc   Tr   r\   )r`   r   Útypingr   rQ   Únltk.tokenize.apir   Ú	EMOTICONSÚURLSÚFLAGSÚPHONE_REGEXrT   r[   rR   rB   rU   rV   rW   r;   r*   rh   r   r,   r.   rA   r@   rj   rc   r   r   ú<module>rr      s  ðñó@ Ý ã å (ð$	€	ð$)€ðf	€ð 	€ð$ 	ààààà(à.ð	ð 
ð
ð/"€ðJ ˜‘˜[Ð7¨7°1°2¨;Ñ7€ð ˆ%�-‰-Ð/Ó
0€ð ˆe�m‰m˜I u§}¡}°u·w±wÑ'>ÀÇÁÑ'NÓO€ð 
ˆ�‰Ð.Ó	/€ð ˆU�]‰]ðHó€
óó8Hô|h)�Zô h)ò`(ò%ð ØØØôr   