Ë
    çÍ:jà$  ã                   ór   — d dl Z d dlZd dlmZ d dlmZmZ d dlmZ d dl	m
Z
  G d„ d«      Z G d„ d	e«      Zy)
é    N)ÚIterator)ÚListÚTuple)Ú
TokenizerI)Úalign_tokensc                   ó(   — e Zd ZdZg d¢ZddgZddgZy)ÚMacIntyreContractionszI
    List of contractions adapted from Robert MacIntyre's tokenizer.
    )z(?i)\b(can)(?#X)(not)\bz(?i)\b(d)(?#X)('ye)\bz(?i)\b(gim)(?#X)(me)\bz(?i)\b(gon)(?#X)(na)\bz(?i)\b(got)(?#X)(ta)\bz(?i)\b(lem)(?#X)(me)\bz(?i)\b(more)(?#X)('n)\bz(?i)\b(wan)(?#X)(na)(?=\s)z(?i) ('t)(?#X)(is)\bz(?i) ('t)(?#X)(was)\bz(?i)\b(whad)(dd)(ya)\bz(?i)\b(wha)(t)(cha)\bN)Ú__name__Ú
__module__Ú__qualname__Ú__doc__ÚCONTRACTIONS2ÚCONTRACTIONS3ÚCONTRACTIONS4© ó    ún/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/tokenize/destructive.pyr	   r	      s&   „ ñò	€Mð -Ð.FÐG€MØ.Ð0HÐI�Mr   r	   c                   óè  — e Zd ZdZ ej
                  dej                  «      df ej
                  d«      df ej
                  d«      df ej
                  d«      df ej
                  d	ej                  «      d
fgZ ej
                  dej                  «      df ej
                  d«      df ej
                  d«      df ej
                  d«      df ej
                  d«      df ej
                  d«      dfgZ ej
                  dej                  «      df ej
                  d«      df ej
                  d«      df ej
                  dej                  «      df ej
                  d«      df ej
                  dej                  «      df ej
                  d«      df ej
                  d«      df ej
                  d «      d!f ej
                  d"ej                  «      dfg
Z
 ej
                  d#«      dfZ ej
                  d$«      d%f ej
                  d&«      d'f ej
                  d(«      d)f ej
                  d*«      d+f ej
                  d,«      d-f ej
                  d.«      d/fgZ ej
                  d0«      d1fZ e«       Z e eej
                  ej$                  «      «      Z e eej
                  ej&                  «      «      Z	 d9d2ed3ed4ed5ee   fd6„Zd2ed5eeeef      fd7„Zy8):ÚNLTKWordTokenizeraE  
    The NLTK tokenizer that has improved upon the TreebankWordTokenizer.

    This is the method that is invoked by ``word_tokenize()``.  It assumes that the
    text has already been segmented into sentences, e.g. using ``sent_tokenize()``.

    The tokenizer is "destructive" such that the regexes applied will munge the
    input string to a state beyond re-construction. It is possible to apply
    `TreebankWordDetokenizer.detokenize` to the tokenized outputs of
    `NLTKDestructiveWordTokenizer.tokenize` but there's no guarantees to
    revert to the original string.
    u   ([Â«â€œâ€˜â€ž]|[`]+)z \1 z^\"ú``z(``)z([ \(\[{<])(\"|\'{2})z\1 `` z$(?i)(\')(?!re|ve|ll|m|t|s|d|n)(\w)\bz\1 \2u   ([Â»â€�â€™])ú''z '' ú"z\s+ú z([^' ])('[sS]|'[mM]|'[dD]|') z\1 \2 z)([^' ])('ll|'LL|'re|'RE|'ve|'VE|n't|N'T) u&   ([^\.])(\.)([\]\)}>"\'Â»â€�â€™ ]*)\s*$z	\1 \2 \3 z([:,])([^\d])z \1 \2z([:,])$z\.{2,}z \g<0> z[;@#$%&]z[\u2012-\u2015]z([^\.])(\.)([\]\)}>"\']*)\s*$z\1 \2\3 z[?!]z([^'])' z\1 ' z[*]z[\]\[\(\)\{\}\<\>]z\(z-LRB-z\)z-RRB-z\[z-LSB-z\]z-RSB-z\{z-LCB-z\}z-RCB-z--z -- ÚtextÚconvert_parenthesesÚ
return_strÚreturnc                 ó²  — |rt        j                  dt        d¬«       | j                  D ]  \  }}|j	                  ||«      }Œ | j
                  D ]  \  }}|j	                  ||«      }Œ | j                  \  }}|j	                  ||«      }|r&| j                  D ]  \  }}|j	                  ||«      }Œ | j                  \  }}|j	                  ||«      }d|z   dz   }| j                  D ]  \  }}|j	                  ||«      }Œ | j                  D ]  }|j	                  d|«      }Œ | j                  D ]  }|j	                  d|«      }Œ |j                  «       S )aò  Return a tokenized copy of `text`.

        >>> from nltk.tokenize import NLTKWordTokenizer
        >>> s = '''Good muffins cost $3.88 (roughly 3,36 euros)\nin New York.  Please buy me\ntwo of them.\nThanks.'''
        >>> NLTKWordTokenizer().tokenize(s) # doctest: +NORMALIZE_WHITESPACE
        ['Good', 'muffins', 'cost', '$', '3.88', '(', 'roughly', '3,36',
        'euros', ')', 'in', 'New', 'York.', 'Please', 'buy', 'me', 'two',
        'of', 'them.', 'Thanks', '.']
        >>> NLTKWordTokenizer().tokenize(s, convert_parentheses=True) # doctest: +NORMALIZE_WHITESPACE
        ['Good', 'muffins', 'cost', '$', '3.88', '-LRB-', 'roughly', '3,36',
        'euros', '-RRB-', 'in', 'New', 'York.', 'Please', 'buy', 'me', 'two',
        'of', 'them.', 'Thanks', '.']


        :param text: A string with a sentence or sentences.
        :type text: str
        :param convert_parentheses: if True, replace parentheses to PTB symbols,
            e.g. `(` to `-LRB-`. Defaults to False.
        :type convert_parentheses: bool, optional
        :param return_str: If True, return tokens as space-separated string,
            defaults to False.
        :type return_str: bool, optional
        :return: List of tokens from `text`.
        :rtype: List[str]
        zHParameter 'return_str' has been deprecated and should no longer be used.é   )ÚcategoryÚ
stacklevelr   z \1 \2 )ÚwarningsÚwarnÚDeprecationWarningÚSTARTING_QUOTESÚsubÚPUNCTUATIONÚPARENS_BRACKETSÚCONVERT_PARENTHESESÚDOUBLE_DASHESÚENDING_QUOTESr   r   Úsplit)Úselfr   r   r   ÚregexpÚsubstitutions         r   ÚtokenizezNLTKWordTokenizer.tokenize~   s{  € ñ8 Ü�M‰Mð"ä+Øõ	ð %)×$8Ñ$8ò 	2Ñ ˆF�LØ—:‘:˜l¨DÓ1‰Dð	2ð %)×$4Ñ$4ò 	2Ñ ˆF�LØ—:‘:˜l¨DÓ1‰Dð	2ð  $×3Ñ3Ñˆ�Ø�z‰z˜,¨Ó-ˆáØ(,×(@Ñ(@ò 6Ñ$�˜Ø—z‘z ,°Ó5‘ð6ð  $×1Ñ1Ñˆ�Ø�z‰z˜,¨Ó-ˆð �T‰z˜CÑˆà$(×$6Ñ$6ò 	2Ñ ˆF�LØ—:‘:˜l¨DÓ1‰Dð	2ð ×(Ñ(ò 	0ˆFØ—:‘:˜j¨$Ó/‰Dð	0à×(Ñ(ò 	0ˆFØ—:‘:˜j¨$Ó/‰Dð	0ð �z‰z‹|Ðr   c              #   ó.  K  — | j                  |«      }d|v sd|v rVt        j                  d|«      D �cg c]  }|j                  «       ‘Œ }}|D �cg c]  }|dv r|j	                  d«      n|‘Œ }}n|}t        ||«      E d{  –—†  yc c}w c c}w 7 Œ­w)a}  
        Returns the spans of the tokens in ``text``.
        Uses the post-hoc nltk.tokens.align_tokens to return the offset spans.

            >>> from nltk.tokenize import NLTKWordTokenizer
            >>> s = '''Good muffins cost $3.88\nin New (York).  Please (buy) me\ntwo of them.\n(Thanks).'''
            >>> expected = [(0, 4), (5, 12), (13, 17), (18, 19), (19, 23),
            ... (24, 26), (27, 30), (31, 32), (32, 36), (36, 37), (37, 38),
            ... (40, 46), (47, 48), (48, 51), (51, 52), (53, 55), (56, 59),
            ... (60, 62), (63, 68), (69, 70), (70, 76), (76, 77), (77, 78)]
            >>> list(NLTKWordTokenizer().span_tokenize(s)) == expected
            True
            >>> expected = ['Good', 'muffins', 'cost', '$', '3.88', 'in',
            ... 'New', '(', 'York', ')', '.', 'Please', '(', 'buy', ')',
            ... 'me', 'two', 'of', 'them.', '(', 'Thanks', ')', '.']
            >>> [s[start:end] for start, end in NLTKWordTokenizer().span_tokenize(s)] == expected
            True

        :param text: A string with a sentence or sentences.
        :type text: str
        :yield: Tuple[int, int]
        r   r   z
``|'{2}|\")r   r   r   r   N)r0   ÚreÚfinditerÚgroupÚpopr   )r-   r   Ú
raw_tokensÚmÚmatchedÚtokÚtokenss          r   Úspan_tokenizezNLTKWordTokenizer.span_tokenizeÆ   s¤   è ø€ ð. —]‘] 4Ó(ˆ
ð �4‰K˜T T™\ä*,¯+©+°mÀTÓ*JÖK Q�q—w‘w•yÐKˆGÐKð
 &öàð #&Ð):Ñ":�—‘˜A”ÀÑCðˆFñ ð
  ˆFä ¨Ó-×-Ñ-ùò Lùòð 	.ús(   ‚2B´B	ÁBÁBÁ/BÂBÂBN)FF)r
   r   r   r   r2   ÚcompileÚUr%   r+   ÚUNICODEr'   r(   r)   r*   r	   Ú_contractionsÚlistÚmapr   r   ÚstrÚboolr0   r   ÚtupleÚintr;   r   r   r   r   r   &   s7  „ ñð 
ˆ�‰Ð*¨B¯D©DÓ	1°7Ð;Ø	ˆ�‰�FÓ	˜UÐ#Ø	ˆ�‰�GÓ	˜gÐ&Ø	ˆ�‰Ð,Ó	-¨yÐ9Ø	ˆ�‰Ð;¸R¿T¹TÓ	BÀHÐMð€Oð 
ˆ�‰�N B§D¡DÓ	)¨7Ð3Ø	ˆ�‰�EÓ	˜FÐ#Ø	ˆ�‰�DÓ	˜6Ð"Ø	ˆ�‰�FÓ	˜SÐ!Ø	ˆ�‰Ð4Ó	5°yÐAØ	ˆ�‰Ð@Ó	AÀ9ÐMð€Mð( 
ˆ�‰ÐDÀbÇdÁdÓ	KÈ\ÐZØ	ˆ�‰Ð$Ó	% yÐ1Ø	ˆ�‰�JÓ	 Ð)àˆB�J‰J�y "§$¡$Ó'Øð	
ð 
ˆ�‰�KÓ	  *Ð-àˆB�J‰JÐ)¨2¯:©:Ó6Øð	
ð
 ˆB�J‰JÐ7Ó8Øð	
ð 
ˆ�‰�GÓ	˜jÐ)Ø	ˆ�‰�KÓ	  (Ð+àˆB�J‰J�v˜rŸt™tÓ$Øð	
ð'€Kð4 "�r—z‘zÐ"7Ó8¸*ÐE€Oð 
ˆ�‰�EÓ	˜GÐ$Ø	ˆ�‰�EÓ	˜GÐ$Ø	ˆ�‰�EÓ	˜GÐ$Ø	ˆ�‰�EÓ	˜GÐ$Ø	ˆ�‰�EÓ	˜GÐ$Ø	ˆ�‰�EÓ	˜GÐ$ðÐð  �R—Z‘Z Ó&¨Ð0€Mñ *Ó+€MÙ™˜RŸZ™Z¨×)DÑ)DÓEÓF€MÙ™˜RŸZ™Z¨×)DÑ)DÓEÓF€Mð PUñFØðFØ.2ðFØHLðFà	ˆc‰óFðP). #ð ).¨(°5¸¸c¸±?Ñ*Cô ).r   r   )r2   r"   Úcollections.abcr   Útypingr   r   Únltk.tokenize.apir   Únltk.tokenize.utilr   r	   r   r   r   r   ú<module>rJ      s3   ðó 
Û Ý $ß å (Ý +÷Jñ Jô&I.˜
õ I.r   