Ë
    çÍ:j9A  ã                   ó„   — d Z ddlZddlZddlmZ ddlmZmZ ddlm	Z	 ddl
mZ ddlmZ  G d„ d	e	«      Z G d
„ de	«      Zy)a	  

Penn Treebank Tokenizer

The Treebank tokenizer uses regular expressions to tokenize text as in Penn Treebank.
This implementation is a port of the tokenizer sed script written by Robert McIntyre
and available at http://www.cis.upenn.edu/~treebank/tokenizer.sed.
é    N)ÚIterator)ÚListÚTuple)Ú
TokenizerI)ÚMacIntyreContractions)Úalign_tokensc            
       óD  — e Zd ZdZ ej
                  d«      df ej
                  d«      df ej
                  d«      dfgZ ej
                  d«      d	f ej
                  d
«      df ej
                  d«      df ej
                  d«      df ej
                  d«      df ej
                  d«      df ej
                  d«      dfgZ ej
                  d«      dfZ ej
                  d«      df ej
                  d«      df ej
                  d«      df ej
                  d«      df ej
                  d«      df ej
                  d«      d fgZ	 ej
                  d!«      d"fZ
 ej
                  d#«      d$f ej
                  d%«      d$f ej
                  d&«      d'f ej
                  d(«      d'fgZ e«       Z e eej
                  ej                   «      «      Z e eej
                  ej"                  «      «      Z	 d0d)ed*ed+ed,ee   fd-„Zd)ed,eeeef      fd.„Zy/)1ÚTreebankWordTokenizera	  
    The Treebank tokenizer uses regular expressions to tokenize text as in Penn Treebank.

    This tokenizer performs the following steps:

    - split standard contractions, e.g. ``don't`` -> ``do n't`` and ``they'll`` -> ``they 'll``
    - treat most punctuation characters as separate tokens
    - split off commas and single quotes, when followed by whitespace
    - separate periods that appear at the end of line

    >>> from nltk.tokenize import TreebankWordTokenizer
    >>> s = '''Good muffins cost $3.88\nin New York.  Please buy me\ntwo of them.\nThanks.'''
    >>> TreebankWordTokenizer().tokenize(s)
    ['Good', 'muffins', 'cost', '$', '3.88', 'in', 'New', 'York.', 'Please', 'buy', 'me', 'two', 'of', 'them.', 'Thanks', '.']
    >>> s = "They'll save and invest more."
    >>> TreebankWordTokenizer().tokenize(s)
    ['They', "'ll", 'save', 'and', 'invest', 'more', '.']
    >>> s = "hi, my name can't hello,"
    >>> TreebankWordTokenizer().tokenize(s)
    ['hi', ',', 'my', 'name', 'ca', "n't", 'hello', ',']
    z^\"ú``z(``)z \1 z([ \(\[{<])(\"|\'{2})z\1 `` z([:,])([^\d])z \1 \2z([:,])$z\.\.\.z ... z[;@#$%&]z \g<0> z([^\.])(\.)([\]\)}>"\']*)\s*$z\1 \2\3 z[?!]z([^'])' z\1 ' z[\]\[\(\)\{\}\<\>]z\(ú-LRB-z\)ú-RRB-z\[ú-LSB-z\]ú-RSB-z\{ú-LCB-z\}ú-RCB-ú--ú -- ú''z '' ú"z([^' ])('[sS]|'[mM]|'[dD]|') z\1 \2 z)([^' ])('ll|'LL|'re|'RE|'ve|'VE|n't|N'T) ÚtextÚconvert_parenthesesÚ
return_strÚreturnc                 ó¶  — |durt        j                  dt        d¬«       | j                  D ]  \  }}|j	                  ||«      }Œ | j
                  D ]  \  }}|j	                  ||«      }Œ | j                  \  }}|j	                  ||«      }|r&| j                  D ]  \  }}|j	                  ||«      }Œ | j                  \  }}|j	                  ||«      }d|z   dz   }| j                  D ]  \  }}|j	                  ||«      }Œ | j                  D ]  }|j	                  d|«      }Œ | j                  D ]  }|j	                  d|«      }Œ |j                  «       S )aý  Return a tokenized copy of `text`.

        >>> from nltk.tokenize import TreebankWordTokenizer
        >>> s = '''Good muffins cost $3.88 (roughly 3,36 euros)\nin New York.  Please buy me\ntwo of them.\nThanks.'''
        >>> TreebankWordTokenizer().tokenize(s) # doctest: +NORMALIZE_WHITESPACE
        ['Good', 'muffins', 'cost', '$', '3.88', '(', 'roughly', '3,36',
        'euros', ')', 'in', 'New', 'York.', 'Please', 'buy', 'me', 'two',
        'of', 'them.', 'Thanks', '.']
        >>> TreebankWordTokenizer().tokenize(s, convert_parentheses=True) # doctest: +NORMALIZE_WHITESPACE
        ['Good', 'muffins', 'cost', '$', '3.88', '-LRB-', 'roughly', '3,36',
        'euros', '-RRB-', 'in', 'New', 'York.', 'Please', 'buy', 'me', 'two',
        'of', 'them.', 'Thanks', '.']

        :param text: A string with a sentence or sentences.
        :type text: str
        :param convert_parentheses: if True, replace parentheses to PTB symbols,
            e.g. `(` to `-LRB-`. Defaults to False.
        :type convert_parentheses: bool, optional
        :param return_str: If True, return tokens as space-separated string,
            defaults to False.
        :type return_str: bool, optional
        :return: List of tokens from `text`.
        :rtype: List[str]
        FzHParameter 'return_str' has been deprecated and should no longer be used.é   )ÚcategoryÚ
stacklevelú z \1 \2 )ÚwarningsÚwarnÚDeprecationWarningÚSTARTING_QUOTESÚsubÚPUNCTUATIONÚPARENS_BRACKETSÚCONVERT_PARENTHESESÚDOUBLE_DASHESÚENDING_QUOTESÚCONTRACTIONS2ÚCONTRACTIONS3Úsplit)Úselfr   r   r   ÚregexpÚsubstitutions         úk/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/tokenize/treebank.pyÚtokenizezTreebankWordTokenizer.tokenizef   s€  € ð6 ˜UÑ"Ü�M‰Mð"ä+Øõ	ð %)×$8Ñ$8ò 	2Ñ ˆF�LØ—:‘:˜l¨DÓ1‰Dð	2ð %)×$4Ñ$4ò 	2Ñ ˆF�LØ—:‘:˜l¨DÓ1‰Dð	2ð  $×3Ñ3Ñˆ�Ø�z‰z˜,¨Ó-ˆáØ(,×(@Ñ(@ò 6Ñ$�˜Ø—z‘z ,°Ó5‘ð6ð  $×1Ñ1Ñˆ�Ø�z‰z˜,¨Ó-ˆð �T‰z˜CÑˆà$(×$6Ñ$6ò 	2Ñ ˆF�LØ—:‘:˜l¨DÓ1‰Dð	2ð ×(Ñ(ò 	0ˆFØ—:‘:˜j¨$Ó/‰Dð	0à×(Ñ(ò 	0ˆFØ—:‘:˜j¨$Ó/‰Dð	0ð �z‰z‹|Ðó    c              #   ó.  K  — | j                  |«      }d|v sd|v rVt        j                  d|«      D �cg c]  }|j                  «       ‘Œ }}|D �cg c]  }|dv r|j	                  d«      n|‘Œ }}n|}t        ||«      E d{  –—†  yc c}w c c}w 7 Œ­w)a‰  
        Returns the spans of the tokens in ``text``.
        Uses the post-hoc nltk.tokens.align_tokens to return the offset spans.

            >>> from nltk.tokenize import TreebankWordTokenizer
            >>> s = '''Good muffins cost $3.88\nin New (York).  Please (buy) me\ntwo of them.\n(Thanks).'''
            >>> expected = [(0, 4), (5, 12), (13, 17), (18, 19), (19, 23),
            ... (24, 26), (27, 30), (31, 32), (32, 36), (36, 37), (37, 38),
            ... (40, 46), (47, 48), (48, 51), (51, 52), (53, 55), (56, 59),
            ... (60, 62), (63, 68), (69, 70), (70, 76), (76, 77), (77, 78)]
            >>> list(TreebankWordTokenizer().span_tokenize(s)) == expected
            True
            >>> expected = ['Good', 'muffins', 'cost', '$', '3.88', 'in',
            ... 'New', '(', 'York', ')', '.', 'Please', '(', 'buy', ')',
            ... 'me', 'two', 'of', 'them.', '(', 'Thanks', ')', '.']
            >>> [s[start:end] for start, end in TreebankWordTokenizer().span_tokenize(s)] == expected
            True

        :param text: A string with a sentence or sentences.
        :type text: str
        :yield: Tuple[int, int]
        r   r   z
``|'{2}|\")r   r   r   r   N)r0   ÚreÚfinditerÚgroupÚpopr   )r,   r   Ú
raw_tokensÚmÚmatchedÚtokÚtokenss          r/   Úspan_tokenizez#TreebankWordTokenizer.span_tokenize­   s¤   è ø€ ð. —]‘] 4Ó(ˆ
ð �4‰K˜T T™\ä*,¯+©+°mÀTÓ*JÖK Q�q—w‘w•yÐKˆGÐKð
 &öàð #&Ð):Ñ":�—‘˜A”ÀÑCðˆFñ ð
  ˆFä ¨Ó-×-Ñ-ùò Lùòð 	.ús(   ‚2B´B	ÁBÁBÁ/BÂBÂBN)FF)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r3   Úcompiler"   r$   r%   r&   r'   r(   r   Ú_contractionsÚlistÚmapr)   r*   ÚstrÚboolr0   r   ÚtupleÚintr<   © r1   r/   r
   r
      sw  „ ñð0 
ˆ�‰�FÓ	˜UÐ#Ø	ˆ�‰�GÓ	˜gÐ&Ø	ˆ�‰Ð,Ó	-¨yÐ9ð€Oð 
ˆ�‰Ð$Ó	% yÐ1Ø	ˆ�‰�JÓ	 Ð)Ø	ˆ�‰�IÓ	 Ð)Ø	ˆ�‰�KÓ	  *Ð-àˆB�J‰JÐ7Ó8Øð	
ð 
ˆ�‰�GÓ	˜jÐ)Ø	ˆ�‰�KÓ	  (Ð+ð€Kð "�r—z‘zÐ"7Ó8¸*ÐE€Oð 
ˆ�‰�EÓ	˜GÐ$Ø	ˆ�‰�EÓ	˜GÐ$Ø	ˆ�‰�EÓ	˜GÐ$Ø	ˆ�‰�EÓ	˜GÐ$Ø	ˆ�‰�EÓ	˜GÐ$Ø	ˆ�‰�EÓ	˜GÐ$ðÐð  �R—Z‘Z Ó&¨Ð0€Mð 
ˆ�‰�EÓ	˜FÐ#Ø	ˆ�‰�DÓ	˜6Ð"Ø	ˆ�‰Ð4Ó	5°yÐAØ	ˆ�‰Ð@Ó	AÀ9ÐMð	€Mñ *Ó+€MÙ™˜RŸZ™Z¨×)DÑ)DÓEÓF€MÙ™˜RŸZ™Z¨×)DÑ)DÓEÓF€Mð PUñEØðEØ.2ðEØHLðEà	ˆc‰óEðN). #ð ).¨(°5¸¸c¸±?Ñ*Cô ).r1   r
   c                   óŽ  — e Zd ZdZ e«       Zej                  D � ��cg c]'  }t        j                  |j                  dd«      «      ‘Œ) c}}} Zej                  D � ��cg c]'  }t        j                  |j                  dd«      «      ‘Œ) c}}} Z
 ej                  d«      df ej                  d«      df ej                  d«      df ej                  d	«      df ej                  d
«      df ej                  d«      df ej                  d«      dfgZ ej                  d«      dfZ ej                  d«      df ej                  d«      df ej                  d«      df ej                  d«      df ej                  d«      df ej                  d«      dfgZ ej                  d«      df ej                  d«      df ej                  d «      dfgZ ej                  d!«      d"f ej                  d#«      df ej                  d$«      d%f ej                  d&«      df ej                  d'«      df ej                  d(«      d)f ej                  d*«      d+fgZ ej                  d,«      d-f ej                  d.«      d+f ej                  d/«      dfgZd6d0ee   d1ed2efd3„Zd6d0ee   d1ed2efd4„Zy5c c}}} w c c}}} w )7ÚTreebankWordDetokenizera‹  
    The Treebank detokenizer uses the reverse regex operations corresponding to
    the Treebank tokenizer's regexes.

    Note:

    - There're additional assumption mades when undoing the padding of ``[;@#$%&]``
      punctuation symbols that isn't presupposed in the TreebankTokenizer.
    - There're additional regexes added in reversing the parentheses tokenization,
       such as the ``r'([\]\)\}\>])\s([:;,.])'``, which removes the additional right
       padding added to the closing parentheses precedding ``[:;,.]``.
    - It's not possible to return the original whitespaces as they were because
      there wasn't explicit records of where `'\n'`, `'\t'` or `'\s'` were removed at
      the text.split() operation.

    >>> from nltk.tokenize.treebank import TreebankWordTokenizer, TreebankWordDetokenizer
    >>> s = '''Good muffins cost $3.88\nin New York.  Please buy me\ntwo of them.\nThanks.'''
    >>> d = TreebankWordDetokenizer()
    >>> t = TreebankWordTokenizer()
    >>> toks = t.tokenize(s)
    >>> d.detokenize(toks)
    'Good muffins cost $3.88 in New York. Please buy me two of them. Thanks.'

    The MXPOST parentheses substitution can be undone using the ``convert_parentheses``
    parameter:

    >>> s = '''Good muffins cost $3.88\nin New (York).  Please (buy) me\ntwo of them.\n(Thanks).'''
    >>> expected_tokens = ['Good', 'muffins', 'cost', '$', '3.88', 'in',
    ... 'New', '-LRB-', 'York', '-RRB-', '.', 'Please', '-LRB-', 'buy',
    ... '-RRB-', 'me', 'two', 'of', 'them.', '-LRB-', 'Thanks', '-RRB-', '.']
    >>> expected_tokens == t.tokenize(s, convert_parentheses=True)
    True
    >>> expected_detoken = 'Good muffins cost $3.88 in New (York). Please (buy) me two of them. (Thanks).'
    >>> expected_detoken == d.detokenize(t.tokenize(s, convert_parentheses=True), convert_parentheses=True)
    True

    During tokenization it's safe to add more spaces but during detokenization,
    simply undoing the padding doesn't really help.

    - During tokenization, left and right pad is added to ``[!?]``, when
      detokenizing, only left shift the ``[!?]`` is needed.
      Thus ``(re.compile(r'\s([?!])'), r'\g<1>')``.

    - During tokenization ``[:,]`` are left and right padded but when detokenizing,
      only left shift is necessary and we keep right pad after comma/colon
      if the string after is a non-digit.
      Thus ``(re.compile(r'\s([:,])\s([^\d])'), r'\1 \2')``.

    >>> from nltk.tokenize.treebank import TreebankWordDetokenizer
    >>> toks = ['hello', ',', 'i', 'ca', "n't", 'feel', 'my', 'feet', '!', 'Help', '!', '!']
    >>> twd = TreebankWordDetokenizer()
    >>> twd.detokenize(toks)
    "hello, i can't feel my feet! Help!!"

    >>> toks = ['hello', ',', 'i', "can't", 'feel', ';', 'my', 'feet', '!',
    ... 'Help', '!', '!', 'He', 'said', ':', 'Help', ',', 'help', '?', '!']
    >>> twd.detokenize(toks)
    "hello, i can't feel; my feet! Help!! He said: Help, help?!"
    z(?#X)z\sz+([^' ])\s('ll|'LL|'re|'RE|'ve|'VE|n't|N'T) z\1\2 z([^' ])\s('[sS]|'[mM]|'[dD]|') z([^'\s])\s(\'\')ú\1\2z([,.;:!?'])\s+(\"|\'\')z(\'\')\s([.,:)\]>};%])r   r   z([,.;:!?])"(\')z\1\2"r   r   r   ú(r   ú)r   ú[r   ú]r   ú{r   ú}z([\[\(\{\<])\sz\g<1>z\s([\]\)\}\>])z([\]\)\}\>])\s([:;,.])z([^'])\s'\sz\1' z\s([?!])z([^\.])\s(\.)([\]\)}>"\']*)\s*$z\1\2\3z([#$])\sz\s([;%])z
\s\.\.\.\sz...z\s([:,])z\1z([ (\[{<])\s``z\1``z(``)\sr   r;   r   r   c                 óÂ  — dj                  |«      }d|z   dz   }| j                  D ]  }|j                  d|«      }Œ | j                  D ]  }|j                  d|«      }Œ | j                  D ]  \  }}|j                  ||«      }Œ |j                  «       }| j                  \  }}|j                  ||«      }|r&| j                  D ]  \  }}|j                  ||«      }Œ | j                  D ]  \  }}|j                  ||«      }Œ | j                  D ]  \  }}|j                  ||«      }Œ | j                  D ]  \  }}|j                  ||«      }Œ |j                  «       S )a¥  
        Treebank detokenizer, created by undoing the regexes from
        the TreebankWordTokenizer.tokenize.

        :param tokens: A list of strings, i.e. tokenized text.
        :type tokens: List[str]
        :param convert_parentheses: if True, replace PTB symbols with parentheses,
            e.g. `-LRB-` to `(`. Defaults to False.
        :type convert_parentheses: bool, optional
        :return: str
        r   rL   )Újoinr*   r#   r)   r(   Ústripr'   r&   r%   r$   r"   )r,   r;   r   r   r-   r.   s         r/   r0   z TreebankWordDetokenizer.tokenizeb  sˆ  € ð �x‰x˜Óˆð �T‰z˜CÑˆð ×(Ñ(ò 	-ˆFØ—:‘:˜g tÓ,‰Dð	-à×(Ñ(ò 	-ˆFØ—:‘:˜g tÓ,‰Dð	-ð %)×$6Ñ$6ò 	2Ñ ˆF�LØ—:‘:˜l¨DÓ1‰Dð	2ð �z‰z‹|ˆð  $×1Ñ1Ñˆ�Ø�z‰z˜,¨Ó-ˆáØ(,×(@Ñ(@ò 6Ñ$�˜Ø—z‘z ,°Ó5‘ð6ð %)×$8Ñ$8ò 	2Ñ ˆF�LØ—:‘:˜l¨DÓ1‰Dð	2ð %)×$4Ñ$4ò 	2Ñ ˆF�LØ—:‘:˜l¨DÓ1‰Dð	2ð %)×$8Ñ$8ò 	2Ñ ˆF�LØ—:‘:˜l¨DÓ1‰Dð	2ð �z‰z‹|Ðr1   c                 ó&   — | j                  ||«      S )z&Duck-typing the abstract *tokenize()*.)r0   )r,   r;   r   s      r/   Ú
detokenizez"TreebankWordDetokenizer.detokenize—  s   € à�}‰}˜VÐ%8Ó9Ð9r1   N)F)r=   r>   r?   r@   r   rB   r)   r3   rA   Úreplacer*   r(   r'   r&   r%   r$   r"   rC   rE   rF   r0   rW   )Ú.0Úpatternr3   s   000r/   rK   rK   Ù   s  „ ñ:ñx *Ó+€Mð %×2Ñ2÷ð àô 	�
‰
�7—?‘? 7¨EÓ2Õ3ô€Mð %×2Ñ2÷ð àô 	�
‰
�7—?‘? 7¨EÓ2Õ3ô€Mð 
ˆ�‰ÐBÓ	CÀXÐNØ	ˆ�‰Ð6Ó	7¸ÐBð 
ˆ�‰Ð'Ó	(¨'Ð2à	ˆ�‰Ð.Ó	/°Ð9àˆB�J‰JÐ0Ó1Øð	
ð 
ˆ�‰�EÓ	˜CÐ à	ˆ�‰Ð&Ó	'¨Ð5ð€Mð$  �R—Z‘Z Ó(¨%Ð0€Mð 
ˆ�‰�GÓ	˜cÐ"Ø	ˆ�‰�GÓ	˜cÐ"Ø	ˆ�‰�GÓ	˜cÐ"Ø	ˆ�‰�GÓ	˜cÐ"Ø	ˆ�‰�GÓ	˜cÐ"Ø	ˆ�‰�GÓ	˜cÐ"ðÐð 
ˆ�‰Ð%Ó	&¨Ð1Ø	ˆ�‰Ð%Ó	&¨Ð1Ø	ˆ�‰Ð-Ó	.°Ð8ð€Oð 
ˆ�‰�NÓ	# WÐ-Ø	ˆ�‰�KÓ	  (Ð+à	ˆ�‰Ð6Ó	7¸ÐCð
 
ˆ�‰�KÓ	  (Ð+Ø	ˆ�‰�KÓ	  (Ð+à	ˆ�‰�MÓ	" FÐ+ð ˆB�J‰J�{Ó#Øð	
ð€Kð, 
ˆ�‰Ð%Ó	&¨Ð0Ø	ˆ�‰�IÓ	 Ð&Ø	ˆ�‰�EÓ	˜DÐ!ð€Oñ3˜t C™yð 3¸tð 3ÐPSó 3ñj:  c¡ð :Àð :ÐRUô :ùôAùôs   ž,J9Á,K rK   )r@   r3   r   Úcollections.abcr   Útypingr   r   Únltk.tokenize.apir   Únltk.tokenize.destructiver   Únltk.tokenize.utilr   r
   rK   rI   r1   r/   ú<module>r`      s>   ðñó 
Û Ý $ß å (Ý ;Ý +ôx.˜Jô x.ôv@:˜jõ @:r1   