Ë
    çÍ:jŽ  ã                   óˆ   — d Z ddlmZmZ ddlmZmZ  G d„ de«      Z G d„ de«      Z G d„ d	e«      Z	 G d
„ de«      Z
dd„Zy)aÔ  
Simple Tokenizers

These tokenizers divide strings into substrings using the string
``split()`` method.
When tokenizing using a particular delimiter string, use
the string ``split()`` method directly, as this is more efficient.

The simple tokenizers are *not* available as separate functions;
instead, you should just use the string ``split()`` method directly:

    >>> s = "Good muffins cost $3.88\nin New York.  Please buy me\ntwo of them.\n\nThanks."
    >>> s.split() # doctest: +NORMALIZE_WHITESPACE
    ['Good', 'muffins', 'cost', '$3.88', 'in', 'New', 'York.',
    'Please', 'buy', 'me', 'two', 'of', 'them.', 'Thanks.']
    >>> s.split(' ') # doctest: +NORMALIZE_WHITESPACE
    ['Good', 'muffins', 'cost', '$3.88\nin', 'New', 'York.', '',
    'Please', 'buy', 'me\ntwo', 'of', 'them.\n\nThanks.']
    >>> s.split('\n') # doctest: +NORMALIZE_WHITESPACE
    ['Good muffins cost $3.88', 'in New York.  Please buy me',
    'two of them.', '', 'Thanks.']

The simple tokenizers are mainly useful because they follow the
standard ``TokenizerI`` interface, and so can be used with any code
that expects a tokenizer.  For example, these tokenizers can be used
to specify the tokenization conventions when building a `CorpusReader`.

é    )ÚStringTokenizerÚ
TokenizerI)Úregexp_span_tokenizeÚstring_span_tokenizec                   ó   — e Zd ZdZdZy)ÚSpaceTokenizeraÎ  Tokenize a string using the space character as a delimiter,
    which is the same as ``s.split(' ')``.

        >>> from nltk.tokenize import SpaceTokenizer
        >>> s = "Good muffins cost $3.88\nin New York.  Please buy me\ntwo of them.\n\nThanks."
        >>> SpaceTokenizer().tokenize(s) # doctest: +NORMALIZE_WHITESPACE
        ['Good', 'muffins', 'cost', '$3.88\nin', 'New', 'York.', '',
        'Please', 'buy', 'me\ntwo', 'of', 'them.\n\nThanks.']
    ú N©Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú_string© ó    úi/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/tokenize/simple.pyr   r   *   s   „ ñð �Gr   r   c                   ó   — e Zd ZdZdZy)ÚTabTokenizerzäTokenize a string use the tab character as a delimiter,
    the same as ``s.split('\t')``.

        >>> from nltk.tokenize import TabTokenizer
        >>> TabTokenizer().tokenize('a\tb c\n\t d')
        ['a', 'b c\n', ' d']
    ú	Nr
   r   r   r   r   r   8   s   „ ñð �Gr   r   c                   ó    — e Zd ZdZdZd„ Zd„ Zy)ÚCharTokenizerz„Tokenize a string into individual characters.  If this functionality
    is ever required directly, use ``for char in string``.
    Nc                 ó   — t        |«      S ©N)Úlist©ÚselfÚss     r   ÚtokenizezCharTokenizer.tokenizeK   s   € Ü�A‹wˆr   c              #   ób   K  — t        t        dt        |«      dz   «      «      E d {  –—†  y 7 Œ­w)Né   )Ú	enumerateÚrangeÚlenr   s     r   Úspan_tokenizezCharTokenizer.span_tokenizeN   s#   è ø€ ÜœU 1¤c¨!£f¨q¡jÓ1Ó2×2Ò2ús   ‚%/§-¨/)r   r   r   r   r   r   r$   r   r   r   r   r   D   s   „ ñð €Gòó3r   r   c                   ó$   — e Zd ZdZdd„Zd„ Zd„ Zy)ÚLineTokenizera˜  Tokenize a string into its lines, optionally discarding blank lines.
    This is similar to ``s.split('\n')``.

        >>> from nltk.tokenize import LineTokenizer
        >>> s = "Good muffins cost $3.88\nin New York.  Please buy me\ntwo of them.\n\nThanks."
        >>> LineTokenizer(blanklines='keep').tokenize(s) # doctest: +NORMALIZE_WHITESPACE
        ['Good muffins cost $3.88', 'in New York.  Please buy me',
        'two of them.', '', 'Thanks.']
        >>> # same as [l for l in s.split('\n') if l.strip()]:
        >>> LineTokenizer(blanklines='discard').tokenize(s) # doctest: +NORMALIZE_WHITESPACE
        ['Good muffins cost $3.88', 'in New York.  Please buy me',
        'two of them.', 'Thanks.']

    :param blanklines: Indicates how blank lines should be handled.  Valid values are:

        - ``discard``: strip blank lines out of the token list before returning it.
           A line is considered blank if it contains only whitespace characters.
        - ``keep``: leave all blank lines in the token list.
        - ``discard-eof``: if the string ends with a newline, then do not generate
           a corresponding token ``''`` after that newline.
    c                 óX   — d}||vrt        ddj                  |«      z  «      ‚|| _        y )N)ÚdiscardÚkeepúdiscard-eofzBlank lines must be one of: %sr	   )Ú
ValueErrorÚjoinÚ_blanklines)r   Ú
blanklinesÚvalid_blankliness      r   Ú__init__zLineTokenizer.__init__i   s:   € Ø=ÐØÐ-Ñ-ÜØ0°3·8±8Ð<LÓ3MÑMóð ð &ˆÕr   c                 óú   — |j                  «       }| j                  dk(  r"|D �cg c]  }|j                  «       sŒ|‘Œ }}|S | j                  dk(  r%|r#|d   j                  «       s|j	                  «        |S c c}w )Nr(   r*   éÿÿÿÿ)Ú
splitlinesr-   ÚrstripÚstripÚpop)r   r   ÚlinesÚls       r   r   zLineTokenizer.tokenizer   sp   € Ø—‘“ˆà×Ñ˜yÒ(Ø %Ö4˜1¨¯©­’QÐ4ˆEÐ4ð ˆð ×Ñ Ò.Ù˜U 2™YŸ_™_Ô.Ø—	‘	”Øˆùò	 5s
   ¤A8ºA8c              #   ó„   K  — | j                   dk(  rt        |d«      E d {  –—†  y t        |d«      E d {  –—†  y 7 Œ7 Œ­w)Nr)   z\nz
\n(\s+\n)*)r-   r   r   r   s     r   r$   zLineTokenizer.span_tokenize}   s=   è ø€ Ø×Ñ˜vÒ%Ü+¨A¨uÓ5×5Ñ5ä+¨A¨}Ó=×=Ñ=ð 6øà=ús   ‚A ¡<¢A ¶>·A ¾A N©r(   )r   r   r   r   r0   r   r$   r   r   r   r&   r&   R   s   „ ñó,&òó>r   r&   c                 ó6   — t        |«      j                  | «      S r   )r&   r   )Útextr.   s     r   Úline_tokenizer=   Š   s   € Ü˜Ó$×-Ñ-¨dÓ3Ð3r   Nr:   )r   Únltk.tokenize.apir   r   Únltk.tokenize.utilr   r   r   r   r   r&   r=   r   r   r   ú<module>r@      sH   ðñ÷: :ß Iô�_ô ô	�?ô 	ô3�Oô 3ô/>�Jô />ôp4r   