Ë
    çÍ:jÀn  ã                   óZ  — d Z ddlZddlZddlZddlmZmZmZ ddlm	Z	 ddl
mZ ddlmZ ddlmZ ddlmZ dd	lmZmZ dd
lmZ ddlmZ ddlmZ ddlmZmZmZ  edg d¢«      Z G d„ d«      Z  G d„ d«      Z! G d„ d«      Z" G d„ d«      Z# G d„ de#«      Z$d„ Z%e&dk(  r e%«        g d¢Z'y)a  
This module brings together a variety of NLTK functionality for
text analysis, and provides simple, interactive interfaces.
Functionality includes: concordancing, collocation discovery,
regular expression search over tokenized strings, and
distributional similarity.
é    N)ÚCounterÚdefaultdictÚ
namedtuple)Úreduce)Úlog)ÚBigramCollocationFinder)ÚMLE)Úpadded_everygram_pipeline)ÚBigramAssocMeasuresÚ	f_measure)ÚConditionalFreqDist)ÚFreqDist)Úsent_tokenize)ÚLazyConcatenationÚ
cut_stringÚ	tokenwrapÚConcordanceLine)ÚleftÚqueryÚrightÚoffsetÚ
left_printÚright_printÚlinec                   óL   — e Zd ZdZed„ «       Zddd„ fd„Zd„ Zd„ Zd
d„Z	dd	„Z
y)ÚContextIndexa  
    A bidirectional index between words and their 'contexts' in a text.
    The context of a word is usually defined to be the words that occur
    in a fixed window around the word; but other definitions may also
    be used by providing a custom context function.
    c                 ó–   — |dk7  r| |dz
     j                  «       nd}|t        | «      dz
  k7  r| |dz      j                  «       nd}||fS )z;One left token and one right token, normalized to lowercaser   é   ú*START*ú*END*)ÚlowerÚlen)ÚtokensÚir   r   s       ú^/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/text.pyÚ_default_contextzContextIndex._default_context/   sS   € ð )*¨Qªˆv�a˜!‘e‰}×"Ñ"Ô$°IˆØ)*¬c°&«k¸A©oÒ)=��q˜1‘u‘×#Ñ#Ô%À7ˆØ�eˆ}Ðó    Nc                 ó   — | S ©N© ©Úxs    r%   ú<lambda>zContextIndex.<lambda>6   s   € ÈQ€ r'   c                 ó&  ‡ ‡— |‰ _         ‰‰ _        |r|‰ _        n‰ j                  ‰ _        |r‰D �cg c]  } ||«      sŒ|‘Œ c}Št	        ˆ ˆfd„t        ‰«      D «       «      ‰ _        t	        ˆ ˆfd„t        ‰«      D «       «      ‰ _        y c c}w )Nc              3   ój   •K  — | ]*  \  }}‰j                  |«      ‰j                  ‰|«      f–— Œ, y ­wr)   )Ú_keyÚ_context_func©Ú.0r$   ÚwÚselfr#   s      €€r%   ú	<genexpr>z(ContextIndex.__init__.<locals>.<genexpr>?   s4   øè ø€ ò %
Ù>B¸aÀˆT�Y‰Y�q‹\˜4×-Ñ-¨f°aÓ8Ô9ñ%
ùó   ƒ03c              3   ój   •K  — | ]*  \  }}‰j                  ‰|«      ‰j                  |«      f–— Œ, y ­wr)   )r1   r0   r2   s      €€r%   r6   z(ContextIndex.__init__.<locals>.<genexpr>B   s4   øè ø€ ò %
Ù>B¸aÀˆT×Ñ ¨Ó*¨D¯I©I°a«LÔ9ñ%
ùr7   )r0   Ú_tokensr1   r&   ÚCFDÚ	enumerateÚ_word_to_contextsÚ_context_to_words)r5   r#   Úcontext_funcÚfilterÚkeyÚts   ``    r%   Ú__init__zContextIndex.__init__6   sŠ   ù€ ØˆŒ	ØˆŒÙØ!-ˆDÕà!%×!6Ñ!6ˆDÔÙØ!'Ö5˜A©6°!­9’aÒ5ˆFÜ!$ô %
ÜFOÐPVÓFWô%
ó "
ˆÔô "%ô %
ÜFOÐPVÓFWô%
ó "
ˆÕùò	 6s   ²BÁ Bc                 ó   — | j                   S )zw
        :rtype: list(str)
        :return: The document that this context index was
            created from.
        ©r9   ©r5   s    r%   r#   zContextIndex.tokensF   ó   € ð �|‰|Ðr'   c                 óÐ   — | j                  |«      }t        | j                  |   «      }i }| j                  j                  «       D ]  \  }}t	        |t        |«      «      ||<   Œ |S )z 
        Return a dictionary mapping from words to 'similarity scores,'
        indicating how often these two words occur in the same
        context.
        )r0   Úsetr<   Úitemsr   )r5   ÚwordÚword_contextsÚscoresr4   Ú
w_contextss         r%   Úword_similarity_dictz!ContextIndex.word_similarity_dictN   sj   € ð �y‰y˜‹ˆÜ˜D×2Ñ2°4Ñ8Ó9ˆàˆØ!×3Ñ3×9Ñ9Ó;ò 	B‰MˆAˆzÜ! -´°Z³ÓAˆF�1ŠIð	Bð ˆr'   c                 ó0  — t        t        «      }| j                  | j                  |«         D ]L  }| j                  |   D ]8  }||k7  sŒ	||xx   | j                  |   |   | j                  |   |   z  z  cc<   Œ: ŒN t        ||j                  d¬«      d | S )NT)r@   Úreverse)r   Úintr<   r0   r=   ÚsortedÚget)r5   rJ   ÚnrL   Úcr4   s         r%   Úsimilar_wordszContextIndex.similar_words]   s£   € ÜœSÓ!ˆØ×'Ñ'¨¯	©	°$«Ñ8ò 	ˆAØ×+Ñ+¨AÑ.ò �Ø˜“9Ø˜1“IØ×.Ñ.¨qÑ1°$Ñ7¸$×:PÑ:PÐQRÑ:SÐTUÑ:VÑVñ”Iñð	ô �f &§*¡*°dÔ;¸B¸QÐ?Ð?r'   c                 ó¶  ‡ ‡— |D �cg c]  }‰ j                  |«      ‘Œ }}|D �cg c]  }t        ‰ j                  |   «      ‘Œ }}t        t	        |«      «      D �cg c]  }||   rŒ	||   ‘Œ }}t        t        j                  |«      Š|r|rt        ddj                  |«      «      ‚‰s
t        «       S t        ˆˆ fd„|D «       «      }|S c c}w c c}w c c}w )a§  
        Find contexts where the specified words can all appear; and
        return a frequency distribution mapping each context to the
        number of times that context was used.

        :param words: The words used to seed the similarity search
        :type words: str
        :param fail_on_unknown: If true, then raise a value error if
            any of the given words do not occur at all in the index.
        z%The following word(s) were not found:ú c              3   óT   •K  — | ]  }‰j                   |   D ]  }|‰v sŒ|–— Œ Œ! y ­wr)   )r<   )r3   r4   rU   Úcommonr5   s      €€r%   r6   z/ContextIndex.common_contexts.<locals>.<genexpr>|   s9   øè ø€ ò Ø¨$×*@Ñ*@ÀÑ*CòØ%&ÀqÈFÂ{”ðØñùs   ƒ(Ÿ	()
r0   rH   r<   Úranger"   r   ÚintersectionÚ
ValueErrorÚjoinr   )	r5   ÚwordsÚfail_on_unknownr4   Úcontextsr$   ÚemptyÚfdrZ   s	   `       @r%   Úcommon_contextszContextIndex.common_contextsg   sÊ   ù€ ð (-Ö- !�—‘˜1•Ð-ˆÐ-Ø<AÖB°q”C˜×.Ñ.¨qÑ1Õ2ÐBˆÐBÜ#(¬¨U«Ó#4ÖH˜a¸HÀQ»K��q“ÐHˆÐHÜœ×(Ñ(¨(Ó3ˆÙ‘_ÜÐDÀcÇhÁhÈuÃoÓVÐVÙä“:Ðäô Ø ôó ˆBð ˆIùò .ùÚBùÚHs   ‡C¥CÁ
CÁ'C©é   )F)Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ústaticmethodr&   rB   r#   rN   rV   rd   r*   r'   r%   r   r   '   s>   „ ñð ñó ðð -1¸Á;ó 
ò òó@ôr'   r   c                   ó>   — e Zd ZdZd„ fd„Zd„ Zd„ Zd„ Zd
d„Zdd„Z	y	)ÚConcordanceIndexzs
    An index that can be used to look up the offset locations at which
    a given word occurs in a document.
    c                 ó   — | S r)   r*   r+   s    r%   r-   zConcordanceIndex.<lambda>ˆ   s   € ¨Q€ r'   c                 óÒ   — || _         	 || _        	 t        t        «      | _        	 t        |«      D ]4  \  }}| j                  |«      }| j                  |   j                  |«       Œ6 y)aé  
        Construct a new concordance index.

        :param tokens: The document (list of tokens) that this
            concordance index was created from.  This list can be used
            to access the context of a given word occurrence.
        :param key: A function that maps each token to a normalized
            version that will be used as a key in the index.  E.g., if
            you use ``key=lambda s:s.lower()``, then the index will be
            case-insensitive.
        N)r9   r0   r   ÚlistÚ_offsetsr;   Úappend)r5   r#   r@   ÚindexrJ   s        r%   rB   zConcordanceIndex.__init__ˆ   sg   € ð ˆŒð	 ð ˆŒ	ØDä#¤DÓ)ˆŒØLä$ VÓ,ò 	.‰KˆE�4Ø—9‘9˜T“?ˆDØ�M‰M˜$Ñ×&Ñ& uÕ-ñ	.r'   c                 ó   — | j                   S )z{
        :rtype: list(str)
        :return: The document that this concordance index was
            created from.
        rD   rE   s    r%   r#   zConcordanceIndex.tokens¢   rF   r'   c                 óB   — | j                  |«      }| j                  |   S )zä
        :rtype: list(int)
        :return: A list of the offset positions at which the given
            word occurs.  If a key function was specified for the
            index, then given word's key will be looked up.
        )r0   rq   ©r5   rJ   s     r%   ÚoffsetszConcordanceIndex.offsetsª   s    € ð �y‰y˜‹ˆØ�}‰}˜TÑ"Ð"r'   c                 ó\   — dt        | j                  «      t        | j                  «      fz  S )Nz+<ConcordanceIndex for %d tokens (%d types)>)r"   r9   rq   rE   s    r%   Ú__repr__zConcordanceIndex.__repr__´   s-   € Ø<Ü�—‘ÓÜ�—‘Óð@
ñ 
ð 	
r'   c           
      óH  — t        |t        «      r|}n|g}dj                  |«      }t        d„ |D «       «      }||z
  dz
  dz  }|dz  }g }| j	                  |d   «      }	t        |dd «      D ]C  \  }
}| j	                  |«      D �ch c]
  }||
z
  dz
  ’Œ }}t        |j                  |	«      «      }	ŒE |	rç|	D ]â  }
dj                  | j                  |
|
t        |«      z    «      }| j                  t        d|
|z
  «      |
 }| j                  |
t        |«      z   |
|z    }t        dj                  |«      | «      j                  |«      }t        dj                  |«      |«      }dj                  |||g«      }t        ||||
|||«      }|j                  |«       Œä |S c c}w )z‹
        Find all concordance lines given the query word.

        Provided with a list of words, these will be found as a phrase.
        rX   c              3   óL   K  — | ]  }t        j                  |«      rŒd –— Œ y­w)r   N)ÚunicodedataÚ	combining)r3   Úchars     r%   r6   z4ConcordanceIndex.find_concordance.<locals>.<genexpr>Æ   s   è ø€ ÒU˜t¼×9NÑ9NÈtÕ9TœÑUùs   ‚$�$é   é   r   r   N)Ú
isinstancerp   r^   Úsumrw   r;   rR   r\   r9   r"   Úmaxr   Úrjustr   rr   )r5   rJ   ÚwidthÚphraseÚ
phrase_strÚ
phrase_lenÚ
half_widthÚcontextÚconcordance_listrw   r$   r   Úword_offsetsÚ
query_wordÚleft_contextÚright_contextr   r   Ú
line_printÚconcordance_lines                       r%   Úfind_concordancez!ConcordanceIndex.find_concordanceº   sÄ  € ô �dœDÔ!Ø‰Fà�VˆFà—X‘X˜fÓ%ˆ
ÜÑU zÔUÓUˆ
Ø˜jÑ(¨1Ñ,°Ñ2ˆ
Ø˜1‘*ˆð ÐØ—,‘,˜v a™yÓ)ˆÜ  ¨¨ Ó,ò 	A‰GˆAˆtØ9=¿¹ÀdÓ9KÖL¨v˜F Q™J¨›NÐLˆLÐLÜ˜\×6Ñ6°wÓ?Ó@‰Gð	Añ Øò :�Ø ŸX™X d§l¡l°1°q¼3¸v»;±Ð&GÓH�
à#Ÿ|™|¬C°°1°w±;Ó,?À!ÐD�Ø $§¡¨Q´°V³©_¸qÀ7¹{Ð K�ä'¨¯©°Ó(>ÀÀÓL×RÑRØó�
ô )¨¯©°-Ó)@À*ÓM�à ŸX™X z°:¸{Ð&KÓL�
ä#2Ø ØØ!ØØØØó$Ð ð !×'Ñ'Ð(8Õ9ð-:ð.  Ðùò5 Ms   ÂFc                 óü   — | j                  ||¬«      }|st        d«       yt        |t        |«      «      }t        d|› dt        |«      › d�«       t	        |d| «      D ]  \  }}t        |j
                  «       Œ y)a±  
        Print concordance lines given the query word.
        :param word: The target word or phrase (a list of strings)
        :type word: str or list
        :param lines: The number of lines to display (default=25)
        :type lines: int
        :param width: The width of each line, in characters (default=80)
        :type width: int
        :param save: The option to save the concordance.
        :type save: bool
        )r…   z
no matcheszDisplaying z of z	 matches:N)r’   ÚprintÚminr"   r;   r   )r5   rJ   r…   Úlinesr‹   r$   r‘   s          r%   Úprint_concordancez"ConcordanceIndex.print_concordanceê   s†   € ð  ×0Ñ0°¸UÐ0ÓCÐáÜ�,Õä˜œsÐ#3Ó4Ó5ˆEÜ�K ˜w d¬3Ð/?Ó+@Ð*AÀÐKÔLÜ'0Ð1AÀ&À5Ð1IÓ'Jò -Ñ#�Ð#ÜÐ&×+Ñ+Õ,ñ-r'   N)éP   )r˜   é   )
rg   rh   ri   rj   rB   r#   rw   ry   r’   r—   r*   r'   r%   rm   rm   ‚   s+   „ ññ
 $/ó .ò4ò#ò
ó. ô`-r'   rm   c                   ó   — e Zd ZdZd„ Zd„ Zy)ÚTokenSearcheraâ  
    A class that makes it easier to use regular expressions to search
    over tokenized strings.  The tokenized string is converted to a
    string where tokens are marked with angle brackets -- e.g.,
    ``'<the><window><is><still><open>'``.  The regular expression
    passed to the ``findall()`` method is modified to treat angle
    brackets as non-capturing parentheses, in addition to matching the
    token boundaries; and to have ``'.'`` not match the angle brackets.
    c                 ó>   — dj                  d„ |D «       «      | _        y )NÚ c              3   ó,   K  — | ]  }d |z   dz   –— Œ y­w)ú<ú>Nr*   )r3   r4   s     r%   r6   z)TokenSearcher.__init__.<locals>.<genexpr>  s   è ø€ Ò:¨a˜C !™G c�MÑ:ùs   ‚)r^   Ú_raw)r5   r#   s     r%   rB   zTokenSearcher.__init__  s   € Ø—G‘GÑ:°6Ô:Ó:ˆ�	r'   c                 ó´  — t        j                  dd|«      }t        j                  dd|«      }t        j                  dd|«      }t        j                  dd|«      }t        j                  || j                  «      }|D ]0  }|j	                  d«      rŒ|j                  d«      sŒ't        d	«      ‚ |D �cg c]  }|d
d j                  d«      ‘Œ }}|S c c}w )a  
        Find instances of the regular expression in the text.
        The text is a list of tokens, and a regexp pattern to match
        a single token must be surrounded by angle brackets.  E.g.

        >>> from nltk.text import TokenSearcher
        >>> from nltk.book import text1, text5, text9
        >>> text5.findall("<.*><.*><bro>")
        you rule bro; telling you bro; u twizted bro
        >>> text1.findall("<a>(<.*>)<man>")
        monied; nervous; dangerous; white; white; white; pious; queer; good;
        mature; white; Cape; great; wise; wise; butterless; white; fiendish;
        pale; furious; better; certain; complete; dismasted; younger; brave;
        brave; brave; brave
        >>> text9.findall("<th.*>{3,}")
        thread through those; the thought that; that the thing; the thing
        that; that that thing; through these than through; them that the;
        through the thick; them that they; thought that the

        :param regexp: A regular expression
        :type regexp: str
        z\sr�   rŸ   z(?:<(?:r    z)>)z	(?<!\\)\.z[^>]z$Bad regexp for TokenSearcher.findallr   éÿÿÿÿz><)ÚreÚsubÚfindallr¡   Ú
startswithÚendswithr]   Úsplit©r5   ÚregexpÚhitsÚhs       r%   r¦   zTokenSearcher.findall  sÅ   € ô0 —‘˜˜r 6Ó*ˆÜ—‘˜˜i¨Ó0ˆÜ—‘˜˜e VÓ,ˆÜ—‘˜ f¨fÓ5ˆô �z‰z˜& $§)¡)Ó,ˆð ò 	IˆAØ—<‘< Õ$¨¯©°C­Ü Ð!GÓHÐHð	Ið
 .2Ö2¨��!�B�—‘˜dÕ#Ð2ˆÐ2Øˆùò 3s   Â6CN)rg   rh   ri   rj   rB   r¦   r*   r'   r%   r›   r›     s   „ ñò;ó'r'   r›   c                   óÈ   — e Zd ZdZdZdd„Zd„ Zd„ Zdd„Zdd„Z	dd	„Z
dd
„Zd„ Zd„ Zd„ Zdd„Zdd„Zd„ Zdd„Zdd„Zd„ Zd„ Zd„ Z ej0                  d«      Zd„ Zd„ Zd„ Zy) ÚTextaÛ  
    A wrapper around a sequence of simple (string) tokens, which is
    intended to support initial exploration of texts (via the
    interactive console).  Its methods perform a variety of analyses
    on the text's contexts (e.g., counting, concordancing, collocation
    discovery), and display the results.  If you wish to write a
    program which makes use of these analyses, then you should bypass
    the ``Text`` class, and use the appropriate analysis function or
    class directly instead.

    A ``Text`` is typically initialized from a given document or
    corpus.  E.g.:

    >>> import nltk.corpus
    >>> from nltk.text import Text
    >>> moby = Text(nltk.corpus.gutenberg.words('melville-moby_dick.txt'))

    TNc                 ó  — | j                   rt        |«      }|| _        |r|| _        yd|dd v r5|dd j	                  d«      }dj                  d„ |d| D «       «      | _        ydj                  d„ |dd D «       «      d	z   | _        y)
zv
        Create a Text object.

        :param tokens: The source text.
        :type tokens: sequence of str
        ú]Nrf   rX   c              3   ó2   K  — | ]  }t        |«      –— Œ y ­wr)   ©Ústr©r3   Útoks     r%   r6   z Text.__init__.<locals>.<genexpr>b  s   è ø€ Ò C¨c¤ S§Ñ Cùó   ‚r   c              3   ó2   K  — | ]  }t        |«      –— Œ y ­wr)   r³   rµ   s     r%   r6   z Text.__init__.<locals>.<genexpr>d  s   è ø€ Ò @¨c¤ S§Ñ @ùr·   é   z...)Ú_COPY_TOKENSrp   r#   Únamers   r^   )r5   r#   r»   Úends       r%   rB   zText.__init__S  sˆ   € ð ×ÒÜ˜&“\ˆFØˆŒáØˆD�IØ�F˜3˜B�KÑØ˜˜"�+×#Ñ# CÓ(ˆCØŸ™Ñ C°V¸A¸c°]Ô CÓCˆD�IàŸ™Ñ @°V¸B¸Q°ZÔ @Ó@À5ÑHˆD�Ir'   c                 ó    — | j                   |   S r)   )r#   )r5   r$   s     r%   Ú__getitem__zText.__getitem__j  s   € Ø�{‰{˜1‰~Ðr'   c                 ó,   — t        | j                  «      S r)   )r"   r#   rE   s    r%   Ú__len__zText.__len__m  s   € Ü�4—;‘;ÓÐr'   c                 ó’   — d| j                   vrt        | j                  d„ ¬«      | _        | j                  j	                  |||«      S )aÌ  
        Prints a concordance for ``word`` with the specified context window.
        Word matching is not case-sensitive.

        :param word: The target word or phrase (a list of strings)
        :type word: str or list
        :param width: The width of each line, in characters (default=80)
        :type width: int
        :param lines: The number of lines to display (default=25)
        :type lines: int

        :seealso: ``ConcordanceIndex``
        Ú_concordance_indexc                 ó"   — | j                  «       S r)   ©r!   ©Úss    r%   r-   z"Text.concordance.<locals>.<lambda>„  ó   € ¨1¯7©7«9€ r'   ©r@   )Ú__dict__rm   r#   rÂ   r—   ©r5   rJ   r…   r–   s       r%   ÚconcordancezText.concordancet  sD   € ð   t§}¡}Ñ4Ü&6Ø—‘Ñ!4ô'ˆDÔ#ð ×&Ñ&×8Ñ8¸¸uÀeÓLÐLr'   c                 ó–   — d| j                   vrt        | j                  d„ ¬«      | _        | j                  j	                  ||«      d| S )aÎ  
        Generate a concordance for ``word`` with the specified context window.
        Word matching is not case-sensitive.

        :param word: The target word or phrase (a list of strings)
        :type word: str or list
        :param width: The width of each line, in characters (default=80)
        :type width: int
        :param lines: The number of lines to display (default=25)
        :type lines: int

        :seealso: ``ConcordanceIndex``
        rÂ   c                 ó"   — | j                  «       S r)   rÄ   rÅ   s    r%   r-   z'Text.concordance_list.<locals>.<lambda>™  rÇ   r'   rÈ   N)rÉ   rm   r#   rÂ   r’   rÊ   s       r%   r‹   zText.concordance_list‰  sI   € ð   t§}¡}Ñ4Ü&6Ø—‘Ñ!4ô'ˆDÔ#ð ×&Ñ&×7Ñ7¸¸eÓDÀVÀeÐLÐLr'   c                 ó°  ‡— d| j                   v r| j                  |k(  r| j                  |k(  sž|| _        || _        ddlm} |j                  d«      Št        j                  | j                  |«      }|j                  d«       |j                  ˆfd„«       t        «       }t        |j                  |j                  |«      «      | _        | j                  S )aÚ  
        Return collocations derived from the text, ignoring stopwords.

            >>> from nltk.book import text4
            >>> text4.collocation_list()[:2]
            [('United', 'States'), ('fellow', 'citizens')]

        :param num: The maximum number of collocations to return.
        :type num: int
        :param window_size: The number of tokens spanned by a collocation (default=2)
        :type window_size: int
        :rtype: list(tuple(str, str))
        Ú_collocationsr   )Ú	stopwordsÚenglishr   c                 óH   •— t        | «      dk  xs | j                  «       ‰v S )Né   )r"   r!   )r4   Úignored_wordss    €r%   r-   z'Text.collocation_list.<locals>.<lambda>¹  s   ø€ ¬s°1«v¸©zÒ/W¸Q¿W¹W»YÈ-Ð=W€ r'   )rÉ   Ú_numÚ_window_sizeÚnltk.corpusrÐ   r_   r   Ú
from_wordsr#   Úapply_freq_filterÚapply_word_filterr   rp   ÚnbestÚlikelihood_ratiorÏ   )r5   ÚnumÚwindow_sizerÐ   ÚfinderÚbigram_measuresrÔ   s         @r%   Úcollocation_listzText.collocation_list�  s·   ø€ ð ˜tŸ}™}Ñ,Ø—	‘	˜SÒ Ø×!Ñ! [Ò0àˆDŒIØ +ˆDÔõ .à%ŸO™O¨IÓ6ˆMÜ,×7Ñ7¸¿¹À[ÓQˆFØ×$Ñ$ QÔ'Ø×$Ñ$Ó%WÔXÜ1Ó3ˆOÜ!%Ø—‘˜_×=Ñ=¸sÓCó"ˆDÔð ×!Ñ!Ð!r'   c                 ó�   — | j                  ||«      D ��cg c]  \  }}|dz   |z   ‘Œ }}}t        t        |d¬«      «       yc c}}w )aý  
        Print collocations derived from the text, ignoring stopwords.

            >>> from nltk.book import text4
            >>> text4.collocations() # doctest: +NORMALIZE_WHITESPACE
            United States; fellow citizens; years ago; four years; Federal
            Government; General Government; Vice President; American people; God
            bless; Chief Justice; one another; fellow Americans; Old World;
            Almighty God; Fellow citizens; Chief Magistrate; every citizen; Indian
            tribes; public debt; foreign nations


        :param num: The maximum number of collocations to print.
        :type num: int
        :param window_size: The number of tokens spanned by a collocation (default=2)
        :type window_size: int
        rX   ú; )Ú	separatorN)rá   r”   r   )r5   rÝ   rÞ   Úw1Úw2Úcollocation_stringss         r%   ÚcollocationszText.collocationsÀ  sO   € ð( )-×(=Ñ(=¸cÀ;Ó(O÷
Ù$˜b "ˆB�‰H�r‹Mð
Ðñ 
ô 	ŒiÐ+°tÔ<Õ=ùó
s   –Ac                 ó8   — | j                   j                  |«      S )zJ
        Count the number of times this word appears in the text.
        )r#   Úcountrv   s     r%   rê   z
Text.countØ  ó   € ð �{‰{× Ñ  Ó&Ð&r'   c                 ó8   — | j                   j                  |«      S )zQ
        Find the index of the first occurrence of the word in the text.
        )r#   rs   rv   s     r%   rs   z
Text.indexÞ  rë   r'   c                 ó   — t         ‚r)   )ÚNotImplementedError)r5   Úmethods     r%   ÚreadabilityzText.readabilityä  s   € ä!Ð!r'   c                 óÊ  ‡‡‡— d| j                   vrt        | j                  d„ d„ ¬«      | _        ‰j	                  «       Š| j                  j
                  Š‰‰j                  «       v rjt        ‰‰   «      Št        ˆˆˆfd„‰j                  «       D «       «      }|j                  |«      D ��cg c]  \  }}|‘Œ	 }}}t        t        |«      «       yt        d«       yc c}}w )a~  
        Distributional similarity: find other words which appear in the
        same contexts as the specified word; list most similar words first.

        :param word: The word used to seed the similarity search
        :type word: str
        :param num: The number of words to generate (default=20)
        :type num: int
        :seealso: ContextIndex.similar_words()
        Ú_word_context_indexc                 ó"   — | j                  «       S r)   )Úisalphar+   s    r%   r-   zText.similar.<locals>.<lambda>ö  s   € ¨a¯i©i«k€ r'   c                 ó"   — | j                  «       S r)   rÄ   rÅ   s    r%   r-   zText.similar.<locals>.<lambda>ö  s   € ÈÏÉË€ r'   )r?   r@   c              3   óH   •K  — | ]  }‰|   D ]  }|‰v r	|‰k(  s|–— Œ Œ y ­wr)   r*   )r3   r4   rU   ra   ÚwcirJ   s      €€€r%   r6   zText.similar.<locals>.<genexpr>ÿ  s?   øè ø€ ò àØ˜Q™òð Ø˜‘=¨¨dªô ðØñùs   ƒ"z
No matchesN)rÉ   r   r#   rò   r!   r<   Ú
conditionsrH   r   Úmost_commonr”   r   )	r5   rJ   rÝ   rc   r4   Ú_r_   ra   r÷   s	    `     @@r%   ÚsimilarzText.similarè  s¿   ú€ ð !¨¯©Ñ5ä'3Ø—‘Ñ$9Ñ?Rô(ˆDÔ$ð �z‰z‹|ˆØ×&Ñ&×8Ñ8ˆØ�3—>‘>Ó#Ñ#Ü˜3˜t™9“~ˆHÜõ àŸ™Ó)ôó ˆBð $&§>¡>°#Ó#6×7™4˜1˜a’QÐ7ˆEÑ7Ü”)˜EÓ"Õ#ä�,Õùó 8s   Â/Cc                 óz  — d| j                   vrt        | j                  d„ ¬«      | _        	 | j                  j	                  |d«      }|st        d«       y|j                  |«      D ��cg c]  \  }}|‘Œ	 }}}t        t        d„ |D «       «      «       yc c}}w # t        $ r}t        |«       Y d}~yd}~ww xY w)aY  
        Find contexts where the specified words appear; list
        most frequent common contexts first.

        :param words: The words used to seed the similarity search
        :type words: str
        :param num: The number of words to generate (default=20)
        :type num: int
        :seealso: ContextIndex.common_contexts()
        rò   c                 ó"   — | j                  «       S r)   rÄ   rÅ   s    r%   r-   z&Text.common_contexts.<locals>.<lambda>  rÇ   r'   rÈ   TzNo common contexts were foundc              3   ó2   K  — | ]  \  }}|d z   |z   –— Œ y­w)rú   Nr*   )r3   rå   ræ   s      r%   r6   z'Text.common_contexts.<locals>.<genexpr>!  s   è ø€ ÒL±&°"°b  S¡¨2¥ÑLùr·   N)	rÉ   r   r#   rò   rd   r”   rù   r   r]   )r5   r_   rÝ   rc   r4   rú   Úranked_contextsÚes           r%   rd   zText.common_contexts
  s¤   € ð !¨¯©Ñ5ä'3Ø—‘Ñ!4ô(ˆDÔ$ð		Ø×)Ñ)×9Ñ9¸%ÀÓFˆBÙÜÐ5Õ6à13·±ÀÓ1D×"E©¨¨A¢1Ð"E�Ñ"EÜ”iÑL¸OÔLÓLÕMùó #Føô ò 	Ü�!�H‰Hûð	ús/   ­)B ÁB Á+BÁ7B ÂB Â	B:Â%B5Â5B:c                 ó"   — ddl m}  || |«       y)zü
        Produce a plot showing the distribution of the words through the text.
        Requires pylab to be installed.

        :param words: The words to be plotted
        :type words: list(str)
        :seealso: nltk.draw.dispersion_plot()
        r   )Údispersion_plotN)Ú	nltk.drawr  )r5   r_   r  s      r%   r  zText.dispersion_plot&  s   € õ 	.á˜˜eÕ$r'   c                 ó`   — t        ||«      \  }}t        |¬«      }|j                  ||«       |S )N)Úorder)r
   r	   Úfit)r5   Útokenized_sentsrT   Ú
train_dataÚpadded_sentsÚmodels         r%   Ú_train_default_ngram_lmzText._train_default_ngram_lm3  s/   € Ü#<¸QÀÓ#PÑ ˆ
�LÜ˜!”ˆØ�	‰	�*˜lÔ+Øˆr'   c                 ó�  — t        dj                  | j                  «      «      D �cg c]  }|j                  d«      ‘Œ c}| _        t        | d«      s=t        dt        j                  ¬«       | j                  | j                  d¬«      | _
        g }|dkD  sJ d«       ‚t        |«      |k  rat        | j                  j                  |||¬	«      «      D ]#  \  }}|d
k(  rŒ|dk(  r n|j                  |«       Œ% |dz  }t        |«      |k  rŒa|rdj                  |«      dz   nd}|t        |d| «      z   }	t        |	«       |	S c c}w )a  
        Print random text, generated using a trigram language model.
        See also `help(nltk.lm)`.

        :param length: The length of text to generate (default=100)
        :type length: int

        :param text_seed: Generation can be conditioned on preceding context.
        :type text_seed: list(str)

        :param random_seed: A random seed or an instance of `random.Random`. If provided,
            makes the random sampling part of generation reproducible. (default=42)
        :type random_seed: int
        rX   Ú_trigram_modelzBuilding ngram index...)ÚfilerÓ   )rT   r   z!The `length` must be more than 0.)Ú	text_seedÚrandom_seedz<s>z</s>r   r�   N)r   r^   r#   r©   Ú_tokenized_sentsÚhasattrr”   ÚsysÚstderrr  r  r"   r;   Úgeneraterr   r   )
r5   Úlengthr  r  ÚsentÚgenerated_tokensÚidxÚtokenÚprefixÚ
output_strs
             r%   r  zText.generate9  sX  € ô" )6°c·h±h¸t¿{¹{Ó6KÓ(Lö!
Ø $ˆD�J‰J�s�Oò!
ˆÔô �tÐ-Ô.ÜÐ+´#·*±*Õ=Ø"&×">Ñ">Ø×%Ñ%¨ð #?ó #ˆDÔð Ðà˜ŠzÐ>Ð>Ó>ˆzÜÐ"Ó# fÒ,Ü'Ø×#Ñ#×,Ñ,Ø i¸[ð -ó óò 	/‘
��Uð
 ˜E’>ØØ˜F’?ÙØ ×'Ñ'¨Õ.ð	/ð ˜1ÑˆKô Ð"Ó# fÓ,ñ /8�—‘˜)Ó$ sÒ*¸RˆØœiÐ(8¸¸&Ð(AÓBÑBˆ
ÜˆjÔØÐùò9!
s   §Ec                 ó<   —  | j                  «       j                  |Ž S )zc
        See documentation for FreqDist.plot()
        :seealso: nltk.prob.FreqDist.plot()
        )ÚvocabÚplot)r5   Úargss     r%   r  z	Text.plotg  s   € ð
 !ˆt�z‰z‹|× Ñ  $Ð'Ð'r'   c                 óV   — d| j                   vrt        | «      | _        | j                  S )z.
        :seealso: nltk.prob.FreqDist
        Ú_vocab)rÉ   r   r"  rE   s    r%   r  z
Text.vocabn  s%   € ð ˜4Ÿ=™=Ñ(ä" 4›.ˆDŒKØ�{‰{Ðr'   c                 óæ   — d| j                   vrt        | «      | _        | j                  j                  |«      }|D �cg c]  }dj	                  |«      ‘Œ }}t        t        |d«      «       yc c}w )aÓ  
        Find instances of the regular expression in the text.
        The text is a list of tokens, and a regexp pattern to match
        a single token must be surrounded by angle brackets.  E.g.

        >>> from nltk.book import text1, text5, text9
        >>> text5.findall("<.*><.*><bro>")
        you rule bro; telling you bro; u twizted bro
        >>> text1.findall("<a>(<.*>)<man>")
        monied; nervous; dangerous; white; white; white; pious; queer; good;
        mature; white; Cape; great; wise; wise; butterless; white; fiendish;
        pale; furious; better; certain; complete; dismasted; younger; brave;
        brave; brave; brave
        >>> text9.findall("<th.*>{3,}")
        thread through those; the thought that; that the thing; the thing
        that; that that thing; through these than through; them that the;
        through the thick; them that they; thought that the

        :param regexp: A regular expression
        :type regexp: str
        Ú_token_searcherrX   rã   N)rÉ   r›   r$  r¦   r^   r”   r   rª   s       r%   r¦   zText.findallw  sb   € ð.  D§M¡MÑ1Ü#0°Ó#6ˆDÔ à×#Ñ#×+Ñ+¨FÓ3ˆØ%)Ö* �—‘˜•Ð*ˆÐ*ÜŒi˜˜dÓ#Õ$ùò +s   ¾A.z\w+|[\.\!\?]c                 ó´  — |dz
  }|dk\  rG| j                   j                  ||   «      s)|dz  }|dk\  r| j                   j                  ||   «      sŒ)|dk7  r||   nd}|dz   }|t        |«      k  rP| j                   j                  ||   «      s2|dz  }|t        |«      k  r| j                   j                  ||   «      sŒ2|t        |«      k7  r||   nd}||fS )zÙ
        One left & one right token, both case-normalized.  Skip over
        non-sentence-final punctuation.  Used by the ``ContextIndex``
        that is created for ``similar()`` and ``common_contexts()``.
        r   r   r   r    )Ú_CONTEXT_REÚmatchr"   )r5   r#   r$   Újr   r   s         r%   Ú_contextzText._context›  sß   € ð �‰EˆØ�1Šf˜T×-Ñ-×3Ñ3°F¸1±IÔ>Ø�‰FˆAð �1Šf˜T×-Ñ-×3Ñ3°F¸1±IÕ>à šFˆv�aŠy¨	ˆð �‰EˆØ”#�f“+Šo d×&6Ñ&6×&<Ñ&<¸VÀA¹YÔ&GØ�‰FˆAð ”#�f“+Šo d×&6Ñ&6×&<Ñ&<¸VÀA¹YÕ&Gà¤# f£+Ò-��q’	°7ˆà�eˆ}Ðr'   c                 ó    — d| j                   z  S ©Nz
<Text: %s>©r»   rE   s    r%   Ú__str__zText.__str__³  ó   € Ø˜dŸi™iÑ'Ð'r'   c                 ó    — d| j                   z  S r+  r,  rE   s    r%   ry   zText.__repr__¶  r.  r'   r)   )éO   r™   )rf   r   re   )rÓ   )éd   Né*   )rg   rh   ri   rj   rº   rB   r¾   rÀ   rË   r‹   rá   rè   rê   rs   rð   rû   rd   r  r  r  r  r  r¦   r¤   Úcompiler&  r)  r-  ry   r*   r'   r%   r¯   r¯   9  s�   „ ñð. €LóIò.ò óMó*Mó(!"óF>ò0'ò'ò"ó  óDò8%óó,ò\(òò%ðD �"—*‘*˜_Ó-€Kòò0(ó(r'   r¯   c                   ó(   — e Zd ZdZd„ Zd„ Zd„ Zd„ Zy)ÚTextCollectiona;  A collection of texts, which can be loaded with list of texts, or
    with a corpus consisting of one or more texts, and which supports
    counting, concordancing, collocation discovery, etc.  Initialize a
    TextCollection as follows:

    >>> import nltk.corpus
    >>> from nltk.text import TextCollection
    >>> from nltk.book import text1, text2, text3
    >>> gutenberg = TextCollection(nltk.corpus.gutenberg)
    >>> mytexts = TextCollection([text1, text2, text3])

    Iterating over a TextCollection produces all the tokens of all the
    texts in order.
    c                 óØ   — t        |d«      r,|j                  «       D �cg c]  }|j                  |«      ‘Œ }}|| _        t        j                  | t        |«      «       i | _        y c c}w )Nr_   )r  Úfileidsr_   Ú_textsr¯   rB   r   Ú
_idf_cache)r5   ÚsourceÚfs      r%   rB   zTextCollection.__init__Ë  sV   € Ü�6˜7Ô#Ø/5¯~©~Ó/?Ö@¨!�f—l‘l 1•oÐ@ˆFÐ@àˆŒÜ�‰�dÔ-¨fÓ5Ô6Øˆ�ùò	 As   ŸA'c                 ó<   — |j                  |«      t        |«      z  S )z"The frequency of the term in text.)rê   r"   ©r5   ÚtermÚtexts      r%   ÚtfzTextCollection.tfÓ  s   € à�z‰z˜$Ó¤# d£)Ñ+Ð+r'   c                 óH  — | j                   j                  |«      }|€t        | j                  D �cg c]	  }||v sŒd‘Œ c}«      }t        | j                  «      dk(  rt	        d«      ‚|r!t        t        | j                  «      |z  «      nd}|| j                   |<   |S c c}w )z¤The number of texts in the corpus divided by the
        number of texts that the term appears in.
        If a term does not appear in the corpus, 0.0 is returned.Tr   z+IDF undefined for empty document collectiong        )r9  rS   r"   r8  r]   r   )r5   r>  Úidfr?  Úmatchess        r%   rB  zTextCollection.idf×  sŽ   € ð
 �o‰o×!Ñ! $Ó'ˆØˆ;Ü¨D¯K©KÖH D¸4À4º<š4ÒHÓIˆGÜ�4—;‘;Ó 1Ò$Ü Ð!NÓOÐOÙ5<”#”c˜$Ÿ+™+Ó&¨Ñ0Ô1À#ˆCØ$'ˆD�O‰O˜DÑ!Øˆ
ùò Is
   ±	B»Bc                 óJ   — | j                  ||«      | j                  |«      z  S r)   )r@  rB  r=  s      r%   Útf_idfzTextCollection.tf_idfå  s    € Ø�w‰w�t˜TÓ" T§X¡X¨d£^Ñ3Ð3r'   N)rg   rh   ri   rj   rB   r@  rB  rE  r*   r'   r%   r5  r5  »  s   „ ñòò,òó4r'   r5  c                  óz  — ddl m}  t        | j                  d¬«      «      }t	        |«       t	        «        t	        d«       |j                  d«       t	        «        t	        d«       |j                  d«       t	        «        t	        d«       |j                  «        t	        «        t	        d«       |j                  g d	¢«       t	        «        t	        d
«       |j                  d«       t	        «        t	        d«       t	        d|d   «       t	        d|dd «       t	        d|j                  «       d   «       y )Nr   )ÚbrownÚnews)Ú
categorieszConcordance:zDistributionally similar words:zCollocations:zDispersion plot:)rH  ÚreportÚsaidÚ	announcedzVocabulary plot:é2   z	Indexing:ztext[3]:rÓ   z
text[3:5]:é   ztext.vocab()['news']:)r×   rG  r¯   r_   r”   rË   rû   rè   r  r  r  )rG  r?  s     r%   ÚdemorO  é  sè   € Ý!ä�—‘ v�Ó.Ó/€DÜ	ˆ$„KÜ	„GÜ	ˆ.ÔØ×Ñ�VÔÜ	„GÜ	Ð
+Ô,Ø‡L�L�ÔÜ	„GÜ	ˆ/ÔØ×ÑÔÜ	„Gô 
Ð
ÔØ×ÑÒ@ÔAÜ	„GÜ	Ð
ÔØ‡I�Iˆb„MÜ	„GÜ	ˆ+ÔÜ	ˆ*�d˜1‘gÔÜ	ˆ,˜˜Q˜q˜	Ô"Ü	Ð
! 4§:¡:£<°Ñ#7Õ8r'   Ú__main__)r   rm   r›   r¯   r5  )(rj   r¤   r  r|   Úcollectionsr   r   r   Ú	functoolsr   Úmathr   Únltk.collocationsr   Únltk.lmr	   Únltk.lm.preprocessingr
   Únltk.metricsr   r   Únltk.probabilityr   r:   r   Únltk.tokenizer   Ú	nltk.utilr   r   r   r   r   rm   r›   r¯   r5  rO  rg   Ú__all__r*   r'   r%   ú<module>r\     s¦   ðñó 
Û 
Û ß 8Ñ 8Ý Ý å 5Ý Ý ;ß 7Ý 7Ý %Ý 'ß >Ñ >áØÚMó€÷Xñ X÷v|-ñ |-÷~5ñ 5÷p~(ñ ~(ôD+4�Tô +4ò\9ð< ˆzÒÙ„Fò�r'   