Ë
    çÍ:jï   ã                   ó°   — d Z ddlZddlZddlZddlmZmZ ddlmZ ddlm	Z	 ddl
mZ ddlmZ ddlmZ  G d	„ d
e¬«      Zd„ Zd„ Zdd„Z G d„ de¬«      Zy)zLanguage Model Interface.é    N)ÚABCMetaÚabstractmethod)Úbisect)Ú
accumulate)ÚNgramCounter)Ú	log_base2)Ú
Vocabularyc                   ó6   — e Zd ZdZd„ Zed„ «       Zed„ «       Zy)Ú	SmoothingzìNgram Smoothing Interface

    Implements Chen & Goodman 1995's idea that all smoothing algorithms have
    certain features in common. This should ideally allow smoothing algorithms to
    work both with Backoff and Interpolation.
    c                 ó    — || _         || _        y)zä
        :param vocabulary: The Ngram vocabulary object.
        :type vocabulary: nltk.lm.vocab.Vocabulary
        :param counter: The counts of the vocabulary items.
        :type counter: nltk.lm.counter.NgramCounter
        N)ÚvocabÚcounts)ÚselfÚ
vocabularyÚcounters      ú`/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/lm/api.pyÚ__init__zSmoothing.__init__   s   € ð  ˆŒ
Øˆ�ó    c                 ó   — t        «       ‚©N©ÚNotImplementedError)r   Úwords     r   Úunigram_scorezSmoothing.unigram_score'   ó   € ä!Ó#Ð#r   c                 ó   — t        «       ‚r   r   ©r   r   Úcontexts      r   Úalpha_gammazSmoothing.alpha_gamma+   r   r   N)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r   r   © r   r   r   r      s4   „ ñòð ñ$ó ð$ð ñ$ó ñ$r   r   )Ú	metaclassc                 óD   — t        j                  | «      t        | «      z  S )z0Return average (aka mean) for sequence of items.)ÚmathÚfsumÚlen)Úitemss    r   Ú_meanr+   0   s   € ä�9‰9�UÓœc %›jÑ(Ð(r   c                 ód   — t        | t        j                  «      r| S t        j                  | «      S r   )Ú
isinstanceÚrandomÚRandom)Úseed_or_generators    r   Ú_random_generatorr1   5   s'   € ÜÐ#¤V§]¡]Ô3Ø Ð Ü�=‰=Ð*Ó+Ð+r   c                 óö   — | st        d«      ‚t        | «      t        |«      k7  rt        d«      ‚t        t        |«      «      }t	        j
                  |«      }|j                  «       }| t        |||z  «         S )z`Like random.choice, but with weights.

    Heavily inspired by python 3.6 `random.choices`.
    z"Can't choose from empty populationz3The number of weights does not match the population)Ú
ValueErrorr)   Úlistr   r'   r(   r.   r   )Ú
populationÚweightsÚrandom_generatorÚcum_weightsÚtotalÚ	thresholds         r   Ú_weighted_choicer;   ;   sq   € ñ
 ÜÐ=Ó>Ð>Ü
ˆ:ƒœ#˜g›,Ò&ÜÐNÓOÐOÜ”z 'Ó*Ó+€KÜ�I‰I�gÓ€EØ ×'Ñ'Ó)€IØ”f˜[¨%°)Ñ*;Ó<Ñ=Ð=r   c                   ó\   — e Zd ZdZdd„Zdd„Zdd„Zedd„«       Zdd„Z	d„ Z
d	„ Zd
„ Zdd„Zy)ÚLanguageModelzKABC for Language Models.

    Cannot be directly instantiated itself.

    Nc                 óì   — || _         |r?t        |t        «      s/t        j                  d| j
                  j                  ›d�d¬«       |€
t        «       n|| _        |€t        «       | _	        y|| _	        y)ap  Creates new LanguageModel.

        :param vocabulary: If provided, this vocabulary will be used instead
            of creating a new one when training.
        :type vocabulary: `nltk.lm.Vocabulary` or None
        :param counter: If provided, use this object to count ngrams.
        :type counter: `nltk.lm.NgramCounter` or None
        :param ngrams_fn: If given, defines how sentences in training text are turned to ngram
            sequences.
        :type ngrams_fn: function or None
        :param pad_fn: If given, defines how sentences in training text are padded.
        :type pad_fn: function or None
        z$The `vocabulary` argument passed to z- must be an instance of `nltk.lm.Vocabulary`.é   )Ú
stacklevelN)
Úorderr-   r	   ÚwarningsÚwarnÚ	__class__r    r   r   r   )r   rA   r   r   s       r   r   zLanguageModel.__init__Q   sh   € ð ˆŒ
Ùœj¨´ZÔ@Ü�M‰MØ6°t·~±~×7NÑ7NÐ6Qð R?ð ?àõð
 &0Ð%7”Z”\¸ZˆŒ
Ø(/¨”l“nˆ�¸Wˆ�r   c                 ó¶   ‡ — ‰ j                   s(|€t        d«      ‚‰ j                   j                  |«       ‰ j                  j                  ˆ fd„|D «       «       y)zeTrains the model on a text.

        :param text: Training text as a sequence of sentences.

        Nz:Cannot fit without a vocabulary or text to create it from.c              3   óT   •K  — | ]  }‰j                   j                  |«      –— Œ! y ­wr   )r   Úlookup)Ú.0Úsentr   s     €r   ú	<genexpr>z$LanguageModel.fit.<locals>.<genexpr>u   s    øè ø€ ÒD°t˜4Ÿ:™:×,Ñ,¨T×2ÑDùs   ƒ%()r   r3   Úupdater   )r   ÚtextÚvocabulary_texts   `  r   ÚfitzLanguageModel.fiti   sN   ø€ ð �zŠzØÐ&Ü ØPóð ð �J‰J×Ñ˜oÔ.Ø�‰×ÑÓD¸tÔDÕDr   c                 óš   — | j                  | j                  j                  |«      |r| j                  j                  |«      «      S d«      S )z©Masks out of vocab (OOV) words and computes their model score.

        For model-specific logic of calculating scores, see the `unmasked_score`
        method.
        N)Úunmasked_scorer   rG   r   s      r   ÚscorezLanguageModel.scorew   sI   € ð ×"Ñ"Ø�J‰J×Ñ˜dÓ#Á7 T§Z¡Z×%6Ñ%6°wÓ%?ó
ð 	
ØPTó
ð 	
r   c                 ó   — t        «       ‚)aÒ  Score a word given some optional context.

        Concrete models are expected to provide an implementation.
        Note that this method does not mask its arguments with the OOV label.
        Use the `score` method for that.

        :param str word: Word for which we want the score
        :param tuple(str) context: Context the word is in.
            If `None`, compute unigram score.
        :param context: tuple(str) or None
        :rtype: float
        r   r   s      r   rP   zLanguageModel.unmasked_score�   s   € ô "Ó#Ð#r   c                 ó8   — t        | j                  ||«      «      S )z‡Evaluate the log score of this word in this context.

        The arguments are the same as for `score` and `unmasked_score`.

        )r   rQ   r   s      r   ÚlogscorezLanguageModel.logscore‘   s   € ô ˜Ÿ™ D¨'Ó2Ó3Ð3r   c                 ón   — |r| j                   t        |«      dz      |   S | j                   j                  S )z²Helper method for retrieving counts for a given context.

        Assumes context has been checked and oov words in it masked.
        :type context: tuple(str) or None

        é   )r   r)   Úunigrams)r   r   s     r   Úcontext_countszLanguageModel.context_counts™   s7   € ñ 7>ˆD�K‰Kœ˜G› qÑ(Ñ)¨'Ñ2ð	
ØCGÇ;Á;×CWÑCWð	
r   c                 óp   — dt        |D �cg c]  }| j                  |d   |dd «      ‘Œ c}«      z  S c c}w )a?  Calculate cross-entropy of model for given evaluation text.

        This implementation is based on the Shannon-McMillan-Breiman theorem,
        as used and referenced by Dan Jurafsky and Jordan Boyd-Graber.

        :param Iterable(tuple(str)) text_ngrams: A sequence of ngram tuples.
        :rtype: float

        éÿÿÿÿN)r+   rT   )r   Útext_ngramsÚngrams      r   ÚentropyzLanguageModel.entropy¤   s>   € ð ”EØ?JÖK°eˆT�]‰]˜5 ™9 e¨C¨R jÕ1ÒKó
ñ 
ð 	
ùÚKs   ‹3
c                 ó8   — t        d| j                  |«      «      S )zŽCalculates the perplexity of the given text.

        This is simply 2 ** cross-entropy for the text, so the arguments are the same.

        g       @)Úpowr]   )r   r[   s     r   Ú
perplexityzLanguageModel.perplexity²   s   € ô �3˜Ÿ™ [Ó1Ó2Ð2r   c                 óL  ‡ ‡— |€g n
t        |«      }t        |«      }|dk(  rÊt        |«      ‰ j                  k\  r|‰ j                   dz   d n|Š‰ j	                  ‰ j
                  j                  ‰«      «      }‰rF|sDt        ‰«      dkD  r‰dd ng Š‰ j	                  ‰ j
                  j                  ‰«      «      }‰r|sŒDt        |«      }t        |t        ˆˆ fd„|D «       «      |«      S g }t        |«      D ](  }|j                  ‰ j                  d||z   |¬«      «       Œ* |S )aà  Generate words from the model.

        :param int num_words: How many words to generate. By default 1.
        :param text_seed: Generation can be conditioned on preceding context.
        :param random_seed: A random seed or an instance of `random.Random`. If provided,
            makes the random sampling part of generation reproducible.
        :return: One (str) word or a list of words generated from model.

        Examples:

        >>> from nltk.lm import MLE
        >>> lm = MLE(2)
        >>> lm.fit([[("a", "b"), ("b", "c")]], vocabulary_text=['a', 'b', 'c'])
        >>> lm.fit([[("a",), ("b",), ("c",)]])
        >>> lm.generate(random_seed=3)
        'a'
        >>> lm.generate(text_seed=['a'])
        'b'

        NrV   c              3   óB   •K  — | ]  }‰j                  |‰«      –— Œ y ­wr   )rQ   )rH   Úwr   r   s     €€r   rJ   z)LanguageModel.generate.<locals>.<genexpr>â   s   øè ø€ Ò>°�d—j‘j  G×,Ñ>ùs   ƒ)Ú	num_wordsÚ	text_seedÚrandom_seed)r4   r1   r)   rA   rX   r   rG   Úsortedr;   ÚtupleÚrangeÚappendÚgenerate)	r   rd   re   rf   r7   ÚsamplesÚ	generatedÚ_r   s	   `       @r   rk   zLanguageModel.generateº   s8  ù€ ð* $Ð+‘B´°i³ˆ	Ü,¨[Ó9Ðà˜Š>ô �y“> T§Z¡ZÒ/ð ˜4Ÿ:™:˜+¨™/Ð+Ñ,àð ð
 ×)Ñ)¨$¯*©*×*;Ñ*;¸GÓ*DÓEˆGÙ¡'Ü),¨W«¸Ò)9˜' ! "™+¸r�Ø×-Ñ-¨d¯j©j×.?Ñ.?ÀÓ.HÓI�ñ ¢'ô ˜W“oˆGÜ#ØÜÔ>°gÔ>Ó>Ø óð ð ˆ	Ü�yÓ!ò 	ˆAØ×ÑØ—‘ØØ'¨)Ñ3Ø 0ð ó õð	ð Ðr   )NNr   )rV   NN)r    r!   r"   r#   r   rN   rQ   r   rP   rT   rX   r]   r`   rk   r$   r   r   r=   r=   J   sE   „ ñóEó0Eó
ð ò$ó ð$ó4ò	
ò
ò3ô5r   r=   r   )r#   r'   r.   rB   Úabcr   r   r   Ú	itertoolsr   Únltk.lm.counterr   Únltk.lm.utilr   Únltk.lm.vocabularyr	   r   r+   r1   r;   r=   r$   r   r   ú<module>rt      sN   ðñ  ã Û Û ß 'Ý Ý  å (Ý "Ý )ô$˜'õ $ò6)ò
,ó>ôe˜gö er   