Ë
    çÍ:jÉ(  ã                   ón   — d Z ddlmZ ddlmZ ddlmZmZmZm	Z	  G d„ de«      Z
d„ Zedk(  r e«        y	y	)
aê  
A classifier based on the Naive Bayes algorithm.  In order to find the
probability for a label, this algorithm first uses the Bayes rule to
express P(label|features) in terms of P(label) and P(features|label):

|                       P(label) * P(features|label)
|  P(label|features) = ------------------------------
|                              P(features)

The algorithm then makes the 'naive' assumption that all features are
independent, given the label:

|                       P(label) * P(f1|label) * ... * P(fn|label)
|  P(label|features) = --------------------------------------------
|                                         P(features)

Rather than computing P(features) explicitly, the algorithm just
calculates the numerator for each label, and normalizes them so they
sum to one:

|                       P(label) * P(f1|label) * ... * P(fn|label)
|  P(label|features) = --------------------------------------------
|                        SUM[l]( P(l) * P(f1|l) * ... * P(fn|l) )
é    )Údefaultdict)ÚClassifierI)ÚDictionaryProbDistÚELEProbDistÚFreqDistÚsum_logsc                   óL   — e Zd ZdZd„ Zd„ Zd„ Zd„ Zd
d„Zdd„Z	e
efd„«       Zy	)ÚNaiveBayesClassifiera  
    A Naive Bayes classifier.  Naive Bayes classifiers are
    paramaterized by two probability distributions:

      - P(label) gives the probability that an input will receive each
        label, given no information about the input's features.

      - P(fname=fval|label) gives the probability that a given feature
        (fname) will receive a given value (fval), given that the
        label (label).

    If the classifier encounters an input with a feature that has
    never been seen with any label, then rather than assigning a
    probability of 0 to all labels, it will ignore that feature.

    The feature value 'None' is reserved for unseen feature values;
    you generally should not use 'None' as a feature value for one of
    your own features.
    c                 ó\   — || _         || _        t        |j                  «       «      | _        y)a=  
        :param label_probdist: P(label), the probability distribution
            over labels.  It is expressed as a ``ProbDistI`` whose
            samples are labels.  I.e., P(label) =
            ``label_probdist.prob(label)``.

        :param feature_probdist: P(fname=fval|label), the probability
            distribution for feature values, given labels.  It is
            expressed as a dictionary whose keys are ``(label, fname)``
            pairs and whose values are ``ProbDistI`` objects over feature
            values.  I.e., P(fname=fval|label) =
            ``feature_probdist[label,fname].prob(fval)``.  If a given
            ``(label,fname)`` is not a key in ``feature_probdist``, then
            it is assumed that the corresponding P(fname=fval|label)
            is 0 for all values of ``fval``.
        N)Ú_label_probdistÚ_feature_probdistÚlistÚsamplesÚ_labels)ÚselfÚlabel_probdistÚfeature_probdists      úm/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/classify/naivebayes.pyÚ__init__zNaiveBayesClassifier.__init__@   s)   € ð"  .ˆÔØ!1ˆÔÜ˜N×2Ñ2Ó4Ó5ˆ�ó    c                 ó   — | j                   S ©N)r   )r   s    r   ÚlabelszNaiveBayesClassifier.labelsU   s   € Ø�|‰|Ðr   c                 ó@   — | j                  |«      j                  «       S r   )Úprob_classifyÚmax)r   Ú
featuresets     r   ÚclassifyzNaiveBayesClassifier.classifyX   s   € Ø×!Ñ! *Ó-×1Ñ1Ó3Ð3r   c                 ó"  — |j                  «       }t        |j                  «       «      D ](  }| j                  D ]  }||f| j                  v sŒ Œ% ||= Œ* i }| j                  D ]   }| j
                  j                  |«      ||<   Œ" | j                  D ]n  }|j                  «       D ]Y  \  }}||f| j                  v r.| j                  ||f   }||xx   |j                  |«      z  cc<   ŒD||xx   t        g «      z  cc<   Œ[ Œp t        |dd¬«      S )NT)Ú	normalizeÚlog)
Úcopyr   Úkeysr   r   r   ÚlogprobÚitemsr   r   )r   r   ÚfnameÚlabelr$   ÚfvalÚfeature_probss          r   r   z"NaiveBayesClassifier.prob_classify[   s(  € ð  —_‘_Ó&ˆ
Ü˜*Ÿ/™/Ó+Ó,ò 	&ˆEØŸ™ò &�Ø˜5�> T×%;Ñ%;Ò;Ùð&ð
 ˜uÑ%ð	&ð ˆØ—\‘\ò 	AˆEØ!×1Ñ1×9Ñ9¸%Ó@ˆG�EŠNð	Að —\‘\ò 		3ˆEØ)×/Ñ/Ó1ò 3‘��tØ˜5�> T×%;Ñ%;Ñ;Ø$(×$:Ñ$:¸5À%¸<Ñ$H�MØ˜E“N m×&;Ñ&;¸DÓ&AÑA”Nð
 ˜E“N¤h¨r£lÑ2”Nñ3ð		3ô " '°T¸tÔDÐDr   c                 óä  ‡‡‡‡	— | j                   Št        d«       | j                  |«      D ]Á  \  ŠŠˆˆˆfd„Š	t        ˆˆˆfd„| j                  D «       ˆ	fd„d¬«      }t        |«      dk(  rŒB|d   }|d	   }‰|‰f   j                  ‰«      dk(  rd
}n0d‰|‰f   j                  ‰«      ‰|‰f   j                  ‰«      z  z  }t        ‰d›d‰d›dd|z  d d d›dd|z  d d d›d|›d�
«       ŒÃ y )NzMost Informative Featuresc                 ó0   •— ‰| ‰f   j                  ‰«      S r   )Úprob)ÚlÚcpdistr&   r(   s    €€€r   Ú	labelprobzFNaiveBayesClassifier.show_most_informative_features.<locals>.labelprobƒ   s   ø€ Ø˜a ˜hÑ'×,Ñ,¨TÓ2Ð2r   c              3   óR   •K  — | ]  }‰‰|‰f   j                  «       v sŒ|–— Œ  y ­wr   )r   )Ú.0r-   r.   r&   r(   s     €€€r   ú	<genexpr>zFNaiveBayesClassifier.show_most_informative_features.<locals>.<genexpr>‡   s*   øè ø€ ÒO�q¨D°F¸1¸e¸8Ñ4D×4LÑ4LÓ4NÒ,N”ÑOùs   ƒ' 'c                 ó   •—  ‰| «       | fS r   © )Úelementr/   s    €r   ú<lambda>zENaiveBayesClassifier.show_most_informative_features.<locals>.<lambda>ˆ   s   ø€ ¡i°Ó&8Ð%8¸'Ð$B€ r   T)ÚkeyÚreverseé   r   éÿÿÿÿÚINFz%8.1fz>24z = Ú14ú z%sé   z>6z : Ú6z : 1.0)r   ÚprintÚmost_informative_featuresÚsortedr   Úlenr,   )
r   Únr   Úl0Úl1Úratior.   r&   r(   r/   s
         @@@@r   Úshow_most_informative_featuresz3NaiveBayesClassifier.show_most_informative_features|   s  û€ à×'Ñ'ˆÜÐ)Ô*à×9Ñ9¸!Ó<ò 	‰KˆE�4ö3ô ÝO˜DŸL™LÔOÛBØôˆFô
 �6‹{˜aÒØØ˜‘ˆBØ˜‘ˆBØ�b˜%�iÑ ×%Ñ% dÓ+¨qÒ0Ø‘àØ˜2˜u˜9Ñ%×*Ñ*¨4Ó0°6¸"¸e¸)Ñ3D×3IÑ3IÈ$Ó3OÑOñ�ô ã›$ ¨¡¨B¨Q£°$¸±)¸R¸a³Â%ðIõñ)	r   c                 ó  ‡	‡
— t        | d«      r| j                  d| S t        «       }t        t        «      Š	t        d„ «      Š
| j
                  j                  «       D ]�  \  \  }}}|j                  «       D ]f  }||f}|j                  |«       |j                  |«      }t        |‰	|   «      ‰	|<   t        |‰
|   «      ‰
|<   ‰
|   dk(  sŒV|j                  |«       Œh Œƒ t        |ˆ	ˆ
fd„¬«      | _        | j                  d| S )a—  
        Return a list of the 'most informative' features used by this
        classifier.  For the purpose of this function, the
        informativeness of a feature ``(fname,fval)`` is equal to the
        highest value of P(fname=fval|label), for any label, divided by
        the lowest value of P(fname=fval|label), for any label:

        |  max[ P(fname=fval|label1) / P(fname=fval|label2) ]
        Ú_most_informative_featuresNc                   ó   — y)Ng      ð?r4   r4   r   r   r6   z@NaiveBayesClassifier.most_informative_features.<locals>.<lambda>¬   s   � r   r   c                 óf   •— ‰|    ‰|    z  | d   | d   dv t        | d   «      j                  «       fS )Nr   r9   )NFT)ÚstrÚlower)Úfeature_ÚmaxprobÚminprobs    €€r   r6   z@NaiveBayesClassifier.most_informative_features.<locals>.<lambda>¼   sE   ø€ Ø˜HÑ%¨°Ñ(9Ñ9Ø˜Q‘KØ˜Q‘KÐ#6Ð6Ü˜ ™Ó$×*Ñ*Ó,ð	&€ r   )r7   )ÚhasattrrJ   Úsetr   Úfloatr   r%   r   Úaddr,   r   ÚminÚdiscardrB   )r   rD   Úfeaturesr'   r&   Úprobdistr(   ÚfeatureÚprP   rQ   s            @@r   rA   z.NaiveBayesClassifier.most_informative_featuresš   s  ù€ ô �4Ð5Ô6Ø×2Ñ2°2°AÐ6Ð6ô “uˆHô "¤%Ó(ˆGÜ!¡+Ó.ˆGà,0×,BÑ,B×,HÑ,HÓ,Jò 2Ñ(‘�˜ Ø$×,Ñ,Ó.ò 2�DØ$ d˜m�GØ—L‘L Ô)Ø Ÿ™ dÓ+�AÜ'*¨1¨g°gÑ.>Ó'?�G˜GÑ$Ü'*¨1¨g°gÑ.>Ó'?�G˜GÑ$Ø˜wÑ'¨1Ó,Ø ×(Ñ(¨Õ1ñ2ð2ô /5Øôô/ˆDÔ+ð ×.Ñ.¨r°Ð2Ð2r   c                 ó|  — t        «       }t        t         «      }t        t        «      }t        «       }|D ]a  \  }}||xx   dz  cc<   |j                  «       D ]<  \  }	}
|||	f   |
xx   dz  cc<   ||	   j	                  |
«       |j	                  |	«       Œ> Œc |D ]U  }||   }|D ]I  }	|||	f   j                  «       }||z
  dkD  sŒ!|||	f   dxx   ||z
  z  cc<   ||	   j	                  d«       ŒK ŒW  ||«      }i }|j                  «       D ]%  \  \  }}	} ||t        ||	   «      ¬«      }||||	f<   Œ'  | ||«      S )z‹
        :param labeled_featuresets: A list of classified featuresets,
            i.e., a list of tuples ``(featureset, label)``.
        r9   r   N)Úbins)r   r   rS   r%   rU   ÚNrC   )ÚclsÚlabeled_featuresetsÚ	estimatorÚlabel_freqdistÚfeature_freqdistÚfeature_valuesÚfnamesr   r'   r&   r(   Únum_samplesÚcountr   r   ÚfreqdistrY   s                    r   ÚtrainzNaiveBayesClassifier.trainÅ   sŒ  € ô "›ˆÜ&¤xÓ0ÐÜ$¤SÓ)ˆÜ“ˆð "5ò 	"ÑˆJ˜Ø˜5Ó! QÑ&Ó!Ø)×/Ñ/Ó1ò "‘��tà  ¨ Ñ.¨tÓ4¸Ñ9Ó4à˜uÑ%×)Ñ)¨$Ô/à—
‘
˜5Õ!ñ"ð	"ð $ò 	4ˆEØ(¨Ñ/ˆKØò 4�Ø(¨°¨Ñ6×8Ñ8Ó:�ð  Ñ&¨Ó*Ø$ U¨E \Ñ2°4Ó8¸KÈ%Ñ<OÑOÓ8Ø" 5Ñ)×-Ñ-¨dÕ3ñ4ð	4ñ # >Ó2ˆð ÐØ(8×(>Ñ(>Ó(@ò 	6Ñ$‰NˆU�E˜HÙ  ´°NÀ5Ñ4IÓ0JÔKˆHØ-5Ð˜U E˜\Ò*ð	6ñ �>Ð#3Ó4Ð4r   N)é
   )éd   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r   r   rH   rA   Úclassmethodr   ri   r4   r   r   r
   r
   +   s?   „ ñò(6ò*ò4òEóBó<)3ðV Ø2=ò .5ó ñ.5r   r
   c                  ó\   — ddl m}   | t        j                  «      }|j	                  «        y )Nr   )Ú
names_demo)Únltk.classify.utilrr   r
   ri   rH   )rr   Ú
classifiers     r   Údemoru   ü   s"   € Ý-áÔ0×6Ñ6Ó7€JØ×-Ñ-Õ/r   Ú__main__N)ro   Úcollectionsr   Únltk.classify.apir   Únltk.probabilityr   r   r   r   r
   ru   rl   r4   r   r   ú<module>rz      s@   ðñõ2 $å )ß PÓ PôI5˜;ô I5òb0ð ˆzÒÙ…Fð r   