Ë
    çÍ:jÿî  ã                   óÐ  — d Z 	 ddlZddlZddlZddlmZ ddlmZ ddl	m
Z
mZmZ ddlmZmZmZ ddlmZmZmZ ddlmZ dd	lmZ dd
lmZ dZ G d„ de«      ZeZ G d„ d«      Z G d„ de«      Z G d„ de«      Z  G d„ de «      Z! G d„ de «      Z" G d„ de«      Z#	 d(d„Z$d„ Z%d„ Z&	 d(d„Z'd„ Z(d„ Z)	 d)d „Z* G d!„ d"e«      Z+d#„ Z,d*d$„Z-d%„ Z.d&„ Z/e0d'k(  r e/«        yy# e$ r Y ŒÜw xY w)+aä  
A classifier model based on maximum entropy modeling framework.  This
framework considers all of the probability distributions that are
empirically consistent with the training data; and chooses the
distribution with the highest entropy.  A probability distribution is
"empirically consistent" with a set of training data if its estimated
frequency with which a class and a feature vector value co-occur is
equal to the actual frequency in the data.

Terminology: 'feature'
======================
The term *feature* is usually used to refer to some property of an
unlabeled token.  For example, when performing word sense
disambiguation, we might define a ``'prevword'`` feature whose value is
the word preceding the target word.  However, in the context of
maxent modeling, the term *feature* is typically used to refer to a
property of a "labeled" token.  In order to prevent confusion, we
will introduce two distinct terms to disambiguate these two different
concepts:

  - An "input-feature" is a property of an unlabeled token.
  - A "joint-feature" is a property of a labeled token.

In the rest of the ``nltk.classify`` module, the term "features" is
used to refer to what we will call "input-features" in this module.

In literature that describes and discusses maximum entropy models,
input-features are typically called "contexts", and joint-features
are simply referred to as "features".

Converting Input-Features to Joint-Features
-------------------------------------------
In maximum entropy models, joint-features are required to have numeric
values.  Typically, each input-feature ``input_feat`` is mapped to a
set of joint-features of the form:

|   joint_feat(token, label) = { 1 if input_feat(token) == feat_val
|                              {      and label == some_label
|                              {
|                              { 0 otherwise

For all values of ``feat_val`` and ``some_label``.  This mapping is
performed by classes that implement the ``MaxentFeatureEncodingI``
interface.
é    N)Údefaultdict)ÚClassifierI)Ú
call_megamÚparse_megam_weightsÚwrite_megam_file)Ú	call_tadmÚparse_tadm_weightsÚwrite_tadm_file)ÚCutoffCheckerÚaccuracyÚlog_likelihood)Úgzip_open_unicode)ÚDictionaryProbDist)ÚOrderedDictz
epytext enc                   óx   — e Zd ZdZdd„Zd„ Zd„ Zd„ Zd„ Zd„ Z	dd„Z
dd	„Zdd
„Zd„ Zg d¢Ze	 	 	 	 	 dd„«       Zy)ÚMaxentClassifieraî  
    A maximum entropy classifier (also known as a "conditional
    exponential classifier").  This classifier is parameterized by a
    set of "weights", which are used to combine the joint-features
    that are generated from a featureset by an "encoding".  In
    particular, the encoding maps each ``(featureset, label)`` pair to
    a vector.  The probability of each label is then computed using
    the following equation::

                                dotprod(weights, encode(fs,label))
      prob(fs|label) = ---------------------------------------------------
                       sum(dotprod(weights, encode(fs,l)) for l in labels)

    Where ``dotprod`` is the dot product::

      dotprod(a,b) = sum(x*y for (x,y) in zip(a,b))
    c                 ój   — || _         || _        || _        |j                  «       t	        |«      k(  sJ ‚y)a{  
        Construct a new maxent classifier model.  Typically, new
        classifier models are created using the ``train()`` method.

        :type encoding: MaxentFeatureEncodingI
        :param encoding: An encoding that is used to convert the
            featuresets that are given to the ``classify`` method into
            joint-feature vectors, which are used by the maxent
            classifier model.

        :type weights: list of float
        :param weights:  The feature weight vector for this classifier.

        :type logarithmic: bool
        :param logarithmic: If false, then use non-logarithmic weights.
        N)Ú	_encodingÚ_weightsÚ_logarithmicÚlengthÚlen)ÚselfÚencodingÚweightsÚlogarithmics       úi/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/classify/maxent.pyÚ__init__zMaxentClassifier.__init__a   s3   € ð" "ˆŒØˆŒØ'ˆÔà�‰Ó ¤C¨£LÒ0Ð0Ñ0ó    c                 ó6   — | j                   j                  «       S ©N)r   Úlabels©r   s    r   r"   zMaxentClassifier.labelsx   s   € Ø�~‰~×$Ñ$Ó&Ð&r   c                 ób   — || _         | j                  j                  «       t        |«      k(  sJ ‚y)z¨
        Set the feature weight vector for this classifier.
        :param new_weights: The new feature weight vector.
        :type new_weights: list of float
        N)r   r   r   r   )r   Únew_weightss     r   Úset_weightszMaxentClassifier.set_weights{   s+   € ð $ˆŒØ�~‰~×$Ñ$Ó&¬#¨kÓ*:Ò:Ð:Ñ:r   c                 ó   — | j                   S )zg
        :return: The feature weight vector for this classifier.
        :rtype: list of float
        )r   r#   s    r   r   zMaxentClassifier.weights„   s   € ð
 �}‰}Ðr   c                 ó@   — | j                  |«      j                  «       S r!   )Úprob_classifyÚmax)r   Ú
featuresets     r   ÚclassifyzMaxentClassifier.classify‹   s   € Ø×!Ñ! *Ó-×1Ñ1Ó3Ð3r   c                 ó^  — i }| j                   j                  «       D ]w  }| j                   j                  ||«      }| j                  r'd}|D ]  \  }}|| j                  |   |z  z  }Œ |||<   ŒRd}|D ]  \  }}|| j                  |   |z  z  }Œ |||<   Œy t        || j                  d¬«      S )Ng        ç      ð?T)ÚlogÚ	normalize)r   r"   Úencoder   r   r   )	r   r+   Ú	prob_dictÚlabelÚfeature_vectorÚtotalÚf_idÚf_valÚprods	            r   r)   zMaxentClassifier.prob_classifyŽ   sÏ   € Øˆ	Ø—^‘^×*Ñ*Ó,ò 	(ˆEØ!Ÿ^™^×2Ñ2°:¸uÓEˆNà× Ò Ø�Ø#1ò 9‘K�D˜%Ø˜TŸ]™]¨4Ñ0°5Ñ8Ñ8‘Eð9à#(�	˜%Ò ð �Ø#1ò 9‘K�D˜%Ø˜DŸM™M¨$Ñ/°5Ñ8Ñ8‘Dð9à#'�	˜%Ò ð	(ô  " )°×1BÑ1BÈdÔSÐSr   c           	      ót  ‡ ‡‡— d}dt        |dz
  «      z   dz   }‰ j                  |«      Št        ‰j                  «       ‰j                  d¬«      }|d| }t        dj                  |«      d	j                  d
„ |D «       «      z   «       t        dd|dz
  dt        |«      z  z   z  z   «       t        t        «      Št        |«      D ]ã  \  }}‰ j                  j                  ||«      }|j                  ˆ fd„d¬«       |D ]§  \  }	}
‰ j                  r‰ j                   |	   |
z  }n‰ j                   |	   |
z  }‰ j                  j#                  |	«      }|j%                  d«      d   }|d|
z  z  }t        |«      dkD  r|dd dz   }t        |||dz  dz  |fz  «       ‰|xx   |z  cc<   Œ© Œå t        dd|dz
  dt        |«      z  z   z  z   «       t        dj                  |«      d	j                  ˆfd„|D «       «      z   «       t        dj                  |«      d	j                  ˆfd„|D «       «      z   «       y)zË
        Print a table showing the effect of each of the features in
        the given feature set, and how they combine to determine the
        probabilities of each label for that featureset.
        é2   z  %-é   zs%s%8.3fT©ÚkeyÚreverseNz	  FeatureÚ c              3   ó2   K  — | ]  }d d|z  dd z  –— Œ y­w)z%8sú%sNé   © )Ú.0Úls     r   ú	<genexpr>z+MaxentClassifier.explain.<locals>.<genexpr>°   s   è ø€ Ò?°1�e  q¡¨"¨1˜~Õ.Ñ?ùs   ‚z  ú-é   c                 ó:   •— t        ‰j                  | d      «      S )Nr   ©Úabsr   )Úfid__r   s    €r   ú<lambda>z*MaxentClassifier.explain.<locals>.<lambda>·   s   ø€ ¤# d§m¡m°E¸!±HÑ&=Ó">€ r   ú and label is r   z (%s)é/   é,   z...ú é   z  TOTAL:c              3   ó.   •K  — | ]  }d ‰|   z  –— Œ y­w©z%8.3fNrC   )rD   rE   Úsumss     €r   rF   z+MaxentClassifier.explain.<locals>.<genexpr>Ç   s   øè ø€ Ò3VÈ!°G¸dÀ1¹gÕ4EÑ3Vùs   ƒz  PROBS:c              3   óF   •K  — | ]  }d ‰j                  |«      z  –— Œ y­wrT   )Úprob)rD   rE   Úpdists     €r   rF   z+MaxentClassifier.explain.<locals>.<genexpr>Ë   s   øè ø€ Ò>°!�g §
¡
¨1£Õ-Ñ>ùs   ƒ!)Ústrr)   ÚsortedÚsamplesrW   ÚprintÚljustÚjoinr   r   ÚintÚ	enumerater   r1   Úsortr   r   ÚdescribeÚsplit)r   r+   ÚcolumnsÚdescr_widthÚTEMPLATEr"   Úir3   r4   r6   r7   ÚscoreÚdescrrX   rU   s   `            @@r   ÚexplainzMaxentClassifier.explain¢   s6  ú€ ð ˆØœC ¨a¡Ó0Ñ0°:Ñ=ˆà×"Ñ" :Ó.ˆÜ˜Ÿ™›¨U¯Z©ZÀÔFˆØ˜˜Ð!ˆÜØ×Ñ˜kÓ*Ø�g‰gÑ?¸Ô?Ó?ñ@ô	
ô 	ˆd�S˜K¨!™O¨a´#°f³+©oÑ=Ñ>Ñ>Ô?Üœ3ÓˆÜ! &Ó)ò 	%‰HˆAˆuØ!Ÿ^™^×2Ñ2°:¸uÓEˆNØ×ÑÛ>Èð  ô ð  .ò %‘��eØ×$Ò$Ø ŸM™M¨$Ñ/°%Ñ7‘Eà ŸM™M¨$Ñ/°5Ñ8�EØŸ™×/Ñ/°Ó5�ØŸ™Ð$4Ó5°aÑ8�Ø˜ 5™Ñ(�Ü�u“: ’?Ø! # 2˜J¨Ñ.�EÜ�h %¨¨Q©°©°eÐ!<Ñ<Ô=Ø�U“˜uÑ$”ñ%ð	%ô" 	ˆd�S˜K¨!™O¨a´#°f³+©oÑ=Ñ>Ñ>Ô?ÜØ×Ñ˜[Ó)¨B¯G©GÓ3VÈvÔ3VÓ,VÑVô	
ô 	Ø×Ñ˜[Ó)Ø�g‰gÓ>°vÔ>Ó>ñ?õ	
r   c           	      óÎ   ‡ — t        ‰ d«      r‰ j                  d| S t        t        t	        t        ‰ j                  «      «      «      ˆ fd„d¬«      ‰ _        ‰ j                  d| S )zW
        Generates the ranked list of informative features from most to least.
        Ú_most_informative_featuresNc                 ó4   •— t        ‰j                  |    «      S r!   rJ   )Úfidr   s    €r   rM   z<MaxentClassifier.most_informative_features.<locals>.<lambda>×   s   ø€ ¤ D§M¡M°#Ñ$6Ó 7€ r   Tr<   )Úhasattrrl   rZ   ÚlistÚranger   r   )r   Úns   ` r   Úmost_informative_featuresz*MaxentClassifier.most_informative_featuresÎ   sa   ø€ ô �4Ð5Ô6Ø×2Ñ2°2°AÐ6Ð6ä.4Ü”Uœ3˜tŸ}™}Ó-Ó.Ó/Û7Øô/ˆDÔ+ð
 ×2Ñ2°2°AÐ6Ð6r   c                 óZ  — | j                  d«      }|dk(  r#|D �cg c]  }| j                  |   dkD  sŒ|‘Œ }}n'|dk(  r"|D �cg c]  }| j                  |   dk  sŒ|‘Œ }}|d| D ]9  }t        | j                  |   d›d| j                  j	                  |«      › �«       Œ; yc c}w c c}w )z«
        :param show: all, neg, or pos (for negative-only or positive-only)
        :type show: str
        :param n: The no. of top features
        :type n: int
        NÚposr   Únegz8.3frQ   )rs   r   r\   r   rb   )r   rr   ÚshowÚfidsrn   s        r   Úshow_most_informative_featuresz/MaxentClassifier.show_most_informative_featuresÜ   s·   € ð ×-Ñ-¨dÓ3ˆØ�5Š=Ø#'ÖB˜C¨4¯=©=¸Ñ+=ÀÓ+A’CÐBˆDÑBØ�UŠ]Ø#'ÖB˜C¨4¯=©=¸Ñ+=ÀÓ+A’CÐBˆDÐBØ˜˜�8ò 	OˆCÜ�T—]‘] 3Ñ'¨Ð-¨Q¨t¯~©~×/FÑ/FÀsÓ/KÐ.LÐMÕNñ	Oùò CùâBs   ›B#³B#ÁB(ÁB(c                 ó‚   — dt        | j                  j                  «       «      | j                  j                  «       fz  S )Nz:<ConditionalExponentialClassifier: %d labels, %d features>)r   r   r"   r   r#   s    r   Ú__repr__zMaxentClassifier.__repr__ì   s:   € ØKÜ�—‘×%Ñ%Ó'Ó(Ø�N‰N×!Ñ!Ó#ðO
ñ 
ð 	
r   )ÚGISÚIISÚMEGAMÚTADMNc                 óT  — |€d}|D ]  }|dvsŒt        d|z  «      ‚ |j                  «       }|dk(  rt        ||||fi |¤ŽS |dk(  rt        ||||fi |¤ŽS |dk(  rt	        |||||fi |¤ŽS |dk(  r,|}	||	d<   ||	d<   ||	d	<   ||	d
<   t        j                  |fi |	¤ŽS t        d|z  «      ‚)a®	  
        Train a new maxent classifier based on the given corpus of
        training samples.  This classifier will have its weights
        chosen to maximize entropy while remaining empirically
        consistent with the training corpus.

        :rtype: MaxentClassifier
        :return: The new maxent classifier

        :type train_toks: list
        :param train_toks: Training data, represented as a list of
            pairs, the first member of which is a featureset,
            and the second of which is a classification label.

        :type algorithm: str
        :param algorithm: A case-insensitive string, specifying which
            algorithm should be used to train the classifier.  The
            following algorithms are currently available.

            - Iterative Scaling Methods: Generalized Iterative Scaling (``'GIS'``),
              Improved Iterative Scaling (``'IIS'``)
            - External Libraries (requiring megam):
              LM-BFGS algorithm, with training performed by Megam (``'megam'``)

            The default algorithm is ``'IIS'``.

        :type trace: int
        :param trace: The level of diagnostic tracing output to produce.
            Higher values produce more verbose output.
        :type encoding: MaxentFeatureEncodingI
        :param encoding: A feature encoding, used to convert featuresets
            into feature vectors.  If none is specified, then a
            ``BinaryMaxentFeatureEncoding`` will be built based on the
            features that are attested in the training corpus.
        :type labels: list(str)
        :param labels: The set of possible labels.  If none is given, then
            the set of all labels attested in the training data will be
            used instead.
        :param gaussian_prior_sigma: The sigma value for a gaussian
            prior on model weights.  Currently, this is supported by
            ``megam``. For other algorithms, its value is ignored.
        :param cutoffs: Arguments specifying various conditions under
            which the training should be halted.  (Some of the cutoff
            conditions are not supported by some algorithms.)

            - ``max_iter=v``: Terminate after ``v`` iterations.
            - ``min_ll=v``: Terminate after the negative average
              log-likelihood drops under ``v``.
            - ``min_lldelta=v``: Terminate if a single iteration improves
              log likelihood by less than ``v``.
        Úiis)	Úmax_iterÚmin_llÚmin_lldeltaÚmax_accÚmin_accdeltaÚcount_cutoffÚnormÚexplicitÚ	bernoullizUnexpected keyword arg %rÚgisÚmegamÚtadmÚtracer   r"   Úgaussian_prior_sigmazUnknown algorithm %s)Ú	TypeErrorÚlowerÚ train_maxent_classifier_with_iisÚ train_maxent_classifier_with_gisÚ"train_maxent_classifier_with_megamÚTadmMaxentClassifierÚtrainÚ
ValueError)
ÚclsÚ
train_toksÚ	algorithmrŽ   r   r"   r�   Úcutoffsr=   Úkwargss
             r   r–   zMaxentClassifier.trainö   s&  € ð| ÐØˆIØò 	CˆCØð 
ò 
ô  Ð ;¸cÑ AÓBÐBð	Cð —O‘OÓ%ˆ	Ø˜ÒÜ3Ø˜E 8¨VñØ7>ñð ð ˜%ÒÜ3Ø˜E 8¨VñØ7>ñð ð ˜'Ò!Ü5Ø˜E 8¨VÐ5IñØMTñð ð ˜&Ò ØˆFØ#ˆF�7‰OØ!)ˆF�:ÑØ%ˆF�8ÑØ-AˆFÐ)Ñ*Ü'×-Ñ-¨jÑC¸FÑCÐCäÐ3°iÑ?Ó@Ð@r   )T)é   )é
   )rž   Úall)Né   NNr   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r"   r&   r   r,   r)   rj   rs   ry   r{   Ú
ALGORITHMSÚclassmethodr–   rC   r   r   r   r   N   sj   „ ñó$1ò.'ò;òò4òTó(*
óX7óOò 
ò 1€Jàð ØØØØòaAó ñaAr   r   c                   ó.   — e Zd ZdZd„ Zd„ Zd„ Zd„ Zd„ Zy)ÚMaxentFeatureEncodingIa´  
    A mapping that converts a set of input-feature values to a vector
    of joint-feature values, given a label.  This conversion is
    necessary to translate featuresets into a format that can be used
    by maximum entropy models.

    The set of joint-features used by a given encoding is fixed, and
    each index in the generated joint-feature vectors corresponds to a
    single joint-feature.  The length of the generated joint-feature
    vectors is therefore constant (for a given encoding).

    Because the joint-feature vectors generated by
    ``MaxentFeatureEncodingI`` are typically very sparse, they are
    represented as a list of ``(index, value)`` tuples, specifying the
    value of each non-zero joint-feature.

    Feature encodings are generally created using the ``train()``
    method, which generates an appropriate encoding based on the
    input-feature values and labels that are present in a given
    corpus.
    c                 ó   — t        «       ‚)aC  
        Given a (featureset, label) pair, return the corresponding
        vector of joint-feature values.  This vector is represented as
        a list of ``(index, value)`` tuples, specifying the value of
        each non-zero joint-feature.

        :type featureset: dict
        :rtype: list(tuple(int, int))
        ©ÚNotImplementedError©r   r+   r3   s      r   r1   zMaxentFeatureEncodingI.encode{  ó   € ô "Ó#Ð#r   c                 ó   — t        «       ‚)z’
        :return: The size of the fixed-length joint-feature vectors
            that are generated by this encoding.
        :rtype: int
        rª   r#   s    r   r   zMaxentFeatureEncodingI.length‡  ó   € ô "Ó#Ð#r   c                 ó   — t        «       ‚)zÞ
        :return: A list of the "known labels" -- i.e., all labels
            ``l`` such that ``self.encode(fs,l)`` can be a nonzero
            joint-feature vector for some value of ``fs``.
        :rtype: list
        rª   r#   s    r   r"   zMaxentFeatureEncodingI.labels�  s   € ô "Ó#Ð#r   c                 ó   — t        «       ‚)z¦
        :return: A string describing the value of the joint-feature
            whose index in the generated feature vectors is ``fid``.
        :rtype: str
        rª   ©r   rn   s     r   rb   zMaxentFeatureEncodingI.describe˜  r¯   r   c                 ó   — t        «       ‚)ao  
        Construct and return new feature encoding, based on a given
        training corpus ``train_toks``.

        :type train_toks: list(tuple(dict, str))
        :param train_toks: Training data, represented as a list of
            pairs, the first member of which is a feature dictionary,
            and the second of which is a classification label.
        rª   )r˜   r™   s     r   r–   zMaxentFeatureEncodingI.train   r­   r   N)	r¡   r¢   r£   r¤   r1   r   r"   rb   r–   rC   r   r   r¨   r¨   d  s    „ ñò,
$ò$ò$ò$ó
$r   r¨   c                   ó.   — e Zd ZdZd„ Zd„ Zd„ Zd„ Zd„ Zy)Ú#FunctionBackedMaxentFeatureEncodingz‹
    A feature encoding that calls a user-supplied function to map a
    given featureset/label pair to a sparse joint-feature vector.
    c                 ó.   — || _         || _        || _        y)ag  
        Construct a new feature encoding based on the given function.

        :type func: (callable)
        :param func: A function that takes two arguments, a featureset
             and a label, and returns the sparse joint feature vector
             that encodes them::

                 func(featureset, label) -> feature_vector

             This sparse joint feature vector (``feature_vector``) is a
             list of ``(index,value)`` tuples.

        :type length: int
        :param length: The size of the fixed-length joint-feature
            vectors that are generated by this encoding.

        :type labels: list
        :param labels: A list of the "known labels" for this
            encoding -- i.e., all labels ``l`` such that
            ``self.encode(fs,l)`` can be a nonzero joint-feature vector
            for some value of ``fs``.
        N)Ú_lengthÚ_funcÚ_labels)r   Úfuncr   r"   s       r   r   z,FunctionBackedMaxentFeatureEncoding.__init__³  s   € ð0 ˆŒØˆŒ
Øˆ�r   c                 ó&   — | j                  ||«      S r!   )r¸   r¬   s      r   r1   z*FunctionBackedMaxentFeatureEncoding.encodeÏ  s   € Ø�z‰z˜* eÓ,Ð,r   c                 ó   — | j                   S r!   ©r·   r#   s    r   r   z*FunctionBackedMaxentFeatureEncoding.lengthÒ  ó   € Ø�|‰|Ðr   c                 ó   — | j                   S r!   ©r¹   r#   s    r   r"   z*FunctionBackedMaxentFeatureEncoding.labelsÕ  r¾   r   c                  ó   — y)Nzno description availablerC   r²   s     r   rb   z,FunctionBackedMaxentFeatureEncoding.describeØ  s   € Ø)r   N)	r¡   r¢   r£   r¤   r   r1   r   r"   rb   rC   r   r   rµ   rµ   ­  s    „ ñò
ò8-òòó*r   rµ   c                   óB   — e Zd ZdZd	d„Zd„ Zd„ Zd„ Zd„ Ze	d
d„«       Z
y)ÚBinaryMaxentFeatureEncodingaø  
    A feature encoding that generates vectors containing a binary
    joint-features of the form:

    |  joint_feat(fs, l) = { 1 if (fs[fname] == fval) and (l == label)
    |                      {
    |                      { 0 otherwise

    Where ``fname`` is the name of an input-feature, ``fval`` is a value
    for that input-feature, and ``label`` is a label.

    Typically, these features are constructed based on a training
    corpus, using the ``train()`` method.  This method will create one
    feature for each combination of ``fname``, ``fval``, and ``label``
    that occurs at least once in the training corpus.

    The ``unseen_features`` parameter can be used to add "unseen-value
    features", which are used whenever an input feature has a value
    that was not encountered in the training corpus.  These features
    have the form:

    |  joint_feat(fs, l) = { 1 if is_unseen(fname, fs[fname])
    |                      {      and l == label
    |                      {
    |                      { 0 otherwise

    Where ``is_unseen(fname, fval)`` is true if the encoding does not
    contain any joint features that are true when ``fs[fname]==fval``.

    The ``alwayson_features`` parameter can be used to add "always-on
    features", which have the form::

    |  joint_feat(fs, l) = { 1 if (l == label)
    |                      {
    |                      { 0 otherwise

    These always-on features allow the maxent model to directly model
    the prior probabilities of each label.
    c                 óª  — t        |j                  «       «      t        t        t        |«      «      «      k7  rt	        d«      ‚t        |«      | _        	 || _        	 t        |«      | _        	 d| _	        	 d| _
        	 |rYt        |«      D ��ci c]  \  }}||| j                  z   “Œ c}}| _	        | xj                  t        | j                  «      z  c_        |rg|D ���ch c]  \  }}}|’Œ
 }	}}}t        |	«      D ��ci c]  \  }}||| j                  z   “Œ c}}| _
        | xj                  t        |	«      z  c_        yyc c}}w c c}}}w c c}}w )a»  
        :param labels: A list of the "known labels" for this encoding.

        :param mapping: A dictionary mapping from ``(fname,fval,label)``
            tuples to corresponding joint-feature indexes.  These
            indexes must be the set of integers from 0...len(mapping).
            If ``mapping[fname,fval,label]=id``, then
            ``self.encode(..., fname:fval, ..., label)[id]`` is 1;
            otherwise, it is 0.

        :param unseen_features: If true, then include unseen value
           features in the generated joint-feature vectors.

        :param alwayson_features: If true, then include always-on
           features in the generated joint-feature vectors.
        úHMapping values must be exactly the set of integers from 0...len(mapping)N©ÚsetÚvaluesrq   r   r—   rp   r¹   Ú_mappingr·   Ú	_alwaysonÚ_unseenr`   ©
r   r"   ÚmappingÚunseen_featuresÚalwayson_featuresrg   r3   ÚfnameÚfvalÚfnamess
             r   r   z$BinaryMaxentFeatureEncoding.__init__  ó-  € ô" ˆw�~‰~ÓÓ ¤C¬¬c°'«lÓ(;Ó$<Ò<Üð8óð ô
 ˜F“|ˆŒØ(àˆŒØ9ä˜7“|ˆŒØ<àˆŒØ,àˆŒØ,áä:CÀFÓ:K÷Ù,6¨Q°��q˜4Ÿ<™<Ñ'Ñ'óˆDŒNð �LŠLœC §¡Ó/Ñ/�LáØ8?×@Ð@Ñ 4 ¨¨e’eÐ@ˆFÒ@ÜFOÐPVÓFW×X¹
¸¸E˜E 1 t§|¡|Ñ#3Ñ3ÓXˆDŒLØ�LŠLœC ›KÑ'ŽLð ùóùô AùÛXó   ÂEÃ EÃ?Ec                 óØ  — g }|j                  «       D ]š  \  }}|||f| j                  v r$|j                  | j                  |||f   df«       Œ;| j                  sŒH| j                  D ]  }|||f| j                  v sŒ Œk || j                  v sŒ{|j                  | j                  |   df«       Œœ | j
                  r.|| j
                  v r |j                  | j
                  |   df«       |S ©NrR   )ÚitemsrÉ   ÚappendrË   r¹   rÊ   ©r   r+   r3   r   rÐ   rÑ   Úlabel2s          r   r1   z"BinaryMaxentFeatureEncoding.encode6  sê   € àˆð &×+Ñ+Ó-ò 	B‰KˆE�4à�t˜UÐ# t§}¡}Ñ4Ø—‘ §¡¨u°d¸EÐ/AÑ!BÀAÐ FÕGð —“à"Ÿl™lò B�FØ˜t VÐ,°·±Ò=ÙðBð
  §¡Ò,Ø Ÿ™¨¯©°eÑ)<¸aÐ(@ÕAð	Bð" �>Š>˜e t§~¡~Ñ5Ø�O‰O˜TŸ^™^¨EÑ2°AÐ6Ô7àˆr   c                 óì  — t        |t        «      st        d«      ‚	 | j                   |t        | j                  «      k  r| j                  |   \  }}}|› d|›d|›�S | j                  rK|| j                  j                  «       v r/| j                  j                  «       D ]  \  }}||k(  sŒd|z  c S  y | j                  rK|| j                  j                  «       v r/| j                  j                  «       D ]  \  }}||k(  sŒd|z  c S  y t        d«      ‚# t        $ rS dgt        | j                  «      z  | _        | j                  j                  «       D ]  \  }}|| j                  |<   Œ Y �ŒIw xY w©Nzdescribe() expected an intéÿÿÿÿz==rN   zlabel is %rz%s is unseenzBad feature id©Ú
isinstancer_   r�   Ú_inv_mappingÚAttributeErrorr   rÉ   r×   rÊ   rÈ   rË   r—   ©r   r6   Úinforg   rÐ   rÑ   r3   Úf_id2s           r   rb   z$BinaryMaxentFeatureEncoding.describeQ  ój  € ä˜$¤Ô$ÜÐ8Ó9Ð9ð	,Ø×Òð ”#�d—m‘mÓ$Ò$Ø#'×#4Ñ#4°TÑ#:Ñ ˆU�D˜%Ø�W˜B˜t˜h n°U°IÐ>Ð>Ø�^Š^ ¨¯©×(=Ñ(=Ó(?Ñ ?Ø $§¡× 4Ñ 4Ó 6ò 1‘��uØ˜5“=Ø(¨5Ñ0Ò0ñ1ð �\Š\˜d d§l¡l×&9Ñ&9Ó&;Ñ;Ø $§¡× 2Ñ 2Ó 4ò 2‘��uØ˜5“=Ø)¨EÑ1Ò1ñ2ô Ð-Ó.Ð.øô# ò 	,Ø!# ¤s¨4¯=©=Ó'9Ñ 9ˆDÔØŸ=™=×.Ñ.Ó0ò ,‘��aØ'+�×!Ñ! !Ò$ó,ð	,úó   �D ÄAE3Å2E3c                 ó   — | j                   S r!   rÀ   r#   s    r   r"   z"BinaryMaxentFeatureEncoding.labelsj  ó   € à�|‰|Ðr   c                 ó   — | j                   S r!   r½   r#   s    r   r   z"BinaryMaxentFeatureEncoding.lengthn  rè   r   Nc                 óH  — i }t        «       }t        t        «      }|D ]u  \  }}	|r|	|vrt        d|	z  «      ‚|j	                  |	«       |j                  «       D ]8  \  }
}||
|fxx   dz  cc<   ||
|f   |k\  sŒ |
||	f|vsŒ(t        |«      ||
||	f<   Œ: Œw |€|} | ||fi |¤ŽS )aª  
        Construct and return new feature encoding, based on a given
        training corpus ``train_toks``.  See the class description
        ``BinaryMaxentFeatureEncoding`` for a description of the
        joint-features that will be included in this encoding.

        :type train_toks: list(tuple(dict, str))
        :param train_toks: Training data, represented as a list of
            pairs, the first member of which is a feature dictionary,
            and the second of which is a classification label.

        :type count_cutoff: int
        :param count_cutoff: A cutoff value that is used to discard
            rare joint-features.  If a joint-feature's value is 1
            fewer than ``count_cutoff`` times in the training corpus,
            then that joint-feature is not included in the generated
            encoding.

        :type labels: list
        :param labels: A list of labels that should be used by the
            classifier.  If not specified, then the set of labels
            attested in ``train_toks`` will be used.

        :param options: Extra parameters for the constructor, such as
            ``unseen_features`` and ``alwayson_features``.
        úUnexpected label %srR   )rÇ   r   r_   r—   Úaddr×   r   ©r˜   r™   r‡   r"   ÚoptionsrÍ   Úseen_labelsÚcountÚtokr3   rÐ   rÑ   s               r   r–   z!BinaryMaxentFeatureEncoding.trainr  sÜ   € ð8 ˆÜ“eˆÜœCÓ ˆà$ò 	C‰JˆC�Ù˜% vÑ-Ü Ð!6¸Ñ!>Ó?Ð?Ø�O‰O˜EÔ"ð  #Ÿy™y›{ò C‘��tð �e˜T�kÓ" aÑ'Ó"Ø˜ ˜Ñ%¨Ó5Ø˜t UÐ+°7Ò:Ü69¸'³l˜  t¨UÐ 2Ò3ñCð	Cð ˆ>Ø ˆFÙ�6˜7Ñ. gÑ.Ð.r   ©FF©r   N©r¡   r¢   r£   r¤   r   r1   rb   r"   r   r¦   r–   rC   r   r   rÃ   rÃ   Ü  s6   „ ñ&óP/(òbò6/ò2òð ò0/ó ñ0/r   rÃ   c                   ó<   — e Zd ZdZ	 dd„Zed„ «       Zd„ Zd„ Zd„ Z	y)	ÚGISEncodinga  
    A binary feature encoding which adds one new joint-feature to the
    joint-features defined by ``BinaryMaxentFeatureEncoding``: a
    correction feature, whose value is chosen to ensure that the
    sparse vector always sums to a constant non-negative number.  This
    new feature is used to ensure two preconditions for the GIS
    training algorithm:

      - At least one feature vector index must be nonzero for every
        token.
      - The feature vector must sum to a constant non-negative number
        for every token.
    Nc           
      óž   — t         j                  | ||||«       |€$t        |D ���ch c]  \  }}}|’Œ
 c}}}«      dz   }|| _        yc c}}}w )a	  
        :param C: The correction constant.  The value of the correction
            feature is based on this value.  In particular, its value is
            ``C - sum([v for (f,v) in encoding])``.
        :seealso: ``BinaryMaxentFeatureEncoding.__init__``
        NrR   )rÃ   r   r   Ú_C)	r   r"   rÍ   rÎ   rÏ   ÚCrÐ   rÑ   r3   s	            r   r   zGISEncoding.__init__µ  sV   € ô 	$×,Ñ,Ø�&˜' ?Ð4Eô	
ð ˆ9Ü°w×?Ð?Ñ3  t¨U’UÔ?Ó@À1ÑDˆAØˆ�ùô @s   §Ac                 ó   — | j                   S )zOThe non-negative constant that all encoded feature vectors
        will sum to.)rø   r#   s    r   rù   zGISEncoding.CÅ  s   € ð �w‰wˆr   c                 óö   — t         j                  | ||«      }t         j                  | «      }t        d„ |D «       «      }|| j                  k\  rt        d«      ‚|j                  || j                  |z
  f«       |S )Nc              3   ó&   K  — | ]	  \  }}|–— Œ y ­wr!   rC   )rD   ÚfÚvs      r   rF   z%GISEncoding.encode.<locals>.<genexpr>Ñ  s   è ø€ Ò-™&˜1˜a”AÑ-ùó   ‚z&Correction feature is not high enough!)rÃ   r1   r   Úsumrø   r—   rØ   )r   r+   r3   r   Úbase_lengthr5   s         r   r1   zGISEncoding.encodeË  sp   € ä.×5Ñ5°d¸JÈÓNˆÜ1×8Ñ8¸Ó>ˆô Ñ- HÔ-Ó-ˆØ�D—G‘GÒÜÐEÓFÐFØ�‰˜ d§g¡g°¡oÐ6Ô7ð ˆr   c                 ó2   — t         j                  | «      dz   S rÖ   )rÃ   r   r#   s    r   r   zGISEncoding.lengthÙ  s   € Ü*×1Ñ1°$Ó7¸!Ñ;Ð;r   c                 ó|   — |t         j                  | «      k(  rd| j                  z  S t         j                  | |«      S )NzCorrection feature (%s))rÃ   r   rø   rb   )r   r6   s     r   rb   zGISEncoding.describeÜ  s8   € ØÔ.×5Ñ5°dÓ;Ò;Ø,¨t¯w©wÑ6Ð6ä.×7Ñ7¸¸dÓCÐCr   )FFN)
r¡   r¢   r£   r¤   r   Úpropertyrù   r1   r   rb   rC   r   r   rö   rö   ¦  s7   „ ñð RVóð  ñó ðò
ò<óDr   rö   c                   ó>   — e Zd Zdd„Zd„ Zd„ Zd„ Zd„ Zed	d„«       Z	y)
ÚTadmEventMaxentFeatureEncodingc                 óˆ   — t        |«      | _        t        «       | _        t        j	                  | || j                  ||«       y r!   )r   rÉ   Ú_label_mappingrÃ   r   )r   r"   rÍ   rÎ   rÏ   s        r   r   z'TadmEventMaxentFeatureEncoding.__init__ä  s6   € Ü# GÓ,ˆŒÜ)›mˆÔÜ#×,Ñ,Ø�&˜$Ÿ-™-¨Ð:Kõ	
r   c                 ó   — g }|j                  «       D ]¸  \  }}||f| j                  vr$t        | j                  «      | j                  ||f<   || j                  vrBt	        |t
        «      s#t        | j                  «      | j                  |<   n|| j                  |<   |j                  | j                  ||f   | j                  |   f«       Œº |S r!   )r×   rÉ   r   r  rß   r_   rØ   )r   r+   r3   r   ÚfeatureÚvalues         r   r1   z%TadmEventMaxentFeatureEncoding.encodeë  sÇ   € ØˆØ(×.Ñ.Ó0ò 
	‰NˆG�UØ˜Ð t§}¡}Ñ4Ü25°d·m±mÓ2D�—‘˜w¨Ð.Ñ/Ø˜D×/Ñ/Ñ/Ü! %¬Ô-Ü14°T×5HÑ5HÓ1I�D×'Ñ'¨Ò.à16�D×'Ñ'¨Ñ.Ø�O‰OØ—‘ ¨Ð/Ñ0°$×2EÑ2EÀeÑ2LÐMõð
	ð ˆr   c                 ó   — | j                   S r!   rÀ   r#   s    r   r"   z%TadmEventMaxentFeatureEncoding.labelsú  r¾   r   c                 ó`   — | j                   D ]  \  }}| j                   ||f   |k(  sŒ||fc S  y r!   )rÉ   )r   rn   r
  r3   s       r   rb   z'TadmEventMaxentFeatureEncoding.describeý  s:   € Ø"Ÿm™mò 	(‰NˆG�UØ�}‰}˜g uÐ-Ñ.°#Ó5Ø Ð'Ò'ñ	(r   c                 ó,   — t        | j                  «      S r!   )r   rÉ   r#   s    r   r   z%TadmEventMaxentFeatureEncoding.length  s   € Ü�4—=‘=Ó!Ð!r   Nc                 óæ   — t        «       }|sg }t        |«      }|D ]  \  }}||vsŒ|j                  |«       Œ |D ]*  \  }}|D ]   }|D ]  }||f|vsŒ
t        |«      |||f<   Œ Œ" Œ,  | ||fi |¤ŽS r!   )r   rp   rØ   r   )	r˜   r™   r‡   r"   rî   rÍ   r+   r3   r
  s	            r   r–   z$TadmEventMaxentFeatureEncoding.train  s³   € ä“-ˆÙØˆFô ˜*Ó%ˆ
à!+ò 	%ÑˆJ˜Ø˜FÒ"Ø—‘˜eÕ$ð	%ð ",ò 	AÑˆJ˜Øò A�Ø)ò A�GØ Ð'¨wÒ6Ü47¸³L˜ ¨%Ð 0Ò1ñAñAð	Añ �6˜7Ñ. gÑ.Ð.r   rò   ró   )
r¡   r¢   r£   r   r1   r"   rb   r   r¦   r–   rC   r   r   r  r  ã  s/   „ ó
òòò(ò
"ð ò/ó ñ/r   r  c                   óB   — e Zd ZdZd	d„Zd„ Zd„ Zd„ Zd„ Ze	d
d„«       Z
y)ÚTypedMaxentFeatureEncodingaZ  
    A feature encoding that generates vectors containing integer,
    float and binary joint-features of the form:

    Binary (for string and boolean features):

    |  joint_feat(fs, l) = { 1 if (fs[fname] == fval) and (l == label)
    |                      {
    |                      { 0 otherwise

    Value (for integer and float features):

    |  joint_feat(fs, l) = { fval if     (fs[fname] == type(fval))
    |                      {         and (l == label)
    |                      {
    |                      { not encoded otherwise

    Where ``fname`` is the name of an input-feature, ``fval`` is a value
    for that input-feature, and ``label`` is a label.

    Typically, these features are constructed based on a training
    corpus, using the ``train()`` method.

    For string and boolean features [type(fval) not in (int, float)]
    this method will create one feature for each combination of
    ``fname``, ``fval``, and ``label`` that occurs at least once in the
    training corpus.

    For integer and float features [type(fval) in (int, float)] this
    method will create one feature for each combination of ``fname``
    and ``label`` that occurs at least once in the training corpus.

    For binary features the ``unseen_features`` parameter can be used
    to add "unseen-value features", which are used whenever an input
    feature has a value that was not encountered in the training
    corpus.  These features have the form:

    |  joint_feat(fs, l) = { 1 if is_unseen(fname, fs[fname])
    |                      {      and l == label
    |                      {
    |                      { 0 otherwise

    Where ``is_unseen(fname, fval)`` is true if the encoding does not
    contain any joint features that are true when ``fs[fname]==fval``.

    The ``alwayson_features`` parameter can be used to add "always-on
    features", which have the form:

    |  joint_feat(fs, l) = { 1 if (l == label)
    |                      {
    |                      { 0 otherwise

    These always-on features allow the maxent model to directly model
    the prior probabilities of each label.
    c                 óª  — t        |j                  «       «      t        t        t        |«      «      «      k7  rt	        d«      ‚t        |«      | _        	 || _        	 t        |«      | _        	 d| _	        	 d| _
        	 |rYt        |«      D ��ci c]  \  }}||| j                  z   “Œ c}}| _	        | xj                  t        | j                  «      z  c_        |rg|D ���ch c]  \  }}}|’Œ
 }	}}}t        |	«      D ��ci c]  \  }}||| j                  z   “Œ c}}| _
        | xj                  t        |	«      z  c_        yyc c}}w c c}}}w c c}}w )a½  
        :param labels: A list of the "known labels" for this encoding.

        :param mapping: A dictionary mapping from ``(fname,fval,label)``
            tuples to corresponding joint-feature indexes.  These
            indexes must be the set of integers from 0...len(mapping).
            If ``mapping[fname,fval,label]=id``, then
            ``self.encode({..., fname:fval, ...``, label)[id]} is 1;
            otherwise, it is 0.

        :param unseen_features: If true, then include unseen value
           features in the generated joint-feature vectors.

        :param alwayson_features: If true, then include always-on
           features in the generated joint-feature vectors.
        rÅ   NrÆ   rÌ   s
             r   r   z#TypedMaxentFeatureEncoding.__init__T  rÓ   rÔ   c                 ó”  — g }|j                  «       D ]ø  \  }}t        |t        t        f«      rH|t	        |«      |f| j
                  v sŒ7|j                  | j
                  |t	        |«      |f   |f«       Œd|||f| j
                  v r$|j                  | j
                  |||f   df«       Œ™| j                  sŒ¦| j                  D ]  }|||f| j
                  v sŒ ŒÉ || j                  v sŒÙ|j                  | j                  |   df«       Œú | j                  r.|| j                  v r |j                  | j                  |   df«       |S rÖ   )
r×   rß   r_   ÚfloatÚtyperÉ   rØ   rË   r¹   rÊ   rÙ   s          r   r1   z!TypedMaxentFeatureEncoding.encode…  s6  € àˆð &×+Ñ+Ó-ò 	F‰KˆE�4Ü˜$¤¤e Ô-àœ4 ›: uÐ-°·±Ò>Ø—O‘O T§]¡]°5¼$¸t»*ÀeÐ3KÑ%LÈdÐ$SÕTð ˜4 Ð'¨4¯=©=Ñ8Ø—O‘O T§]¡]°5¸$ÀÐ3EÑ%FÈÐ$JÕKð —\“\à"&§,¡,ò F˜Ø! 4¨Ð0°D·M±MÒAÙ!ðFð
 ! D§L¡LÒ0Ø$ŸO™O¨T¯\©\¸%Ñ-@À!Ð,DÕEð'	Fð, �>Š>˜e t§~¡~Ñ5Ø�O‰O˜TŸ^™^¨EÑ2°AÐ6Ô7àˆr   c                 óì  — t        |t        «      st        d«      ‚	 | j                   |t        | j                  «      k  r| j                  |   \  }}}|› d|›d|›�S | j                  rK|| j                  j                  «       v r/| j                  j                  «       D ]  \  }}||k(  sŒd|z  c S  y | j                  rK|| j                  j                  «       v r/| j                  j                  «       D ]  \  }}||k(  sŒd|z  c S  y t        d«      ‚# t        $ rS dgt        | j                  «      z  | _        | j                  j                  «       D ]  \  }}|| j                  |<   Œ Y �ŒIw xY wrÜ   rÞ   râ   s           r   rb   z#TypedMaxentFeatureEncoding.describe¥  rå   ræ   c                 ó   — | j                   S r!   rÀ   r#   s    r   r"   z!TypedMaxentFeatureEncoding.labels¾  rè   r   c                 ó   — | j                   S r!   r½   r#   s    r   r   z!TypedMaxentFeatureEncoding.lengthÂ  rè   r   Nc                 óŒ  — i }t        «       }t        t        «      }|D ]—  \  }}	|r|	|vrt        d|	z  «      ‚|j	                  |	«       |j                  «       D ]Z  \  }
}t        |«      t        t        fv rt        |«      }||
|fxx   dz  cc<   ||
|f   |k\  sŒB|
||	f|vsŒJt        |«      ||
||	f<   Œ\ Œ™ |€|} | ||fi |¤ŽS )a)  
        Construct and return new feature encoding, based on a given
        training corpus ``train_toks``.  See the class description
        ``TypedMaxentFeatureEncoding`` for a description of the
        joint-features that will be included in this encoding.

        Note: recognized feature values types are (int, float), over
        types are interpreted as regular binary features.

        :type train_toks: list(tuple(dict, str))
        :param train_toks: Training data, represented as a list of
            pairs, the first member of which is a feature dictionary,
            and the second of which is a classification label.

        :type count_cutoff: int
        :param count_cutoff: A cutoff value that is used to discard
            rare joint-features.  If a joint-feature's value is 1
            fewer than ``count_cutoff`` times in the training corpus,
            then that joint-feature is not included in the generated
            encoding.

        :type labels: list
        :param labels: A list of labels that should be used by the
            classifier.  If not specified, then the set of labels
            attested in ``train_toks`` will be used.

        :param options: Extra parameters for the constructor, such as
            ``unseen_features`` and ``alwayson_features``.
        rë   rR   )	rÇ   r   r_   r—   rì   r×   r  r  r   rí   s               r   r–   z TypedMaxentFeatureEncoding.trainÆ  sõ   € ð> ˆÜ“eˆÜœCÓ ˆà$ò 	C‰JˆC�Ù˜% vÑ-Ü Ð!6¸Ñ!>Ó?Ð?Ø�O‰O˜EÔ"ð  #Ÿy™y›{ò 	C‘��tÜ˜“:¤#¤u Ñ-Ü ›:�Dð �e˜T�kÓ" aÑ'Ó"Ø˜ ˜Ñ%¨Ó5Ø˜t UÐ+°7Ò:Ü69¸'³l˜  t¨UÐ 2Ò3ñ	Cð	Cð" ˆ>Ø ˆFÙ�6˜7Ñ. gÑ.Ð.r   rò   ró   rô   rC   r   r   r  r    s7   „ ñ6óp/(òbò@/ò2òð ò5/ó ñ5/r   r  c                 ó~  — |j                  dd«       t        |«      }|€t        j                  | |¬«      }t	        |d«      st        d«      ‚d|j                  z  }t        | |«      }t        t        j                  |dk(  «      d   «      }t        j                  t        |«      d«      }	|D ]  }
t        j                  |	|
<   Œ t        ||	«      }t        j                  |«      }~|dkD  rt!        d	|d   z  «       |d
kD  r t!        «        t!        d«       t!        d«       	 	 |d
kD  rQ|j"                  xs t%        || «      }|j&                  xs t)        || «      }|j*                  }t!        d|||fz  «       t-        || |«      }|D ]  }
||
xx   dz  cc<   Œ t        j                  |«      }~|j/                  «       }	|	||z
  |z  z  }	|j1                  |	«       |j3                  || «      rnŒÍ	 |d
kD  r+t%        || «      }t)        || «      }t!        d|d›d|d›�«       |S # t4        $ r t!        d«       Y ŒHw xY w)a†  
    Train a new ``ConditionalExponentialClassifier``, using the given
    training samples, using the Generalized Iterative Scaling
    algorithm.  This ``ConditionalExponentialClassifier`` will encode
    the model that maximizes entropy from all the models that are
    empirically consistent with ``train_toks``.

    :see: ``train_maxent_classifier()`` for parameter descriptions.
    r‚   éd   ©r"   rù   zJThe GIS algorithm requires an encoding that defines C (e.g., GISEncoding).r.   r   Údú  ==> Training (%d iterations)r;   ú-      Iteration    Log Likelihood    Accuracyú-      ---------------------------------------ú     %9d    %14.5f    %9.3frR   ú*      Training stopped: keyboard interruptú         Final    ú14.5fú    ú9.3f)Ú
setdefaultr   rö   r–   ro   r�   rù   Úcalculate_empirical_fcountrÇ   ÚnumpyÚnonzeroÚzerosr   ÚNINFÚ ConditionalExponentialClassifierÚlog2r\   Úllr   Úaccr   ÚiterÚcalculate_estimated_fcountr   r&   ÚcheckÚKeyboardInterrupt)r™   rŽ   r   r"   r›   ÚcutoffcheckerÚCinvÚempirical_fcountÚ
unattestedr   rn   Ú
classifierÚlog_empirical_fcountr/  r0  ÚiternumÚestimated_fcountÚlog_estimated_fcounts                     r   r“   r“     sx  € ð ×Ñ�z 3Ô'Ü! 'Ó*€Mð ÐÜ×$Ñ$ Z¸Ð$Ó?ˆä�8˜SÔ!Üð-ó
ð 	
ð �—‘Ñ€Dô 2°*¸hÓGÐô ”U—]‘]Ð#3°qÑ#8Ó9¸!Ñ<Ó=€Jô �k‰kœ#Ð.Ó/°Ó5€GØò "ˆÜ—z‘zˆ�Šð"ä1°(¸GÓD€Jô !Ÿ:™:Ð&6Ó7ÐØàˆq‚yÜÐ.°¸Ñ1DÑDÔEØˆq‚yÜŒÜÐ=Ô>ÜÐ=Ô>ð<ØØ�qŠyØ"×%Ñ%ÒO¬¸
ÀJÓ)O�Ø#×'Ñ'ÒK¬8°JÀ
Ó+K�Ø'×,Ñ,�ÜÐ3°wÀÀCÐ6HÑHÔIô  :Ø˜J¨ó Ðð
 "ò +�Ø  Ó%¨Ñ*Ô%ð+ä#(§:¡:Ð.>Ó#?Ð Ø ð !×(Ñ(Ó*ˆGØÐ,Ð/CÑCÀtÑKÑKˆGØ×"Ñ" 7Ô+ð ×"Ñ" :¨zÔ:Øð5 ð4 ð
 ˆq‚yÜ˜J¨
Ó3ˆÜ�z :Ó.ˆÜÐ" 2 e *¨D°°T°
Ð;Ô<ð Ðøô ò <ÜÐ:Ö;ð<ús   Ä$CH% È%H<È;H<c                 ó°   — t        j                  |j                  «       d«      }| D ],  \  }}|j                  ||«      D ]  \  }}||xx   |z  cc<   Œ Œ. |S ©Nr  )r)  r+  r   r1   )r™   r   Úfcountrñ   r3   ÚindexÚvals          r   r(  r(  d  s_   € Ü�[‰[˜Ÿ™Ó*¨CÓ0€Fà ò !‰
ˆˆUØ"Ÿ/™/¨#¨uÓ5ò 	!‰JˆE�3Ø�5‹M˜SÑ ŒMñ	!ð!ð €Mr   c                 ó$  — t        j                  |j                  «       d«      }|D ]f  \  }}| j                  |«      }|j	                  «       D ]=  }|j                  |«      }|j                  ||«      D ]  \  }}	||xx   ||	z  z  cc<   Œ Œ? Œh |S r?  )r)  r+  r   r)   r[   rW   r1   )
r9  r™   r   r@  rñ   r3   rX   rW   rn   rÑ   s
             r   r2  r2  n  s–   € Ü�[‰[˜Ÿ™Ó*¨CÓ0€Fà ò +‰
ˆˆUØ×(Ñ(¨Ó-ˆØ—]‘]“_ò 	+ˆEØ—:‘:˜eÓ$ˆDØ%Ÿ_™_¨S°%Ó8ò +‘	��TØ�s“˜t d™{Ñ*”ñ+ñ	+ð+ð €Mr   c           
      óx  — |j                  dd«       t        |«      }|€t        j                  | |¬«      }t	        | |«      t        | «      z  }t        | |«      }t        j                  t        ||j                  ¬«      d«      }t        j                  |t        |«      df«      }	t        t        j                  |dk(  «      d   «      }
t        j                  t        |«      d«      }|
D ]  }t        j                  ||<   Œ t!        ||«      }|dkD  rt#        d|d   z  «       |d	kD  r t#        «        t#        d
«       t#        d«       	 	 |d	kD  rQ|j$                  xs t'        || «      }|j(                  xs t+        || «      }|j,                  }t#        d|||fz  «       t/        | ||
||||	|«      }|j1                  «       }||z  }|j3                  |«       |j5                  || «      rnŒ¢	 |d	kD  r+t'        || «      }t+        || «      }t#        d|d›d|d›�«       |S # t6        $ r t#        d«       Y ŒHw xY w)a‚  
    Train a new ``ConditionalExponentialClassifier``, using the given
    training samples, using the Improved Iterative Scaling algorithm.
    This ``ConditionalExponentialClassifier`` will encode the model
    that maximizes entropy from all the models that are empirically
    consistent with ``train_toks``.

    :see: ``train_maxent_classifier()`` for parameter descriptions.
    r‚   r  r  )r=   r  rR   r   r  r;   r  r   r!  r"  r#  r$  r%  r&  )r'  r   rÃ   r–   r(  r   Úcalculate_nfmapr)  ÚarrayrZ   Ú__getitem__ÚreshaperÇ   r*  r+  r,  r-  r\   r/  r   r0  r   r1  Úcalculate_deltasr   r&   r3  r4  )r™   rŽ   r   r"   r›   r5  Úempirical_ffreqÚnfmapÚnfarrayÚnftransposer8  r   rn   r9  r/  r0  r;  Údeltass                     r   r’   r’   €  sQ  € ð ×Ñ�z 3Ô'Ü! 'Ó*€Mð ÐÜ.×4Ñ4°ZÈÐ4ÓOˆô 1°¸XÓFÌÈZËÑX€Oô ˜J¨Ó1€EÜ�k‰kœ& ¨E×,=Ñ,=Ô>ÀÓD€GÜ—-‘- ¬#¨g«,¸Ð):Ó;€Kô ”U—]‘] ?°aÑ#7Ó8¸Ñ;Ó<€Jô �k‰kœ#˜oÓ.°Ó4€GØò "ˆÜ—z‘zˆ�Šð"ä1°(¸GÓD€Jàˆq‚yÜÐ.°¸Ñ1DÑDÔEØˆq‚yÜŒÜÐ=Ô>ÜÐ=Ô>ð<ØØ�qŠyØ"×%Ñ%ÒO¬¸
ÀJÓ)O�Ø#×'Ñ'ÒK¬8°JÀ
Ó+K�Ø'×,Ñ,�ÜÐ3°wÀÀCÐ6HÑHÔIô &ØØØØØØØØó	ˆFð !×(Ñ(Ó*ˆGØ�vÑˆGØ×"Ñ" 7Ô+ð ×"Ñ" :¨zÔ:Øð5 ð4 ð
 ˆq‚yÜ˜J¨
Ó3ˆÜ�z :Ó.ˆÜÐ" 2 e *¨D°°T°
Ð;Ô<ð Ðøô ò <ÜÐ:Ö;ð<ús   ÅB#H" È"H9È8H9c                 ó   — t        «       }| D ]K  \  }}|j                  «       D ]3  }|j                  t        d„ |j	                  ||«      D «       «      «       Œ5 ŒM t        |«      D ��ci c]  \  }}||“Œ
 c}}S c c}}w )aó  
    Construct a map that can be used to compress ``nf`` (which is
    typically sparse).

    *nf(feature_vector)* is the sum of the feature values for
    *feature_vector*.

    This represents the number of features that are active for a
    given labeled text.  This method finds all values of *nf(t)*
    that are attested for at least one token in the given list of
    training tokens; and constructs a dictionary mapping these
    attested values to a continuous range *0...N*.  For example,
    if the only values of *nf()* that were attested were 3, 5, and
    7, then ``_nfmap`` might return the dictionary ``{3:0, 5:1, 7:2}``.

    :return: A map that can be used to compress ``nf`` to a dense
        vector.
    :rtype: dict(int -> int)
    c              3   ó&   K  — | ]	  \  }}|–— Œ y ­wr!   rC   ©rD   ÚidrB  s      r   rF   z"calculate_nfmap.<locals>.<genexpr>ò  s   è ø€ ÒK¡) 2 sœ#ÑKùrÿ   )rÇ   r"   rì   r   r1   r`   )r™   r   Únfsetrñ   Ú_r3   rg   Únfs           r   rE  rE  Ú  s}   € ô* ‹E€EØò M‰ˆˆQØ—_‘_Ó&ò 	MˆEØ�I‰I”cÑK¨x¯©¸sÀEÓ/JÔKÓKÕLñ	MðMô "+¨5Ó!1×2‘g�q˜"ˆB�‰EÓ2Ð2ùÓ2s   Á)A:c           	      ón  — d}d}	t        j                  |j                  «       d«      }
t        j                  t	        |«      |j                  «       fd«      }| D ]}  \  }}|j                  |«      }|j                  «       D ]T  }|j                  ||«      }t        d„ |D «       «      }|D ])  \  }}|||   |fxx   |j                  |«      |z  z  cc<   Œ+ ŒV Œ |t	        | «      z  }t        |	«      D ]¿  }t        j                  ||
«      }d|z  }||z  }t        j                  ||z  d¬«      }t        j                  ||z  d¬«      }|D ]  }||xx   dz  cc<   Œ |
||z
  | z  z  }
t        j                  t        ||z
  «      «      t        j                  t        |
«      «      z  }||k  sŒ½|
c S  |
S )	a
  
    Calculate the update values for the classifier weights for
    this iteration of IIS.  These update weights are the value of
    ``delta`` that solves the equation::

      ffreq_empirical[i]
             =
      SUM[fs,l] (classifier.prob_classify(fs).prob(l) *
                 feature_vector(fs,l)[i] *
                 exp(delta[i] * nf(feature_vector(fs,l))))

    Where:
        - *(fs,l)* is a (featureset, label) tuple from ``train_toks``
        - *feature_vector(fs,l)* = ``encoding.encode(fs,l)``
        - *nf(vector)* = ``sum([val for (id,val) in vector])``

    This method uses Newton's method to solve this equation for
    *delta[i]*.  In particular, it starts with a guess of
    ``delta[i]`` = 1; and iteratively updates ``delta`` with:

    | delta[i] -= (ffreq_empirical[i] - sum1[i])/(-sum2[i])

    until convergence, where *sum1* and *sum2* are defined as:

    |    sum1[i](delta) = SUM[fs,l] f[i](fs,l,delta)
    |    sum2[i](delta) = SUM[fs,l] (f[i](fs,l,delta).nf(feature_vector(fs,l)))
    |    f[i](fs,l,delta) = (classifier.prob_classify(fs).prob(l) .
    |                        feature_vector(fs,l)[i] .
    |                        exp(delta[i] . nf(feature_vector(fs,l))))

    Note that *sum1* and *sum2* depend on ``delta``; so they need
    to be re-computed each iteration.

    The variables ``nfmap``, ``nfarray``, and ``nftranspose`` are
    used to generate a dense encoding for *nf(ltext)*.  This
    allows ``_deltas`` to calculate *sum1* and *sum2* using
    matrices, which yields a significant performance improvement.

    :param train_toks: The set of training tokens.
    :type train_toks: list(tuple(dict, str))
    :param classifier: The current classifier.
    :type classifier: ClassifierI
    :param ffreq_empirical: An array containing the empirical
        frequency for each feature.  The *i*\ th element of this
        array is the empirical frequency for feature *i*.
    :type ffreq_empirical: sequence of float
    :param unattested: An array that is 1 for features that are
        not attested in the training data; and 0 for features that
        are attested.  In other words, ``unattested[i]==0`` iff
        ``ffreq_empirical[i]==0``.
    :type unattested: sequence of int
    :param nfmap: A map that can be used to compress ``nf`` to a dense
        vector.
    :type nfmap: dict(int -> int)
    :param nfarray: An array that can be used to uncompress ``nf``
        from a dense vector.
    :type nfarray: array(float)
    :param nftranspose: The transpose of ``nfarray``
    :type nftranspose: array(float)
    gê-�™—q=i,  r  c              3   ó&   K  — | ]	  \  }}|–— Œ y ­wr!   rC   rQ  s      r   rF   z#calculate_deltas.<locals>.<genexpr>P  s   è ø€ Ò9™Y˜b #”SÑ9ùrÿ   r;   r   )ÚaxisrR   )r)  Úonesr   r+  r   r)   r"   r1   r   rW   rq   ÚouterrK   )r™   r9  r8  Úffreq_empiricalrK  rL  rM  r   ÚNEWTON_CONVERGEÚ
MAX_NEWTONrN  ÚArñ   r3   Údistr4   rU  rR  rB  ÚrangenumÚnf_deltaÚexp_nf_deltaÚnf_exp_nf_deltaÚsum1Úsum2rn   Ún_errors                              r   rI  rI  ö  sÃ  € ðR €OØ€Jä�Z‰Z˜Ÿ™Ó)¨3Ó/€Fô
 	�‰”S˜“Z §¡Ó!2Ð3°SÓ9€Aà ò 
;‰
ˆˆUØ×'Ñ'¨Ó,ˆà—_‘_Ó&ò 	;ˆEà%Ÿ_™_¨S°%Ó8ˆNäÑ9¨.Ô9Ó9ˆBà)ò ;‘��CØ�%˜‘)˜R�-Ó  D§I¡I¨eÓ$4°sÑ$:Ñ:Ô ñ;ñ	;ð
;ð ŒˆZ‹Ñ€Aô ˜*Ó%ò ˆÜ—;‘;˜w¨Ó/ˆØ˜(‘{ˆØ%¨Ñ4ˆÜ�y‰y˜¨Ñ)°Ô2ˆÜ�y‰y˜¨1Ñ,°1Ô5ˆð ò 	ˆCØ�‹I˜‰NŒIð	ð 	�? TÑ)¨d¨UÑ2Ñ2ˆô —)‘)œC °$Ñ 6Ó7Ó8¼5¿9¹9ÄSÈÃ[Ó;QÑQˆØ�_Ó$ØŠMð#ð& €Mr   c                 óÜ  — d}d}d|v r|d   }d|v r|d   }|€,|j                  dd«      }t        j                  | ||d¬«      }n|�t        d«      ‚	 t	        j
                  d	¬
«      \  }	}
t        |
d«      5 }t        | ||||¬«       ddd«       t        j                  |	«       g }|g d¢z  }|r|dgz  }|s|dgz  }|r	d|dz  z  }nd}|dd|z  dgz  }|dk  r|dgz  }d|v r|dd|d   z  gz  }d|v r|ddt        |d   «      z  gz  }t        |d«      r|dgz  }|d|
gz  }t        |«      }	 t        j                  |
«       t!        ||j#                  «       |«      }|t%        j&                  t$        j(                  «      z  }t+        ||«      S # 1 sw Y   �ŒxY w# t        t        f$ r}t        d|z  «      |‚d}~ww xY w# t        $ r}t        d |
› d!|› �«       Y d}~ŒŸd}~ww xY w)"a›  
    Train a new ``ConditionalExponentialClassifier``, using the given
    training samples, using the external ``megam`` library.  This
    ``ConditionalExponentialClassifier`` will encode the model that
    maximizes entropy from all the models that are empirically
    consistent with ``train_toks``.

    :see: ``train_maxent_classifier()`` for parameter descriptions.
    :see: ``nltk.classify.megam``
    Tr‰   rŠ   Nr‡   r   )r"   rÏ   z$Specify encoding or labels, not bothznltk-©ÚprefixÚw)r‰   rŠ   z,Error while creating megam training file: %s)z-nobiasz-repeatÚ10z	-explicitz-fvalsr.   r;   z-lambdaz%.2fz-tuner    z-quietr‚   z-maxirA   Úll_deltaz-dppÚcostz-multilabelÚ
multiclasszWarning: unable to delete z: )ÚgetrÃ   r–   r—   ÚtempfileÚmkstempÚopenr   ÚosÚcloseÚOSErrorrK   ro   r   Úremover\   r   r   r)  r.  Úer   )r™   rŽ   r   r"   r�   rœ   r‰   rŠ   r‡   ÚfdÚtrainfile_nameÚ	trainfilerw  rî   Úinv_varianceÚstdoutr   s                    r   r”   r”   }  sq  € ð €HØ€IØ�VÑØ˜*Ñ%ˆØ�fÑØ˜;Ñ'ˆ	ð Ðð —z‘z .°!Ó4ˆÜ.×4Ñ4Ø˜¨VÀtð 5ó 
‰ð 
Ð	ÜÐ?Ó@Ð@ðTÜ%×-Ñ-°WÔ=ÑˆˆNÜ�. #Ó&ð 	¨)ÜØ˜H i¸(Èiõ÷	ô 	�‰�Œð
 €GØÒ+Ñ+€GÙØ�K�=Ñ ˆÙØ�H�:ÑˆÙð Ð1°1Ñ4Ñ4‰àˆØ�	˜6 LÑ0°'Ð:Ñ:€GØˆq‚yØ�H�:ÑˆØ�VÑØ�G˜T F¨:Ñ$6Ñ6Ð7Ñ7ˆØ�VÑð 	�F˜D¤3 v¨jÑ'9Ó#:Ñ:Ð;Ñ;ˆÜˆx˜Ô Ø�M�?Ñ"ˆØ�˜nÐ-Ñ-€GÜ˜Ó €FðBÜ
�	‰	�.Ô!ô
 " &¨(¯/©/Ó*;¸XÓF€Gð Œu�z‰zœ%Ÿ'™'Ó"Ñ"€Gô ˜H gÓ.Ð.÷c	ñ 	ûô
 ”ZÐ ò TÜÐGÈ!ÑKÓLÐRSÐSûðTûôD ò BÜÐ*¨>Ð*:¸"¸Q¸CÐ@×AÑAûðBúsH   Á%F! Á8FÂ	F! Ä2G ÆFÆF! Æ!GÆ0F?Æ?GÇ	G+ÇG&Ç&G+c                   ó   — e Zd Zed„ «       Zy)r•   c                 ó¾  — |j                  dd«      }|j                  dd«      }|j                  dd «      }|j                  dd «      }|j                  dd«      }|j                  d	d«      }|j                  d
«      }	|j                  d«      }
|st        j                  |||¬«      }t        j                  dd¬«      \  }}t        j                  d¬«      \  }}t        |d«      }t        |||«       |j                  «        g }|j                  dg«       |j                  d|g«       |r|j                  dd|dz  z  g«       |	r|j                  dd|	z  g«       |
r|j                  ddt        |
«      z  g«       |j                  d|g«       |j                  d|g«       |dk  r|j                  dg«       n|j                  dg«       t        |«       t        |«      5 }t        |«      }d d d «       t        j                  |«       t        j                  |«       t        j                   t        j"                  «      z  } | ||«      S # 1 sw Y   ŒbxY w)Nrš   Útao_lmvmrŽ   r    r   r"   r�   r   r‡   r‚   r„   r  znltk-tadm-events-z.gz)ri  Úsuffixznltk-tadm-weights-rh  rj  z-monitorz-methodz-l2z%.6fr;   z-max_itz%dz-fatolz
-events_inz-params_outz2>&1z-summary)ro  r  r–   rp  rq  r   r
   rt  ÚextendrK   r   rr  r	   rs  rv  r)  r.  rw  )r˜   r™   rœ   rš   rŽ   r   r"   Úsigmar‡   r‚   rl  Útrainfile_fdry  Úweightfile_fdÚweightfile_namerz  rî   Ú
weightfiler   s                      r   r–   zTadmMaxentClassifier.trainÚ  s$  € à—J‘J˜{¨JÓ7ˆ	Ø—
‘
˜7 AÓ&ˆØ—:‘:˜j¨$Ó/ˆØ—‘˜H dÓ+ˆØ—
‘
Ð1°1Ó5ˆØ—z‘z .°!Ó4ˆØ—:‘:˜jÓ)ˆØ—:‘:˜mÓ,ˆñ Ü5×;Ñ;Ø˜L°ð <ó ˆHô (0×'7Ñ'7Ø&¨uô(
Ñ$ˆ�nô *2×)9Ñ)9ÐAUÔ)VÑ&ˆ�ä% n°cÓ:ˆ	Ü˜
 H¨iÔ8Ø�‰ÔàˆØ�‰˜
�|Ô$Ø�‰˜	 9Ð-Ô.ÙØ�N‰N˜E 6¨E°1©HÑ#4Ð5Ô6ÙØ�N‰N˜I t¨h¡Ð7Ô8ÙØ�N‰N˜H f¬s°8«}Ñ&<Ð=Ô>Ø�‰˜ nÐ5Ô6Ø�‰˜ Ð7Ô8Ø�1Š9Ø�N‰N˜F˜8Õ$à�N‰N˜J˜<Ô(ä�'Ôä�/Ó"ð 	5 jÜ(¨Ó4ˆG÷	5ô 	�	‰	�.Ô!Ü
�	‰	�/Ô"ð 	”5—:‘:œeŸg™gÓ&Ñ&ˆñ �8˜WÓ%Ð%÷	5ð 	5ús   Ç&IÉIN)r¡   r¢   r£   r¦   r–   rC   r   r   r•   r•   Ù  s   „ Øñ5&ó ñ5&r   r•   c                 ó  — dd l }ddlm} ddlm}  |«       } || d«      5 } |j
                  t        t        |j                  |j                  |«      «      «      «      }d d d «        || d«      5 }|j                  |«      }d d d «        || d«      5 }|j                  |«      }d d d «        || d«      5 }|j                  |«      }	d d d «       	fS # 1 sw Y   ŒxxY w# 1 sw Y   ŒaxY w# 1 sw Y   ŒJxY w# 1 sw Y   Œ3xY w)Nr   )Úopen_datafile)ÚMaxentDecoderzweights.txtzmapping.tabz
labels.txtzalwayson.tab)r)  Ú	nltk.datarˆ  Únltk.tabdatar‰  rF  rp   ÚmapÚfloat64Útxt2listÚtupkey2dictÚ
tab2ivdict)
Útab_dirr)  rˆ  r‰  Úmdecrý   ÚwgtÚmpgÚlabÚaons
             r   Úload_maxent_paramsr—    s  € Ûå'Ý*á‹?€Dá	�w Ó	.ð F°!Øˆe�k‰kœ$œs 5§=¡=°$·-±-ÀÓ2BÓCÓDÓEˆ÷Fñ 
�w Ó	.ð "°!Ø×Ñ˜qÓ!ˆ÷"ñ 
�w Ó	-ð °Ø�m‰m˜AÓˆ÷ñ 
�w Ó	/ð !°1Ø�o‰o˜aÓ ˆ÷!ð ��S˜#ÐÐ÷Fð Fú÷"ð "ú÷ð ú÷!ð !ús/   ¡?CÁ1C#ÂC/Â7C;ÃC Ã#C,Ã/C8Ã;Dc           
      óÒ  — ddl m} ddlm} ddlm}  |«       } ||«      s ||«       t        d|› �«       t        |› d�d«      5 }	|	j                  |j                  t        t        | j                  «       «      «      › «       d d d «       t        |› d�d«      5 }	|	j                  |j                  |«      › «       d d d «       t        |› d	�d«      5 }	|	j                  |j                  |«      › «       d d d «       t        |› d
�d«      5 }	|	j                  |j                  |«      › «       d d d «       y # 1 sw Y   ŒµxY w# 1 sw Y   ŒˆxY w# 1 sw Y   Œ[xY w# 1 sw Y   y xY w)Nr   )Úmkdir)Úisdir)ÚMaxentEncoderzSaving Maxent parameters in z/weights.txtrj  z/mapping.tabz/labels.txtz/alwayson.tab)rs  r™  Úos.pathrš  r‹  r›  r\   rr  ÚwriteÚlist2txtrŒ  ÚreprÚtolistÚtupdict2tabÚ
ivdict2tab)
r“  r”  r•  r–  r‘  r™  rš  r›  Úmencrý   s
             r   Úsave_maxent_paramsr¤  /  sD  € åÝå*á‹?€DÙ�Œ>ÙˆgŒä	Ð(¨¨	Ð
2Ô3ä	��	˜Ð&¨Ó	,ð =°Ø	�‰�4—=‘=¤¤T¨3¯:©:«<Ó!8Ó9Ð:Ô<÷=ä	��	˜Ð&¨Ó	,ð ,°Ø	�‰�4×#Ñ# CÓ(Ð)Ô+÷,ä	��	˜Ð% sÓ	+ð )¨qØ	�‰�4—=‘= Ó%Ð&Ô(÷)ä	��	˜Ð'¨Ó	-ð +°Ø	�‰�4—?‘? 3Ó'Ð(Ô*÷+ð +÷=ð =ú÷,ð ,ú÷)ð )ú÷+ð +ús0   Á>D9Â"EÃ"EÄ"EÄ9EÅEÅEÅE&c                  óŒ   — ddl m}  ddlm}  | d«      }t	        |«      \  }}}}t        t        |||¬«      |«      } ||¬«      S )Nr   )Úfind)ÚClassifierBasedPOSTaggerz/taggers/maxent_treebank_pos_tagger_tab/english/)rÏ   )r9  )rŠ  r¦  Únltk.tag.sequentialr§  r—  r   rÃ   )r¦  r§  r‘  r“  r”  r•  r–  Úmcs           r   Úmaxent_pos_taggerrª  F  sK   € ÝÝ<áÐDÓE€GÜ+¨GÓ4Ñ€Cˆˆc�3Ü	Ü# C¨ÀÔDÀcó
€Bñ $¨rÔ2Ð2r   c                  ó<   — ddl m}   | t        j                  «      }y )Nr   )Ú
names_demo)Únltk.classify.utilr¬  r   r–   )r¬  r9  s     r   Údemor®  U  s   € Ý-áÔ,×2Ñ2Ó3�Jr   Ú__main__)r    NN)r    NNr   )z/tmp)1r¤   r)  ÚImportErrorrs  rp  Úcollectionsr   Únltk.classify.apir   Únltk.classify.megamr   r   r   Únltk.classify.tadmr   r	   r
   r­  r   r   r   rŠ  r   Únltk.probabilityr   Ú	nltk.utilr   Ú__docformat__r   r-  r¨   rµ   rÃ   rö   r  r  r“   r(  r2  r’   rE  rI  r”   r•   r—  r¤  rª  r®  r¡   rC   r   r   ú<module>r¸     s:  ðñ,ðZ	Ûó 
Û Ý #å )ß QÑ Qß MÑ Mß FÑ FÝ 'Ý /Ý !à€ôJA�{ô JAð\ $4Ð  ÷F$ñ F$ôR,*Ð*@ô ,*ô^G/Ð"8ô G/ôT:DÐ-ô :Dôz5/Ð%@ô 5/ôpa/Ð!7ô a/ðT 04ó]ò@ò
ð& 04óWòt3ò8{ðP KLóT/ôx7&Ð+ô 7&ò~ó.+ò.	3ò4ð ˆzÒÙ…Fð øðG1 ò 	Ùð	ús   „C ÃC%Ã$C%