Ë
    çÍ:jæ  ã                   óT   — d Z ddlmZ ddlmZ ddlmZ  G d„ d«      Zd„ Zd„ Z	dd
„Z
y	)aˆ  
Simple classifier for RTE corpus.

It calculates the overlap in words and named entities between text and
hypothesis, and also whether there are words / named entities in the
hypothesis which fail to occur in the text, since this is an indicator that
the hypothesis is more informative than (i.e not entailed by) the text.

TO DO: better Named Entity classification
TO DO: add lemmatization
é    )ÚMaxentClassifier)Úaccuracy)ÚRegexpTokenizerc                   óH   — e Zd ZdZdd„Zd	d„Zd
d„Zed„ «       Zed„ «       Z	y)ÚRTEFeatureExtractorz™
    This builds a bag of words for both the text and the hypothesis after
    throwing away some stopwords, then calculates overlap and difference.
    c                 óH  — || _         h d£| _        h d£| _        t        d«      }|j	                  |j
                  «      | _        |j	                  |j                  «      | _        t        | j                  «      | _
        t        | j                  «      | _        |r\| j                  D �ch c]  }| j                  |«      ’Œ c}| _
        | j                  D �ch c]  }| j                  |«      ’Œ c}| _        | j                   r<| j                  | j                  z
  | _
        | j                  | j                  z
  | _        | j                  | j                  z  | _        | j                  | j                  z
  | _        | j                  | j                  z
  | _        yc c}w c c}w )z­
        :param rtepair: a ``RTEPair`` from which features should be extracted
        :param stop: if ``True``, stopwords are thrown away.
        :type stop: bool
        >   ÚaÚinÚisÚitÚofÚtoÚandÚareÚtheÚhaveÚtheyÚveryÚwereú,ú.>   ÚnoÚnotÚneverÚdeniedÚfailedÚrejectedz[\w.@:/]+|\w+|\$[\d.]+N)ÚstopÚ	stopwordsÚnegwordsr   ÚtokenizeÚtextÚtext_tokensÚhypÚ
hyp_tokensÚsetÚ
text_wordsÚ	hyp_wordsÚ
_lemmatizeÚ_overlapÚ
_hyp_extraÚ
_txt_extra)ÚselfÚrtepairr   Úuse_lemmatizeÚ	tokenizerÚtokens         úo/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/classify/rte_classify.pyÚ__init__zRTEFeatureExtractor.__init__   s3  € ð ˆŒ	ò
ˆŒò$ OˆŒô $Ð$=Ó>ˆ	ð %×-Ñ-¨g¯l©lÓ;ˆÔØ#×,Ñ,¨W¯[©[Ó9ˆŒÜ˜d×.Ñ.Ó/ˆŒÜ˜TŸ_™_Ó-ˆŒáØCG×CSÑCSÖT¸%˜tŸ™¨uÕ5ÒTˆDŒOØBFÇ/Á/ÖR¸˜dŸo™o¨eÕ4ÒRˆDŒNà�9Š9Ø"Ÿo™o°·±Ñ>ˆDŒOØ!Ÿ^™^¨d¯n©nÑ<ˆDŒNàŸ™¨¯©Ñ8ˆŒØŸ.™.¨4¯?©?Ñ:ˆŒØŸ/™/¨D¯N©NÑ:ˆ�ùò UùÚRs   Â)FÃFc                 ó  — | j                   D �ch c]  }| j                  |«      sŒ|’Œ }}|dk(  r|rt        d|«       |S |dk(  r*|rt        d| j                   |z
  «       | j                   |z
  S t        d|z  «      ‚c c}w )z°
        Compute the overlap between text and hypothesis.

        :param toktype: distinguish Named Entities from ordinary words
        :type toktype: 'ne' or 'word'
        Únez
ne overlapÚwordzword overlapzType not recognized:'%s')r*   Ú_neÚprintÚ
ValueError)r-   ÚtoktypeÚdebugr1   Ú
ne_overlaps        r2   ÚoverlapzRTEFeatureExtractor.overlapO   sƒ   € ð *.¯©ÖJ ¸$¿(¹(À5½/’eÐJˆ
ÐJØ�dŠ?ÙÜ�l JÔ/ØÐØ˜ÒÙÜ�n d§m¡m°jÑ&@ÔAØ—=‘= :Ñ-Ð-äÐ7¸'ÑAÓBÐBùò Ks
   �A>¦A>c                 ó´   — | j                   D �ch c]  }| j                  |«      sŒ|’Œ }}|dk(  r|S |dk(  r| j                   |z
  S t        d|z  «      ‚c c}w )z²
        Compute the extraneous material in the hypothesis.

        :param toktype: distinguish Named Entities from ordinary words
        :type toktype: 'ne' or 'word'
        r5   r6   zType not recognized: '%s')r+   r7   r9   )r-   r:   r;   r1   Úne_extras        r2   Ú	hyp_extrazRTEFeatureExtractor.hyp_extrab   s_   € ð (,§¡ÖJ˜e¸$¿(¹(À5½/’EÐJˆÐJØ�dŠ?ØˆOØ˜ÒØ—?‘? XÑ-Ð-äÐ8¸7ÑBÓCÐCùò Ks
   �A¦Ac                 óF   — | j                  «       s| j                  «       ryy)zz
        This just assumes that words in all caps or titles are
        named entities.

        :type token: str
        TF)ÚistitleÚisupper)r1   s    r2   r7   zRTEFeatureExtractor._neq   s   € ð �=‰=Œ?˜eŸm™mœoØØó    c                 óT   — ddl m} |j                  | |j                  ¬«      }|�|S | S )zI
        Use morphy from WordNet to find the base form of verbs.
        r   )Úwordnet)Úpos)Únltk.corpusrF   ÚmorphyÚVERB)r6   ÚwnÚlemmas      r2   r)   zRTEFeatureExtractor._lemmatize}   s-   € õ
 	.à—	‘	˜$ B§G¡G�	Ó,ˆØÐØˆLØˆrD   N)TF)F)T)
Ú__name__Ú
__module__Ú__qualname__Ú__doc__r3   r=   r@   Ústaticmethodr7   r)   © rD   r2   r   r      sA   „ ñó
.;ó`Có&Dð ñ	ó ð	ð ñ	ó ñ	rD   r   c                 ó¦  — t        | «      }i }d|d<   t        |j                  d«      «      |d<   t        |j                  d«      «      |d<   t        |j                  d«      «      |d<   t        |j                  d«      «      |d<   t        |j                  |j
                  z  «      |d	<   t        |j                  |j                  z  «      |d
<   |S )NTÚalwaysonr6   Úword_overlapÚword_hyp_extrar5   r<   Úne_hyp_extraÚneg_txtÚneg_hyp)r   Úlenr=   r@   r    r'   r(   )r.   Ú	extractorÚfeaturess      r2   Úrte_featuresr]   Š   sÉ   € Ü# GÓ,€IØ€HØ€HˆZÑÜ" 9×#4Ñ#4°VÓ#<Ó=€Hˆ^ÑÜ!$ Y×%8Ñ%8¸Ó%@Ó!A€HÐÑÜ  ×!2Ñ!2°4Ó!8Ó9€Hˆ\ÑÜ" 9×#6Ñ#6°tÓ#<Ó=€Hˆ^ÑÜ˜i×0Ñ0°9×3GÑ3GÑGÓH€HˆYÑÜ˜i×0Ñ0°9×3FÑ3FÑFÓG€HˆYÑØ€OrD   c                 óV   — | D �cg c]  }t        |«      |j                  f‘Œ c}S c c}w ©N)r]   Úvalue)Ú	rte_pairsÚpairs     r2   Úrte_featurizerc   —   s$   € Ø9BÖC°Œ\˜$Ó §¡Ò,ÒCÐCùÒCs   …&Nc                 óš  — ddl m} |j                  g d¢«      }|j                  g d¢«      }|�
|d | }|d | }t        |«      }t        |«      }t	        d«       | dv rt        j                  || «      }n1| dv rt        j                  || «      }nt        d«      }t        |«      ‚t	        d	«       t        ||«      }	t	        d
|	z  «       |S )Nr   )Úrte)zrte1_dev.xmlzrte2_dev.xmlzrte3_dev.xml)zrte1_test.xmlzrte2_test.xmlzrte3_test.xmlzTraining classifier...)Úmegam)ÚGISÚIISzFRTEClassifier only supports these algorithms:
 'megam', 'GIS', 'IIS'.
zTesting classifier...zAccuracy: %6.4f)
rH   re   Úpairsrc   r8   r   ÚtrainÚstrÚ	Exceptionr   )
Ú	algorithmÚsample_NÚ
rte_corpusÚ	train_setÚtest_setÚfeaturized_train_setÚfeaturized_test_setÚclfÚerr_msgÚaccs
             r2   Úrte_classifierrw   ›   sá   € Ý-à× Ñ Ò!QÓR€IØ×ÑÒ SÓT€HàÐØ˜i˜xÐ(ˆ	Ø˜I˜XÐ&ˆä(¨Ó3ÐÜ'¨Ó1Ðô 
Ð
"Ô#Ø�IÑÜ×$Ñ$Ð%9¸9ÓE‰Ø	�nÑ	$Ü×$Ñ$Ð%9¸9ÓE‰äð'ó
ˆô ˜Ó Ð Ü	Ð
!Ô"Ü
�3Ð+Ó
,€CÜ	Ð
˜cÑ
!Ô"Ø€JrD   r_   )rP   Únltk.classify.maxentr   Únltk.classify.utilr   Únltk.tokenizer   r   r]   rc   rw   rR   rD   r2   ú<module>r{      s2   ðñ
õ 2Ý 'Ý )÷nñ nòb
òDôrD   