Ë
    çÍ:j„:  ã                   ó<  — d dl Z d dlZd dlZd dlZd dlmZ d dlmZ d dlm	Z	m
Z
mZ d dlmZmZ d dlmZmZ d„ Zd„ Zd	„ Zd
„ Zd„ Zd„ Zd„ Zd„ Zd„ Zd„ Zd„ Zd„ Z	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 dd„Zd„ Zdd„Z  e
ddg«      Z! e
g d¢«      Z"d„ Z#e$dk(  r e«        yy)é    N)Útreebank)Úpickle_load)ÚBrillTaggerTrainerÚRegexpTaggerÚUnigramTagger)ÚPosÚWord)ÚTemplateÚ
error_listc                  ó   — t        «        y)z„
    Run a demo with defaults. See source comments for details,
    or docstrings of any of the more specific demo_* functions.
    N©Úpostag© ó    úb/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/tbl/demo.pyÚdemor      s	   € ô
 …Hr   c                  ó   — t        d¬«       y)úN
    Exemplify repr(Rule) (see also str(Rule) and Rule.format("verbose"))
    Úrepr©Ú
ruleformatNr   r   r   r   Údemo_repr_rule_formatr      s   € ô �fÖr   c                  ó   — t        d¬«       y)r   Ústrr   Nr   r   r   r   Údemo_str_rule_formatr   %   s   € ô �eÖr   c                  ó   — t        d¬«       y)z*
    Exemplify Rule.format("verbose")
    Úverboser   Nr   r   r   r   Údemo_verbose_rule_formatr   ,   s   € ô �iÖ r   c                  óF   — t        t        t        g d¢«      «      g¬«       y)a¾  
    The feature/s of a template takes a list of positions
    relative to the current word where the feature should be
    looked for, conceptually joined by logical OR. For instance,
    Pos([-1, 1]), given a value V, will hold whenever V is found
    one step to the left and/or one step to the right.

    For contiguous ranges, a 2-arg form giving inclusive end
    points can also be used: Pos(-3, -1) is the same as the arg
    below.
    )éýÿÿÿéþÿÿÿéÿÿÿÿ©Ú	templatesN)r   r
   r   r   r   r   Údemo_multiposition_featurer%   3   s   € ô ”hœs¢<Ó0Ó1Ð2Ö3r   c            	      ó\   — t        t        t        dg«      t        ddg«      «      g¬«       y)z8
    Templates can have more than a single feature.
    r   r!   r"   r#   N)r   r
   r	   r   r   r   r   Údemo_multifeature_templater'   B   s$   € ô ”hœt Q C›y¬#¨r°2¨h«-Ó8Ð9Ö:r   c                  ó   — t        dd¬«       y)ah  
    Show aggregate statistics per template. Little used templates are
    candidates for deletion, much used templates may possibly be refined.

    Deleting unused templates is mostly about saving time and/or space:
    training is basically O(T) in the number of templates T
    (also in terms of memory usage, which often will be the limiting factor).
    T)Úincremental_statsÚtemplate_statsNr   r   r   r   Údemo_template_statisticsr+   I   s   € ô ˜T°$Ö7r   c                  ó  — t        j                  g d¢ddgd¬«      } t        j                  g d¢ddgd¬«      }t        t	        j                  | |gd¬	«      «      }t        d
j                  t        |«      «      «       t        |dd¬«       y)a	  
    Template.expand and Feature.expand are class methods facilitating
    generating large amounts of templates. See their documentation for
    details.

    Note: training with 500 templates can easily fill all available
    even on relatively small corpora
    )r"   r   é   r-   é   F)Úexcludezero)r!   r"   r   r-   T)r-   é   )Úcombinationsz8Generated {} templates for transformation-based learning)r$   r)   r*   N)	r	   Úexpandr   Úlistr
   ÚprintÚformatÚlenr   )ÚwordtplsÚtagtplsr$   s      r   Údemo_generated_templatesr9   U   su   € ô �{‰{š:¨¨1 v¸5ÔA€HÜ�j‰jš¨!¨Q¨¸TÔB€GÜ”X—_‘_ h°Ð%8ÀvÔNÓO€IÜ	ØB×IÑIÜ�	‹Nó	
ôô
 �Y°$ÀtÖLr   c                  ó    — t        ddd¬«       y)z‚
    Plot a learning curve -- the contribution on tagging accuracy of
    the individual rules.
    Note: requires matplotlib
    Tzlearningcurve.png)r)   Úseparate_baseline_dataÚlearning_curve_outputNr   r   r   r   Údemo_learning_curver=   i   s   € ô ØØ#Ø1ör   c                  ó   — t        d¬«       y)zW
    Writes a file with context for each erroneous word after tagging testing data
    z
errors.txt)Úerror_outputNr   r   r   r   Údemo_error_analysisr@   v   s   € ô ˜Ö%r   c                  ó   — t        d¬«       y)zm
    Serializes the learned tagger to a file in pickle format; reloads it
    and validates the process.
    z
tagger.pcl)Úserialize_outputNr   r   r   r   Údemo_serialize_taggerrC   }   s   € ô
 ˜LÖ)r   c                  ó    — t        ddd¬«       y)z˜
    Discard rules with low accuracy. This may hurt performance a bit,
    but will often produce rules which are more interesting read to a human.
    i¸  g¸…ëQ¸î?é
   )Ú	num_sentsÚmin_accÚ	min_scoreNr   r   r   r   Údemo_high_accuracy_rulesrI   …   s   € ô
 �T 4°2Ö6r   c           	      óê  — |xs t         }| €ddlm}m}  |«       } t	        |||||«      \  }}}}|r t
        j                  j                  |«      sRt        ||¬«      }t        |d«      5 }t        j                  ||«       ddd«       t        dj                  |«      «       t        |d«      5 }t        |«      }t        d|› �«       ddd«       nt        ||¬«      }t        d	«       |r)t        d
j                  j                  |«      «      «       t!        j                   «       }t#        | ||	¬«      }t        d«       |j%                  ||||«      }t        dt!        j                   «       |z
  d›d�«       |rt        d|j                  |«      z  «       |dk(  rNt        d«       t'        |j)                  «       d«      D ]&  \  }}t        |d›d|j                  |	«      d›�«       Œ( |
r{t        d«       |j+                  ||«      \  } }!t        d«       |st        d«       |j-                  «       }"|r|j/                  |!«       |rLt1        ||!|"|¬«       t        d|› �«       n.t        d«       |j3                  |«      } |r|j/                  «        |�st        |d«      5 }#|#j5                  d|z  «       |#j5                  dj7                  t9        || «      «      j;                  d«      dz   «       ddd«       t        d |› �«       |�¦|j3                  |«      } t        |d«      5 }t        j                  ||«       ddd«       t        d!|› �«       t        |d«      5 }t        |«      }$ddd«       t        d|› �«       $j3                  |«      }%| |%k(  rt        d"«       yt        d#«       yy# 1 sw Y   �Œ8xY w# 1 sw Y   �ŒäxY w# 1 sw Y   ŒÚxY w# 1 sw Y   ŒšxY w# 1 sw Y   ŒxxY w)$a’
  
    Brill Tagger Demonstration
    :param templates: how many sentences of training and testing data to use
    :type templates: list of Template

    :param tagged_data: maximum number of rule instances to create
    :type tagged_data: C{int}

    :param num_sents: how many sentences of training and testing data to use
    :type num_sents: C{int}

    :param max_rules: maximum number of rule instances to create
    :type max_rules: C{int}

    :param min_score: the minimum score for a rule in order for it to be considered
    :type min_score: C{int}

    :param min_acc: the minimum score for a rule in order for it to be considered
    :type min_acc: C{float}

    :param train: the fraction of the the corpus to be used for training (1=all)
    :type train: C{float}

    :param trace: the level of diagnostic tracing output to produce (0-4)
    :type trace: C{int}

    :param randomize: whether the training data should be a random subset of the corpus
    :type randomize: C{bool}

    :param ruleformat: rule output format, one of "str", "repr", "verbose"
    :type ruleformat: C{str}

    :param incremental_stats: if true, will tag incrementally and collect stats for each rule (rather slow)
    :type incremental_stats: C{bool}

    :param template_stats: if true, will print per-template statistics collected in training and (optionally) testing
    :type template_stats: C{bool}

    :param error_output: the file where errors will be saved
    :type error_output: C{string}

    :param serialize_output: the file where the learned tbl tagger will be saved
    :type serialize_output: C{string}

    :param learning_curve_output: filename of plot of learning curve(s) (train and also test, if available)
    :type learning_curve_output: C{string}

    :param learning_curve_take: how many rules plotted
    :type learning_curve_take: C{int}

    :param baseline_backoff_tagger: the file where rules will be saved
    :type baseline_backoff_tagger: tagger

    :param separate_baseline_data: use a fraction of the training data exclusively for training baseline
    :type separate_baseline_data: C{bool}

    :param cache_baseline_tagger: cache baseline tagger to this file (only interesting as a temporary workaround to get
                                  deterministic output from the baseline unigram tagger between python versions)
    :type cache_baseline_tagger: C{string}


    Note on separate_baseline_data: if True, reuse training data both for baseline and rule learner. This
    is fast and fine for a demo, but is likely to generalize worse on unseen data.
    Also cannot be sensibly used for learning curves on training data (the baseline will be artificially high).
    Nr   )Úbrill24Údescribe_template_sets)ÚbackoffÚwbz)Trained baseline tagger, pickled it to {}ÚrbzReloaded pickled tagger from zTrained baseline taggerz!    Accuracy on test set: {:0.4f}r   zTraining tbl tagger...zTrained tbl tagger in z0.2fz secondsz    Accuracy on test set: %.4fr-   z
Learned rules: Ú4dú ÚszJIncrementally tagging the test data, collecting individual rule statisticsz    Rule statistics collectedzbWARNING: train_stats asked for separate_baseline_data=True; the baseline will be artificially high)Útakez Wrote plot of learning curve to zTagging the test dataÚwzErrors for Brill Tagger %r

ú
zutf-8z)Wrote tagger errors including context to zWrote pickled tagger to z4Reloaded tagger tried on test set, results identicalz;PROBLEM: Reloaded tagger gave different results on test set)ÚREGEXP_TAGGERÚnltk.tag.brillrK   rL   Ú_demo_prepare_dataÚosÚpathÚexistsr   ÚopenÚpickleÚdumpr4   r5   r   ÚaccuracyÚtimer   ÚtrainÚ	enumerateÚrulesÚbatch_tag_incrementalÚtrain_statsÚprint_template_statisticsÚ
_demo_plotÚ	tag_sentsÚwriteÚjoinr   Úencode)&r$   Útagged_datarF   Ú	max_rulesrH   rG   ra   ÚtraceÚ	randomizer   r)   r*   r?   rB   r<   Úlearning_curve_takeÚbaseline_backoff_taggerr;   Úcache_baseline_taggerrK   rL   Útraining_dataÚbaseline_dataÚ	gold_dataÚtesting_dataÚbaseline_taggerÚprint_rulesÚtbrillÚtrainerÚbrill_taggerÚrulenoÚruleÚ
taggedtestÚ	teststatsÚ
trainstatsÚfÚbrill_tagger_reloadedÚtaggedtest_reloadeds&                                         r   r   r   �   s  € ðp 6ÒF¼ÐØÐßBñ
 “Iˆ	Ü>PØ�U˜I yÐ2Hó?Ñ;€]�M 9¨lñ Ü�w‰w�~‰~Ð3Ô4Ü+ØÐ'>ôˆOô Ð+¨TÓ2ð :°kÜ—‘˜O¨[Ô9÷:äØ;×BÑBØ)óôô
 Ð'¨Ó.ð 	K°+Ü)¨+Ó6ˆOÜÐ1Ð2GÐ1HÐIÔJ÷	Kð 	Kô (¨Ð?VÔWˆÜÐ'Ô(ÙÜØ/×6Ñ6Ø×(Ñ(¨Ó3óô	
ô �Y‰Y‹[€FÜ Ø˜ E°jô€Gô 
Ð
"Ô#Ø—=‘= °	¸9ÀgÓN€LÜ	Ð"¤4§9¡9£;°Ñ#7¸Ð"=¸XÐ
FÔGÙÜÐ.°×1FÑ1FÀyÓ1QÑQÔRð �‚zÜÐ!Ô"Ü% l×&8Ñ&8Ó&:¸AÓ>ò 	>‰LˆF�DÜ�V˜B�K˜q §¡¨ZÓ!8¸Ð ;Ð<Õ=ð	>ñ
 ÜØXô	
ð #/×"DÑ"DØ˜)ó#
Ñˆ�Yô 	Ð-Ô.Ù%Üð,ôð "×-Ñ-Ó/ˆ
ÙØ×2Ñ2°9Ô=Ù ÜØ% y°*ÐCVõô Ð4Ð5JÐ4KÐLÕMäÐ%Ô&Ø!×+Ñ+¨LÓ9ˆ
ÙØ×2Ñ2Ô4ð ÐÜ�, Ó$ð 	Y¨Ø�G‰GÐ4Ð7GÑGÔHØ�G‰G�D—I‘Iœj¨°JÓ?Ó@×GÑGÈÓPÐSWÑWÔX÷	Yô 	Ð9¸,¸ÐHÔIð Ð#Ø!×+Ñ+¨LÓ9ˆ
ÜÐ" DÓ)ð 	3¨[Ü�K‰K˜ kÔ2÷	3äÐ(Ð)9Ð(:Ð;Ô<ÜÐ" DÓ)ð 	=¨[Ü$/°Ó$<Ð!÷	=äÐ-Ð.>Ð-?Ð@ÔAØ3×=Ñ=¸lÓKÐØÐ,Ò,ÜÐHÕIäÐOÕPð $÷U:ñ :ú÷	Kñ 	Kú÷z	Yð 	Yú÷	3ð 	3ú÷	=ð 	=ús=   Á*N7Â/OÊ'AOÌ-OÍ&O)Î7OÏOÏOÏO&Ï)O2c           	      ó˜  — | €t        d«       t        j                  «       } |�t        | «      |k  rt        | «      }|r3t	        j
                  t        | «      «       t	        j                  | «       t        ||z  «      }| d | }| || }|D ��	cg c]  }|D �	cg c]  }	|	d   ‘Œ	 c}	‘Œ }
}}	|s|}nt        |«      dz  }|d | ||d  }}t        |«      \  }}t        |
«      \  }}t        |«      \  }}t        d|d›d|d›d�«       t        d|d›d|d›d�«       t        d	j                  |||rd
nd«      «       ||||
fS c c}	w c c}	}w )Nz%Loading tagged data from treebank... r   r0   zRead testing data (Údz sents/z wds)zRead training data (z-Read baseline data ({:d} sents/{:d} wds) {:s}Ú z[reused the training set])
r4   r   Útagged_sentsr6   ÚrandomÚseedÚshuffleÚintÚcorpus_sizer5   )rl   ra   rF   ro   r;   Úcutoffrs   ru   ÚsentÚtrv   rt   Ú	bl_cutoffÚ	trainseqsÚtraintokensÚtestseqsÚ
testtokensÚbltrainseqsÚbltraintokenss                      r   rX   rX   R  s|  € ð
 ÐÜÐ5Ô6Ü×+Ñ+Ó-ˆØÐœC Ó,°	Ò9Ü˜Ó$ˆ	ÙÜ�‰”C˜Ó$Ô%Ü�‰�{Ô#Ü�˜UÑ"Ó#€FØ  Ð(€MØ˜F 9Ð-€IØ5>×?¨T 4Ö(˜a�Q�q“TÔ(Ð?€LÑ?Ù!Ø%‰ä˜Ó&¨!Ñ+ˆ	à˜*˜9Ð%Ø˜)˜*Ð%ð &ˆô  +¨=Ó9Ñ€Y�Ü(¨Ó6Ñ€XˆzÜ#.¨}Ó#=Ñ €[�-Ü	Ð ¨˜|¨7°:¸a°.ÀÐ
FÔGÜ	Ð  ¨1 ¨W°[À°OÀ5Ð
IÔJÜ	Ø7×>Ñ>ØØÙ(‰BÐ.Ió	
ôð ˜=¨)°\ÐBÐBùò+ )ùÓ?s   Â	EÂEÂ$EÅEc                 óÖ  — |d   g}|d   D ]  }|j                  |d   |z
  «       Œ |d | D �cg c]  }d||d   z  z
  ‘Œ }}|d   g}|d   D ]  }|j                  |d   |z
  «       Œ |d | D �cg c]  }d||d   z  z
  ‘Œ }}dd lm} t        t	        t        |«      «      «      }	|j                  |	||	|«       |j                  g d¢«       |j                  | «       y c c}w c c}w )NÚinitialerrorsÚ
rulescoresr"   r-   Ú
tokencountr   )NNNg      ð?)	ÚappendÚmatplotlib.pyplotÚpyplotr3   Úranger6   ÚplotÚaxisÚsavefig)
r<   r   r€   rS   Ú	testcurveÚ	rulescoreÚxÚ
traincurveÚpltÚrs
             r   rg   rg   z  s  € Ø˜?Ñ+Ð,€IØ˜|Ñ,ò 4ˆ	Ø×Ñ˜ 2™¨Ñ2Õ3ð4à:CÀEÀTÐ:JÖK°Q��Q˜ <Ñ0Ñ0Ó0ÐK€IÐKà˜_Ñ-Ð.€JØ Ñ-ò 6ˆ	Ø×Ñ˜* R™.¨9Ñ4Õ5ð6à<FÀuÈÐ<MÖN°q�!�a˜* \Ñ2Ñ2Ó2ÐN€JÐNå#äŒU”3�y“>Ó"Ó#€AØ‡H�HˆQ�	˜1˜jÔ)Ø‡H�HÒ$Ô%Ø‡K�KÐ%Õ&ùò Lùò
 Os   ¯C!Á1C&©z^-?[0-9]+(\.[0-9]+)?$ÚCD©z.*ÚNN)	r¨   )z(The|the|A|a|An|an)$ÚAT)z.*able$ÚJJ)z.*ness$r«   )z.*ly$ÚRB)z.*s$ÚNNS)z.*ing$ÚVBG)z.*ed$ÚVBDrª   c                 ó<   — t        | «      t        d„ | D «       «      fS )Nc              3   ó2   K  — | ]  }t        |«      –— Œ y ­w)N)r6   )Ú.0r¤   s     r   ú	<genexpr>zcorpus_size.<locals>.<genexpr>Ÿ  s   è ø€ Ò0 aœ3˜qŸ6Ñ0ùs   ‚)r6   Úsum)Úseqss    r   rŒ   rŒ   ž  s   € Ü�‹I”sÑ0¨4Ô0Ó0Ð1Ð1r   Ú__main__)NNiè  é,  r0   Ngš™™™™™é?r0   Fr   FFNNNr¹   NFN)NN)%rY   r]   rˆ   r`   Únltk.corpusr   Únltk.picklesecr   Únltk.tagr   r   r   rW   r   r	   Únltk.tblr
   r   r   r   r   r   r%   r'   r+   r9   r=   r@   rC   rI   r   rX   rg   ÚNN_CD_TAGGERrV   rŒ   Ú__name__r   r   r   ú<module>rÀ      sì   ðó 
Û Û Û å  Ý &ß DÑ Dß $ß )òòòò!ò4ò;ò	8òMò(
ò&ò*ò7ð ØØØØØØ
Ø
ØØØØØØØØØ Ø Øó'BQòJ%CóP'ñ& Ð=¸}ÐMÓN€áò
ó€ò2ð ˆzÒÙÕð r   