Ë
    çÍ:jsM  ã                   óª   — d dl Z d dlZ	 d dlZd dlmZ dZdZd\  ZZ	d gZ
 G d„ de«      Z G d„ d	«      Z G d
„ d«      Zdd„Zdefd„Zy# e$ r Y ŒCw xY w)é    N)Ú
TokenizerIÚblock_comparisonÚvocabulary_introduction)r   é   c            	       óf   — e Zd ZdZddededdedf	d„Zd	„ Zd
„ Z	d„ Z
d„ Zd„ Zd„ Zd„ Zd„ Zd„ Zd„ Zy)ÚTextTilingTokenizeraû  Tokenize a document into topical sections using the TextTiling algorithm.
    This algorithm detects subtopic shifts based on the analysis of lexical
    co-occurrence patterns.

    The process starts by tokenizing the text into pseudosentences of
    a fixed size w. Then, depending on the method used, similarity
    scores are assigned at sentence gaps. The algorithm proceeds by
    detecting the peak differences between these scores and marking
    them as boundaries. The boundaries are normalized to the closest
    paragraph break and the segmented text is returned.

    :param w: Pseudosentence size
    :type w: int
    :param k: Size (in sentences) of the block used in the block comparison method
    :type k: int
    :param similarity_method: The method used for determining similarity scores:
       `BLOCK_COMPARISON` (default) or `VOCABULARY_INTRODUCTION`.
    :type similarity_method: constant
    :param stopwords: A list of stopwords that are filtered out (defaults to NLTK's stopwords corpus)
    :type stopwords: list(str)
    :param smoothing_method: The method used for smoothing the score plot:
      `DEFAULT_SMOOTHING` (default)
    :type smoothing_method: constant
    :param smoothing_width: The width of the window used by the smoothing method
    :type smoothing_width: int
    :param smoothing_rounds: The number of smoothing passes
    :type smoothing_rounds: int
    :param cutoff_policy: The policy used to determine the number of boundaries:
      `HC` (default) or `LC`
    :type cutoff_policy: constant

    >>> from nltk.corpus import brown
    >>> tt = TextTilingTokenizer(demo_mode=True)
    >>> text = brown.raw()[:4000]
    >>> s, ss, d, b = tt.tokenize(text)
    >>> b
    [0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 1, 0, 0, 0, 0]
    é   é
   Né   r   Fc
                 ó–   — |€ddl m} |j                  d«      }| j                  j	                  t        «       «       | j                  d= y )Nr   )Ú	stopwordsÚenglishÚself)Únltk.corpusr   ÚwordsÚ__dict__ÚupdateÚlocals)
r   ÚwÚkÚsimilarity_methodr   Úsmoothing_methodÚsmoothing_widthÚsmoothing_roundsÚcutoff_policyÚ	demo_modes
             úm/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/tokenize/texttiling.pyÚ__init__zTextTilingTokenizer.__init__A   s;   € ð ÐÝ-à!Ÿ™¨	Ó2ˆIØ�‰×ÑœV›XÔ&Ø�M‰M˜&Ñ!ó    c                 óÌ  — |j                  «       }| j                  |«      }t        |«      }dj                  d„ |D «       «      }| j                  |«      }| j	                  |«      }|D ]3  }|j
                  D �	cg c]  }	|	d   | j                  vsŒ|	‘Œ c}	|_        Œ5 | j                  ||«      }
| j                  t        k(  r| j                  ||
«      }n>| j                  t        k(  r| j                  |«      }nt        d| j                  › d�«      ‚| j                  t        k(  r| j!                  |«      }nt        d| j                  › d�«      ‚| j#                  |«      }| j%                  |«      }| j'                  |||«      }g }d}|D ]  }|dk(  rŒ	|j)                  ||| «       |}Œ  ||k  r|j)                  ||d «       |s|g}| j*                  r||||fS |S c c}	w )zZReturn a tokenized copy of *text*, where each "token" represents
        a separate topic.Ú c              3   óN   K  — | ]  }t        j                  d |«      sŒ|–— Œ y­w)z[a-z\-' \n\t]N)ÚreÚmatch)Ú.0Úcs     r   ú	<genexpr>z/TextTilingTokenizer.tokenize.<locals>.<genexpr>_   s#   è ø€ ò 
Ø¬¯©Ð2BÀAÕ)FŒAñ
ùs   ‚%ž%r   zSimilarity method z not recognizedzSmoothing method N)ÚlowerÚ_mark_paragraph_breaksÚlenÚjoinÚ_divide_to_tokensequencesÚwrdindex_listr   Ú_create_token_tabler   ÚBLOCK_COMPARISONÚ_block_comparisonÚVOCABULARY_INTRODUCTIONÚ_vocabulary_introductionÚ
ValueErrorr   ÚDEFAULT_SMOOTHINGÚ_smooth_scoresÚ_depth_scoresÚ_identify_boundariesÚ_normalize_boundariesÚappendr   )r   ÚtextÚlowercase_textÚparagraph_breaksÚtext_lengthÚnopunct_textÚnopunct_par_breaksÚtokseqsÚtsÚwiÚtoken_tableÚ
gap_scoresÚsmooth_scoresÚdepth_scoresÚsegment_boundariesÚnormalized_boundariesÚsegmented_textÚprevbÚbs                      r   ÚtokenizezTextTilingTokenizer.tokenizeT   s"  € ð Ÿ™›ˆØ×6Ñ6°tÓ<ÐÜ˜.Ó)ˆð
 —w‘wñ 
Ø%ô
ó 
ˆð "×8Ñ8¸ÓFÐà×0Ñ0°Ó>ˆð ò 	ˆBà×-Ñ-ö Ø°°A±¸d¿n¹nÒ1L’ò ˆBÕð	ð
 ×.Ñ.¨wÐ8JÓKˆð ×!Ñ!Ô%5Ò5Ø×/Ñ/°¸ÓE‰JØ×#Ñ#Ô'>Ò>Ø×6Ñ6°wÓ?‰JäØ$ T×%;Ñ%;Ð$<¸OÐLóð ð × Ñ Ô$5Ò5Ø ×/Ñ/°
Ó;‰MäÐ0°×1FÑ1FÐ0GÀÐWÓXÐXð ×)Ñ)¨-Ó8ˆØ!×6Ñ6°|ÓDÐà $× :Ñ :ØÐ$Ð&6ó!
Ðð ˆØˆà&ò 	ˆAØ�AŠvØØ×!Ñ! $ u¨Q -Ô0Ø‰Eð		ð �;ÒØ×!Ñ! $ u v ,Ô/áØ"˜VˆNà�>Š>Ø˜}¨lÐ<NÐNÐNØÐùòa s   Á:G!ÂG!c                 óB  — t        |«      }|dk  rg S |D ���cg c]!  }|j                  D ��ch c]  \  }}|’Œ	 c}}‘Œ# }}}}| j                  dz  }g }t        «       }	t	        |«      D ],  }
t        ||
   |	z
  «      }|j                  |«       |	||
   z  }	Œ. dg|z  }t        «       }t	        |dz
  dd«      D ]   }
t        ||
   |z
  «      }|||
<   |||
   z  }Œ" g }t	        |dz
  «      D ]$  }
||
   ||
dz      z   |z  }|j                  |«       Œ& |S c c}}w c c}}}w )aR  Compute gap scores using the Vocabulary Introduction method.

        From Marti A. Hearst (1997) "TextTiling: Segmenting Text into
        Multi-Paragraph Subtopic Passages", Computational Linguistics,
        23(1), pp. 33-64. https://aclanthology.org/J97-1003.pdf

        Section 3.2:

        The idea behind this approach is that the introduction of a new
        topic in a text is signaled by the distribution of new vocabulary
        items. For each pseudosentence gap, we count the number of new
        word types appearing on each side of the gap that have not been
        seen in earlier pseudosentences on that side.

        Schematically (adapted from Fig 3, Hearst 1997)::

            pseudosentences:  [ s1 ] [ s2 ] [ s3 ] [ s4 ] [ s5 ] ...
                                          ^
                                       gap at i=2
                                          |
              left side of gap:  s1, s2   |   right side: s3, s4, s5, ...
              (scan left->right)          |   (scan right->left)
                                          |
              new_L(i) = words in s_i     |   new_R(i) = words in s_{i+1}
                not seen in s1..s_{i-1}   |   not seen in s_{i+2}..s_N

        The score for each gap is::

            score(i) = ( new_L(i) + new_R(i) ) / (2 * w)

        where ``w`` is the pseudosentence size, so ``2 * w`` normalizes
        by the maximum possible number of new word types across both
        sides of the gap.

        :param tokseqs: list of TokenSequence objects
        :return: list of gap scores (length = len(tokseqs) - 1)
        r   r   r   éÿÿÿÿ)r*   r-   r   ÚsetÚranger9   )r   r@   ÚnÚseqÚtokenÚ_Útokseq_setsÚnormÚnew_leftÚ	seen_leftÚiÚ	new_countÚ	new_rightÚ
seen_rightrD   Úscores                   r   r2   z,TextTilingTokenizer._vocabulary_introduction¡   sZ  € ôL �‹LˆØˆqŠ5ØˆIð MT×TÐTÀS¨c×.?Ñ.?×@¡( %¨šÕ@ÐTˆÒTð �v‰v˜‰zˆð ˆÜ“Eˆ	Ü�q“ò 	(ˆAÜ˜K¨™N¨YÑ6Ó7ˆIØ�O‰O˜IÔ&Ø˜ Q™Ñ'‰Ið	(ð �C˜!‘Gˆ	Ü“Uˆ
Ü�q˜1‘u˜b "Ó%ò 	)ˆAÜ˜K¨™N¨ZÑ7Ó8ˆIØ$ˆI�a‰LØ˜+ a™.Ñ(‰Jð	)ð ˆ
Ü�q˜1‘u“ò 	%ˆAØ˜a‘[ 9¨Q°©UÑ#3Ñ3°tÑ;ˆEØ×Ñ˜eÕ$ð	%ð ÐùóA AùÔTs   ™D­D¹DÄDc                 óp  ‡— ˆfd„}g }t        |«      dz
  }t        |«      D ]ø  }d\  }}}	d}
|| j                  dz
  k  r|dz   }n$||| j                  z
  kD  r||z
  }n| j                  }|||z
  dz   |dz    D �cg c]  }|j                  ‘Œ }}||dz   ||z   dz    D �cg c]  }|j                  ‘Œ }}‰D ]6  }| |||«       |||«      z  z  }| |||«      dz  z  }|	 |||«      dz  z  }	Œ8 	 |t	        j
                  ||	z  «      z  }
|j                  |
«       Œú |S c c}w c c}w # t        $ r Y Œ*w xY w)z&Implements the block comparison methodc                 óf   •‡— t        ˆfd„‰|    j                  «      }t        d„ |D «       «      }|S )Nc                 ó   •— | d   ‰v S ©Nr   © )ÚoÚblocks    €r   ú<lambda>zHTextTilingTokenizer._block_comparison.<locals>.blk_frq.<locals>.<lambda>ò   s   ø€  q¨¡t¨u }€ r   c              3   ó&   K  — | ]	  }|d    –— Œ y­w)r   Nrb   )r%   Útsoccs     r   r'   zITextTilingTokenizer._block_comparison.<locals>.blk_frq.<locals>.<genexpr>ó   s   è ø€ Ò5 E�u˜Q•xÑ5ùs   ‚)ÚfilterÚts_occurencesÚsum)Útokrd   Úts_occsÚfreqrC   s    `  €r   Úblk_frqz6TextTilingTokenizer._block_comparison.<locals>.blk_frqñ   s0   ù€ ÜÓ4°kÀ#Ñ6F×6TÑ6TÓUˆGÜÑ5¨WÔ5Ó5ˆDØˆKr   r   )ç        ro   ro   ro   r   )r*   rP   r   ÚindexÚmathÚsqrtÚZeroDivisionErrorr9   )r   r@   rC   rn   rD   ÚnumgapsÚcurr_gapÚscore_dividendÚscore_divisor_b1Úscore_divisor_b2r]   Úwindow_sizerA   Úb1Úb2Úts     `             r   r0   z%TextTilingTokenizer._block_comparisonî   s„  ø€ ô	ð
 ˆ
Ü�g“, Ñ"ˆä˜g›ò 	%ˆHØANÑ>ˆNÐ,Ð.>ØˆEà˜$Ÿ&™& 1™*Ò$Ø&¨™l‘Ø˜G d§f¡fÑ,Ò,Ø%¨Ñ0‘à"Ÿf™f�à%,¨X¸Ñ-CÀaÑ-GÈ(ÐUVÉ,Ð%WÖX˜r�"—(“(ÐXˆBÐXØ%,¨X¸©\¸HÀ{Ñ<RÐUVÑ<VÐ%WÖX˜r�"—(“(ÐXˆBÐXà ò 8�Ø¡'¨!¨R£.±7¸1¸b³>Ñ"AÑA�Ø ¡G¨A¨r£N°aÑ$7Ñ7Ð Ø ¡G¨A¨r£N°aÑ$7Ñ7Ñ ð8ðØ&¬¯©Ð3CÐFVÑ3VÓ)WÑW�ð ×Ñ˜eÕ$ð/	%ð2 Ðùò YùÚXøô %ò Ùðús   Á9DÂD$Ã/D)Ä)	D5Ä4D5c           	      ót   — t        t        t        j                  |dd «      | j                  dz   ¬«      «      S )z1Wraps the smooth function from the SciPy CookbookNr   )Ú
window_len)ÚlistÚsmoothÚnumpyÚarrayr   )r   rD   s     r   r5   z"TextTilingTokenizer._smooth_scores  s2   € äÜ”5—;‘;˜z©!˜}Ó-¸$×:NÑ:NÐQRÑ:RÔSó
ð 	
r   c                 óú   — d}t        j                  d«      }|j                  |«      }d}dg}|D ]H  }|j                  «       |z
  |k  rŒ|j	                  |j                  «       «       |j                  «       }ŒJ |S )zNIdentifies indented text or line breaks as the beginning of
        paragraphséd   z[ 	]*
[ 	]*
[ 	]*r   )r#   ÚcompileÚfinditerÚstartr9   )r   r:   ÚMIN_PARAGRAPHÚpatternÚmatchesÚ
last_breakÚpbreaksÚpbs           r   r)   z*TextTilingTokenizer._mark_paragraph_breaks  s}   € ð ˆÜ—*‘*ÐGÓHˆØ×"Ñ" 4Ó(ˆàˆ
Ø�#ˆØò 	(ˆBØ�x‰x‹z˜JÑ&¨Ò6Øà—‘˜rŸx™x›zÔ*ØŸX™X›Z‘
ð	(ð ˆr   c           
      ó.  — | j                   }g }t        j                  d|«      }|D ]1  }|j                  |j	                  «       |j                  «       f«       Œ3 t        dt        |«      |«      D �cg c]  }t        ||z  ||||z    «      ‘Œ c}S c c}w )z3Divides the text into pseudosentences of fixed sizez\w+r   )	r   r#   r†   r9   Úgroupr‡   rP   r*   ÚTokenSequence)r   r:   r   r-   rŠ   r$   rY   s          r   r,   z-TextTilingTokenizer._divide_to_tokensequences,  s–   € à�F‰FˆØˆÜ—+‘+˜f dÓ+ˆØò 	AˆEØ× Ñ  %§+¡+£-°·±³Ð!?Õ@ð	Aô ˜1œc -Ó0°!Ó4ö
àô ˜!˜a™% ¨q°1°q±5Ð!9Õ:ò
ð 	
ùò 
s   Á3Bc           
      ó¾  — i }d}d}|j                  «       }t        |«      }|dk(  r	 t        |«      }|D ]ù  }	|	j                  D ]ã  \  }
}	 ||kD  rt        |«      }|dz  }||kD  rŒ|
|v r§||
   xj
                  dz  c_        ||
   j                  |k7  r"|||
   _        ||
   xj                  dz  c_        ||
   j                  |k7  r+|||
   _        ||
   j                  j                  |dg«       Œ¯||
   j                  d   dxx   dz  cc<   ŒÍt        ||dggdd||¬«      ||
<   Œå |dz  }Œû |S # t        $ r}t        d«      |‚d}~ww xY w# t        $ r Y Œõw xY w)z#Creates a table of TokenTableFieldsr   z7No paragraph breaks were found(text too short perhaps?)Nr   rN   )Ú	first_posri   Útotal_countÚ	par_countÚlast_parÚlast_tok_seq)Ú__iter__ÚnextÚStopIterationr3   r-   r“   r•   r”   r–   ri   r9   ÚTokenTableField)r   Útoken_sequencesÚ
par_breaksrC   Úcurrent_parÚcurrent_tok_seqÚpb_iterÚcurrent_par_breakÚerA   Úwordrp   s               r   r.   z'TextTilingTokenizer._create_token_table8  sÄ  € àˆØˆØˆØ×%Ñ%Ó'ˆÜ  ›MÐØ Ò!ðÜ$(¨£MÐ!ð
 "ò  	!ˆBØ!×/Ñ/ò ‘��eðØÐ"3Ò3Ü,0°«MÐ)Ø# qÑ(˜ð  Ð"3Ó3ð ˜;Ñ&Ø Ñ%×1Ò1°QÑ6Õ1à" 4Ñ(×1Ñ1°[Ò@Ø5@˜ DÑ)Ô2Ø# DÑ)×3Ò3°qÑ8Õ3à" 4Ñ(×5Ñ5¸ÒHØ9H˜ DÑ)Ô6Ø# DÑ)×7Ñ7×>Ñ>ÀÐQRÐ?SÕTà# DÑ)×7Ñ7¸Ñ;¸AÓ>À!ÑCÔ>ä(7Ø"'Ø(7¸Ð';Ð&<Ø$%Ø"#Ø!,Ø%4ô)�K Ò%ð-ð> ˜qÑ ‰OðA 	!ðD ÐøôM !ò Ü ØMóàðûðûô %ò áðús)   ¨D3 ÁEÄ3	EÄ<EÅEÅ	EÅEc           
      ó  ‡
— |D �cg c]  }d‘Œ }}t        |«      t        |«      z  }t        j                  |«      }| j                  t
        k(  r||z
  Š
n||dz  z
  Š
t        t        |t        t        |«      «      «      «      }|j                  «        t        t        ˆ
fd„|«      «      }|D ]I  }d||d   <   |D ]:  }	|d   |	d   k7  sŒt        |	d   |d   z
  «      dk  sŒ'||	d      dk(  sŒ3d||d   <   Œ< ŒK |S c c}w )zJIdentifies boundaries at the peaks of similarity score
        differencesr   g       @c                 ó   •— | d   ‰kD  S ra   rb   )ÚxÚcutoffs    €r   re   z:TextTilingTokenizer._identify_boundaries.<locals>.<lambda>z  s   ø€  1 Q¡4¨&¡=€ r   r   é   )rj   r*   r�   Ústdr   ÚLCÚsortedÚziprP   Úreverser   rh   Úabs)r   rF   r¥   Ú
boundariesÚavgÚstdevÚdepth_tuplesÚhpÚdtÚdt2r¦   s             @r   r7   z(TextTilingTokenizer._identify_boundariesj  s  ø€ ð ".Ö.˜A’aÐ.ˆ
Ð.ä�,Ó¤# lÓ"3Ñ3ˆÜ—	‘	˜,Ó'ˆà×Ñ¤Ò#Ø˜5‘[‰Fà˜5 3™;Ñ&ˆFäœc ,´´c¸,Ó6GÓ0HÓIÓJˆØ×ÑÔÜ”&Ó0°,Ó?Ó@ˆàò 	*ˆBØ !ˆJ�r˜!‘uÑØò *�à�q‘E˜S ™V“OÜ˜C ™F R¨¡U™NÓ+¨aÓ/Ø" 3 q¡6Ñ*¨aÓ/à()�J˜r !™uÒ%ñ*ð	*ð Ðùò/ /s   †	C=c                 ó  — |D �cg c]  }d‘Œ }}t        t        t        |«      dz  d«      d«      }|}|||  D ]B  }|}||dd…   D ]  }||k\  r|}Œ n |}	||d D ]  }||	k\  r|}	Œ n ||	z   d|z  z
  ||<   |dz  }ŒD |S c c}w )zzCalculates the depth of each gap, i.e. the average difference
        between the left and right peaks and the gap's scorer   r
   r   é   NrN   r   )ÚminÚmaxr*   )
r   Úscoresr¥   rF   Úcliprp   ÚgapscoreÚlpeakr]   Úrpeaks
             r   r6   z!TextTilingTokenizer._depth_scores‡  sÚ   € ð $*Ö*˜ašÐ*ˆÐ*ô
 ”3”s˜6“{ bÑ(¨!Ó,¨aÓ0ˆØˆà˜t T EÐ*ò 	ˆHØˆEØ  	 r 	Ñ*ò �Ø˜E’>Ø!‘Eáð	ð
 ˆEØ  ˜ò �Ø˜E’>Ø!‘Eáð	ð
 #(¨%¡-°!°h±,Ñ">ˆL˜ÑØ�Q‰J‰Eð	ð  Ðùò1 +s   …	Bc                 óv  — g }d\  }}}d}|D ]©  }	|dz  }|	dv r	|rd}|dz  }|	dvr|sd}|t        |«      k  sŒ,|t        || j                  z  | j                  «      kD  sŒS||   dk(  rJt        |«      }
|D ]%  }|
t        ||z
  «      kD  rt        ||z
  «      }
|}Œ% n |vr|j	                  |«       |dz  }Œ« |S )zSNormalize the boundaries identified to the original text's
        paragraph breaks)r   r   r   Fr   z 	
T)r*   r¸   r   r­   r9   )r   r:   r®   r<   Únorm_boundariesÚ
char_countÚ
word_countÚ	gaps_seenÚ	seen_wordÚcharÚbest_fitÚbrÚbestbrs                r   r8   z)TextTilingTokenizer._normalize_boundaries¥  sù   € ð ˆØ,3Ñ)ˆ
�J 	Øˆ	àò 	ˆDØ˜!‰OˆJØ�w‰¡9Ø!�	Ø˜a‘�
Ø˜7Ñ"©9Ø �	Øœ3˜z›?Ó*¨zÜ�I §¡Ñ&¨¯©Ó/ó0ð ˜iÑ(¨AÒ-ä" 4›y�HØ.ò "˜Ø#¤c¨"¨z©/Ó&:Ò:Ü'*¨2°
©?Ó';˜HØ%'™Fá!ð"ð  _Ñ4Ø'×.Ñ.¨vÔ6Ø˜Q‘‘	ð+	ð. Ðr   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r/   r4   ÚHCr   rL   r2   r0   r5   r)   r,   r.   r7   r6   r8   rb   r   r   r   r      sb   „ ñ%ðR Ø
Ø*ØØ*ØØØØó"ò&KòZKòZ$òL
òò$

ò0òdò:ó<r   r   c                   ó    — e Zd ZdZ	 	 	 	 dd„Zy)rš   z[A field in the token table holding parameters for each token,
    used later in the processNc                 ód   — | j                   j                  t        «       «       | j                   d= y ©Nr   )r   r   r   )r   r’   ri   r“   r”   r•   r–   s          r   r   zTokenTableField.__init__Ë  s$   € ð 	�‰×ÑœV›XÔ&Ø�M‰M˜&Ñ!r   )r   r   r   N©rÈ   rÉ   rÊ   rË   r   rb   r   r   rš   rš   Ç  s   „ ñ!ð ØØØô
"r   rš   c                   ó   — e Zd ZdZdd„Zy)r�   z3A token list with its original length and its indexNc                 ó‚   — |xs t        |«      }| j                  j                  t        «       «       | j                  d= y rÏ   )r*   r   r   r   )r   rp   r-   Úoriginal_lengths       r   r   zTokenSequence.__init__Û  s1   € Ø)Ò?¬S°Ó-?ˆØ�‰×ÑœV›XÔ&Ø�M‰M˜&Ñ!r   )NrÐ   rb   r   r   r�   r�   Ø  s
   „ Ù9ô"r   r�   c                 ó   — | j                   dk7  rt        d| j                   › d�«      ‚| j                  |k  rt        dt        | «      › d|› d�«      ‚|dk  r| S |dvrt        d	«      ‚t        j
                  d
| d   z  | |dd…   z
  | d
| d   z  | d| d…   z
  f   }|dk(  rt	        j                  |d«      }nt        d|z   dz   «      }t	        j                  ||j                  «       z  |d¬«      }||dz
  | dz    S )aÈ  smooth the data using a window with requested size.

    This method is based on the convolution of a scaled window with the signal.
    The signal is prepared by introducing reflected copies of the signal
    (with the window size) in both ends so that transient parts are minimized
    in the beginning and end part of the output signal.

    :param x: the input signal
    :param window_len: the dimension of the smoothing window; should be an odd integer
    :param window: the type of window from 'flat', 'hanning', 'hamming', 'bartlett', 'blackman'
        flat window will produce a moving average smoothing.

    :return: the smoothed signal

    example::

        t=linspace(-2,2,0.1)
        x=sin(t)+randn(len(t))*0.1
        y=smooth(x)

    :see also: numpy.hanning, numpy.hamming, numpy.bartlett, numpy.blackman, numpy.convolve,
        scipy.signal.lfilter

    TODO: the window parameter could be the window itself if an array instead of a string
    r   z,smooth only accepts 1 dimension arrays, was ú.zInput vector (z') needs to be bigger than window size (z).é   )ÚflatÚhanningÚhammingÚbartlettÚblackmanzDWindow is on of 'flat', 'hanning', 'hamming', 'bartlett', 'blackman'r   r   rN   r×   Údznumpy.z(window_len)Úsame)Úmode)
Úndimr3   Úsizer*   r�   Úr_ÚonesÚevalÚconvolverj   )r¥   r~   ÚwindowÚsr   Úys         r   r€   r€   â  s,  € ð6 	‡v�v�‚{ÜÐGÈÏÉÀxÈqÐQÓRÐRà‡v�v�
ÒÜØœS ›V˜HÐ$KÈJÈ<ÐWYÐZó
ð 	
ð �A‚~ØˆàÐKÑKÜØRó
ð 	
ô 	�‰��Q�q‘T‘˜A˜j¨¨2˜oÑ.Ñ.°°1°q¸±u±9¸qÀÀZÀKÐPRÐARÑ?SÑ3SÐSÑT€Að �ÒÜ�J‰J�z 3Ó'‰ä�˜FÑ" ^Ñ3Ó4ˆä�‰�q˜1Ÿ5™5›7‘{ A¨FÔ3€AàˆZ˜!‰^˜z˜k¨A™oÐ.Ð.r   c                 ó„  — ddl m} ddlm} |j	                  «        t        d|¬«      }| €|j                  «       d d } |j                  | «      \  }}}}|j                  d|› �«       |j                  d«       |j                  d	«       |j                  t        t        |«      «      |d	¬
«       |j                  t        t        |«      «      |d¬
«       |j                  t        t        |«      «      |d¬
«       |j                  t        t        |«      «      |«       |j                  «        |j!                  «        y )Nr   )Úpylab)ÚbrownT)r   r   i'  zTextTiling: zSentence Gap indexz
Gap Scores)ÚlabelzSmoothed Gap scoreszDepth scores)Ú
matplotlibré   r   rê   Úfigurer   ÚrawrL   ÚtitleÚxlabelÚylabelÚplotrP   r*   ÚstemÚlegendÚshow)	r:   r   ré   rê   Úttræ   ÚssrÜ   rK   s	            r   Údemorø     sð   € Ý å!à	‡L�L„NÜ	 tÐ?PÔ	Q€BØ€|Ø�y‰y‹{˜6˜EÐ"ˆØ—+‘+˜dÓ#�K€A€rˆ1ˆaØ	‡K�K�,Ð0Ð1Ð2Ô3Ø	‡L�LÐ%Ô&Ø	‡L�L�ÔØ	‡J�JŒu”S˜“V‹}˜a |€JÔ4Ø	‡J�JŒu”S˜“W‹~˜rÐ)>€JÔ?Ø	‡J�JŒu”S˜“V‹}˜a ~€JÔ6Ø	‡J�JŒu”S˜“V‹}˜aÔ Ø	‡L�L„NØ	‡J�J…Lr   )é   r×   )rq   r#   r�   ÚImportErrorÚnltk.tokenize.apir   r/   r1   r©   rÌ   r4   r   rš   r�   r€   rø   rb   r   r   ú<module>rü      s‚   ðó Û 	ð	Ûõ )à%Ð Ø3Ð Ø	�€€BØ�CÐ ôk˜*ô k÷\"ñ "÷""ñ "ó5/ðp Ð&6ô øðY ò 	Ùð	ús   ŠA
 Á
AÁA