§
    'ê[f/B  ã                   ó¾   — d dl Z d dlZ	 d dlZn# e$ r Y nw xY wd dlmZ d\  ZZd\  ZZ	d gZ
 G d„ de¦  «        Z G d„ d¦  «        Z G d„ d	¦  «        Zdd„Zdd„ZdS )é    N)Ú
TokenizerI)r   é   c            	       ób   — e Zd ZdZddededdedf	d„Zd	„ Zd
„ Z	d„ Z
d„ Zd„ Zd„ Zd„ Zd„ Zd„ ZdS )ÚTextTilingTokenizeraû  Tokenize a document into topical sections using the TextTiling algorithm.
    This algorithm detects subtopic shifts based on the analysis of lexical
    co-occurrence patterns.

    The process starts by tokenizing the text into pseudosentences of
    a fixed size w. Then, depending on the method used, similarity
    scores are assigned at sentence gaps. The algorithm proceeds by
    detecting the peak differences between these scores and marking
    them as boundaries. The boundaries are normalized to the closest
    paragraph break and the segmented text is returned.

    :param w: Pseudosentence size
    :type w: int
    :param k: Size (in sentences) of the block used in the block comparison method
    :type k: int
    :param similarity_method: The method used for determining similarity scores:
       `BLOCK_COMPARISON` (default) or `VOCABULARY_INTRODUCTION`.
    :type similarity_method: constant
    :param stopwords: A list of stopwords that are filtered out (defaults to NLTK's stopwords corpus)
    :type stopwords: list(str)
    :param smoothing_method: The method used for smoothing the score plot:
      `DEFAULT_SMOOTHING` (default)
    :type smoothing_method: constant
    :param smoothing_width: The width of the window used by the smoothing method
    :type smoothing_width: int
    :param smoothing_rounds: The number of smoothing passes
    :type smoothing_rounds: int
    :param cutoff_policy: The policy used to determine the number of boundaries:
      `HC` (default) or `LC`
    :type cutoff_policy: constant

    >>> from nltk.corpus import brown
    >>> tt = TextTilingTokenizer(demo_mode=True)
    >>> text = brown.raw()[:4000]
    >>> s, ss, d, b = tt.tokenize(text)
    >>> b
    [0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 1, 0, 0, 0, 0]
    é   é
   Né   r   Fc
                 óœ   — |€ddl m} |                     d¦  «        }| j                             t          ¦   «         ¦  «         | j        d= d S )Nr   ©Ú	stopwordsÚenglishÚself)Únltk.corpusr   ÚwordsÚ__dict__ÚupdateÚlocals)
r   ÚwÚkÚsimilarity_methodr   Úsmoothing_methodÚsmoothing_widthÚsmoothing_roundsÚcutoff_policyÚ	demo_modes
             úL/var/www/piapp/venv/lib/python3.11/site-packages/nltk/tokenize/texttiling.pyÚ__init__zTextTilingTokenizer.__init__@   sW   € ð ÐØ-Ð-Ð-Ð-Ð-Ð-à!Ÿš¨	Ñ2Ô2ˆIØŒ×Ò�V™XœXÑ&Ô&Ð&ØŒM˜&Ð!Ð!Ð!ó    c                 óò  ‡ — |                      ¦   «         }‰                      |¦  «        }t          |¦  «        }d                     d„ |D ¦   «         ¦  «        }‰                      |¦  «        }‰                      |¦  «        }|D ]}ˆ fd„|j        D ¦   «         |_        Œ‰                      ||¦  «        }	‰ j        t          k    r‰  	                    ||	¦  «        }
n7‰ j        t          k    rt          d¦  «        ‚t          d‰ j        › d�¦  «        ‚‰ j        t          k    r‰                      |
¦  «        }nt          d‰ j        › d�¦  «        ‚‰                      |¦  «        }‰                      |¦  «        }‰                      |||¦  «        }g }d}|D ](}|dk    rŒ	|                     |||…         ¦  «         |}Œ)||k     r|                     ||d	…         ¦  «         |s|g}‰ j        r|
|||fS |S )
zZReturn a tokenized copy of *text*, where each "token" represents
        a separate topic.Ú c              3   óD   K  — | ]}t          j        d |¦  «        ¯|V — ŒdS )z[a-z\-' \n\t]N)ÚreÚmatch)Ú.0Úcs     r   ú	<genexpr>z/TextTilingTokenizer.tokenize.<locals>.<genexpr>_   sH   è è € ð 
ð 
Ø­¬Ð2BÀAÑ)FÔ)Fð
Øð
ð 
ð 
ð 
ð 
ð 
r   c                 ó2   •— g | ]}|d          ‰j         v¯|‘ŒS ©r   r   )r$   Úwir   s     €r   ú
<listcomp>z0TextTilingTokenizer.tokenize.<locals>.<listcomp>o   s0   ø€ ð  ð  ð  Ø°°A´¸d¼nÐ1LÐ1L�Ð1LÐ1LÐ1Lr   z'Vocabulary introduction not implementedzSimilarity method z not recognizedzSmoothing method r   N)ÚlowerÚ_mark_paragraph_breaksÚlenÚjoinÚ_divide_to_tokensequencesÚwrdindex_listÚ_create_token_tabler   ÚBLOCK_COMPARISONÚ_block_comparisonÚVOCABULARY_INTRODUCTIONÚNotImplementedErrorÚ
ValueErrorr   ÚDEFAULT_SMOOTHINGÚ_smooth_scoresÚ_depth_scoresÚ_identify_boundariesÚ_normalize_boundariesÚappendr   )r   ÚtextÚlowercase_textÚparagraph_breaksÚtext_lengthÚnopunct_textÚnopunct_par_breaksÚtokseqsÚtsÚtoken_tableÚ
gap_scoresÚsmooth_scoresÚdepth_scoresÚsegment_boundariesÚnormalized_boundariesÚsegmented_textÚprevbÚbs   `                 r   ÚtokenizezTextTilingTokenizer.tokenizeT   s|  ø€ ð Ÿš™œˆØ×6Ò6°tÑ<Ô<ÐÝ˜.Ñ)Ô)ˆð
 —w’wð 
ð 
Ø%ð
ñ 
ô 
ñ 
ô 
ˆð "×8Ò8¸ÑFÔFÐà×0Ò0°Ñ>Ô>ˆð ð 	ð 	ˆBð ð  ð  ð  ØÔ-ð ñ  ô  ˆBÔÐð ×.Ò.¨wÐ8JÑKÔKˆð Ô!Õ%5Ò5Ð5Ø×/Ò/°¸ÑEÔEˆJˆJØÔ#Õ'>Ò>Ð>Ý%Ð&OÑPÔPÐPåØL TÔ%;ÐLÐLÐLñô ð ð Ô Õ$5Ò5Ð5Ø ×/Ò/°
Ñ;Ô;ˆMˆMåÐW°Ô1FÐWÐWÐWÑXÔXÐXð ×)Ò)¨-Ñ8Ô8ˆØ!×6Ò6°|ÑDÔDÐà $× :Ò :ØÐ$Ð&6ñ!
ô !
Ðð ˆØˆà&ð 	ð 	ˆAØ�AŠvˆvØØ×!Ò! $ u¨Q w¤-Ñ0Ô0Ð0ØˆEˆEà�;ÒÐØ×!Ò! $ u v v¤,Ñ/Ô/Ð/àð 	$Ø"˜VˆNàŒ>ð 	OØ˜}¨lÐ<NÐNÐNØÐr   c                 óL  ‡— ˆfd„}g }t          |¦  «        dz
  }t          |¦  «        D ]ù}d\  }}}	d}
|| j        dz
  k     r|dz   }n||| j        z
  k    r||z
  }n| j        }d„ |||z
  dz   |dz   …         D ¦   «         }d„ ||dz   ||z   dz   …         D ¦   «         }‰D ]B}| |||¦  «         |||¦  «        z  z  }| |||¦  «        dz  z  }|	 |||¦  «        dz  z  }	ŒC	 |t          j        ||	z  ¦  «        z  }
n# t
          $ r Y nw xY w|                     |
¦  «         Œú|S )z&Implements the block comparison methodc                 óx   •‡— t          ˆfd„‰|          j        ¦  «        }t          d„ |D ¦   «         ¦  «        }|S )Nc                 ó   •— | d         ‰v S ©Nr   © )ÚoÚblocks    €r   ú<lambda>zHTextTilingTokenizer._block_comparison.<locals>.blk_frq.<locals>.<lambda>¥   s   ø€  q¨¤t¨u }€ r   c              3   ó&   K  — | ]}|d          V — ŒdS )r   NrS   )r$   Útsoccs     r   r&   zITextTilingTokenizer._block_comparison.<locals>.blk_frq.<locals>.<genexpr>¦   s&   è è € Ð5Ð5 E�u˜Q”xÐ5Ð5Ð5Ð5Ð5Ð5r   )ÚfilterÚts_occurencesÚsum)ÚtokrU   Úts_occsÚfreqrE   s    `  €r   Úblk_frqz6TextTilingTokenizer._block_comparison.<locals>.blk_frq¤   sF   øø€ ÝÐ4Ð4Ð4Ð4°kÀ#Ô6FÔ6TÑUÔUˆGÝÐ5Ð5¨WÐ5Ñ5Ô5Ñ5Ô5ˆDØˆKr   r   )ç        r`   r`   r`   c                 ó   — g | ]	}|j         ‘Œ
S rS   ©Úindex©r$   rD   s     r   r*   z9TextTilingTokenizer._block_comparison.<locals>.<listcomp>·   ó   € ÐXÐXÐX˜r�"”(ÐXÐXÐXr   c                 ó   — g | ]	}|j         ‘Œ
S rS   rb   rd   s     r   r*   z9TextTilingTokenizer._block_comparison.<locals>.<listcomp>¸   re   r   r	   )r-   Úranger   ÚmathÚsqrtÚZeroDivisionErrorr<   )r   rC   rE   r_   rF   ÚnumgapsÚcurr_gapÚscore_dividendÚscore_divisor_b1Úscore_divisor_b2ÚscoreÚwindow_sizeÚb1Úb2Úts     `            r   r3   z%TextTilingTokenizer._block_comparison¡   s¾  ø€ ð	ð 	ð 	ð 	ð 	ð
 ˆ
Ý�g‘,”, Ñ"ˆå˜g™œð 	%ð 	%ˆHØANÑ>ˆNÐ,Ð.>ØˆEà˜$œ& 1™*Ò$Ð$Ø&¨™l��Ø˜G d¤fÑ,Ò,Ð,Ø%¨Ñ0��à"œf�àXÐX W¨X¸Ñ-CÀaÑ-GÈ(ÐUVÉ,Ð-VÔ%WÐXÑXÔXˆBØXÐX W¨X¸©\¸HÀ{Ñ<RÐUVÑ<VÐ-VÔ%WÐXÑXÔXˆBà ð 8ð 8�Ø ' '¨!¨R¡.¤.°7°7¸1¸b±>´>Ñ"AÑA�Ø  G G¨A¨r¡N¤N°aÑ$7Ñ7Ð Ø  G G¨A¨r¡N¤N°aÑ$7Ñ7Ð Ð ðØ&­¬Ð3CÐFVÑ3VÑ)WÔ)WÑW��øÝ$ð ð ð Ø�ðøøøð ×Ò˜eÑ$Ô$Ð$Ð$àÐs   Ã#C>Ã>
DÄ
Dc           	      ó‚   — t          t          t          j        |dd…         ¦  «        | j        dz   ¬¦  «        ¦  «        S )z1Wraps the smooth function from the SciPy CookbookNr   )Ú
window_len)ÚlistÚsmoothÚnumpyÚarrayr   )r   rF   s     r   r8   z"TextTilingTokenizer._smooth_scoresÇ   s?   € åÝ•5”;˜z¨!¨!¨!œ}Ñ-Ô-¸$Ô:NÐQRÑ:RÐSÑSÔSñ
ô 
ð 	
r   c                 ó  — d}t          j        d¦  «        }|                     |¦  «        }d}dg}|D ]Y}|                     ¦   «         |z
  |k     rŒ|                     |                     ¦   «         ¦  «         |                     ¦   «         }ŒZ|S )zNIdentifies indented text or line breaks as the beginning of
        paragraphséd   z[ 	]*
[ 	]*
[ 	]*r   )r"   ÚcompileÚfinditerÚstartr<   )r   r=   ÚMIN_PARAGRAPHÚpatternÚmatchesÚ
last_breakÚpbreaksÚpbs           r   r,   z*TextTilingTokenizer._mark_paragraph_breaksÍ   s�   € ð ˆÝ”*ÐGÑHÔHˆØ×"Ò" 4Ñ(Ô(ˆàˆ
Ø�#ˆØð 	(ð 	(ˆBØ�xŠx‰zŒz˜JÑ&¨Ò6Ð6Øà—’˜rŸxšx™zœzÑ*Ô*Ð*ØŸXšX™ZœZ�
�
àˆr   c                 ó  ‡‡— | j         Šg Št          j        d|¦  «        }|D ]=}‰                     |                     ¦   «         |                     ¦   «         f¦  «         Œ>ˆˆfd„t          dt          ‰¦  «        ‰¦  «        D ¦   «         S )z3Divides the text into pseudosentences of fixed sizez\w+c           	      óL   •— g | ] }t          |‰z  ‰||‰z   …         ¦  «        ‘Œ!S rS   )ÚTokenSequence)r$   Úir   r0   s     €€r   r*   zATextTilingTokenizer._divide_to_tokensequences.<locals>.<listcomp>æ   sD   ø€ ð 
ð 
ð 
àõ ˜!˜a™% ¨q°1°q±5¨yÔ!9Ñ:Ô:ð
ð 
ð 
r   r   )r   r"   r~   r<   Úgroupr   rg   r-   )r   r=   r‚   r#   r   r0   s       @@r   r/   z-TextTilingTokenizer._divide_to_tokensequencesß   s¢   øø€ àŒFˆØˆÝ”+˜f dÑ+Ô+ˆØð 	Að 	AˆEØ× Ò  %§+¢+¡-¤-°·²±´Ð!?Ñ@Ô@Ð@Ð@ð
ð 
ð 
ð 
ð 
å˜1�c -Ñ0Ô0°!Ñ4Ô4ð
ñ 
ô 
ð 	
r   c           
      óü  — i }d}d}|                      ¦   «         }t          |¦  «        }|dk    r3	 t          |¦  «        }n"# t          $ r}t          d¦  «        |‚d}~ww xY w|D �]}	|	j        D �]\  }
}	 ||k    rt          |¦  «        }|dz  }||k    °n# t          $ r Y nw xY w|
|v r­||
         xj        dz  c_        ||
         j        |k    r#|||
         _        ||
         xj        dz  c_        ||
         j        |k    r0|||
         _        ||
         j	         
                    |dg¦  «         ŒÇ||
         j	        d         dxx         dz  cc<   Œét          ||dggdd||¬¦  «        ||
<   �Œ|dz  }�Œ|S )z#Creates a table of TokenTableFieldsr   z7No paragraph breaks were found(text too short perhaps?)Nr   éÿÿÿÿ)Ú	first_posrZ   Útotal_countÚ	par_countÚlast_parÚlast_tok_seq)Ú__iter__ÚnextÚStopIterationr6   r0   rŽ   r�   r�   r‘   rZ   r<   ÚTokenTableField)r   Útoken_sequencesÚ
par_breaksrE   Úcurrent_parÚcurrent_tok_seqÚpb_iterÚcurrent_par_breakÚerD   Úwordrc   s               r   r1   z'TextTilingTokenizer._create_token_tableë   s  € àˆØˆØˆØ×%Ò%Ñ'Ô'ˆÝ  ™MœMÐØ Ò!Ð!ðÝ$(¨¡M¤MÐ!Ð!øÝ ð ð ð Ý ØMñô àðøøøøðøøøð "ð  	!ñ  	!ˆBØ!Ô/ð ñ ‘��eðØÐ"3Ò3Ð3Ý,0°©M¬MÐ)Ø# qÑ(˜ð  Ð"3Ò3Ð3øøõ %ð ð ð à�Dðøøøð ˜;Ð&Ð&Ø Ô%Ð1Ô1°QÑ6Ð1Ô1à" 4Ô(Ô1°[Ò@Ð@Ø5@˜ DÔ)Ô2Ø# DÔ)Ð3Ô3°qÑ8Ð3Ô3à" 4Ô(Ô5¸ÒHÐHØ9H˜ DÔ)Ô6Ø# DÔ)Ô7×>Ò>ÀÐQRÐ?SÑTÔTÐTÐTà# DÔ)Ô7¸Ô;¸AÐ>Ð>Ô>À!ÑCÐ>Ð>Ñ>Ð>å(7Ø"'Ø(7¸Ð';Ð&<Ø$%Ø"#Ø!,Ø%4ð)ñ )ô )�K Ñ%Ñ%ð ˜qÑ ˆO‰OàÐs)   ±A Á
A ÁAÁA Á6 BÂ
B$Â#B$c           
      ód  ‡	— d„ |D ¦   «         }t          |¦  «        t          |¦  «        z  }t          j        |¦  «        }| j        t
          k    r||z
  Š	n||dz  z
  Š	t          t          |t          t          |¦  «        ¦  «        ¦  «        ¦  «        }| 	                    ¦   «          t          t          ˆ	fd„|¦  «        ¦  «        }|D ]c}d||d         <   |D ]S}|d         |d         k    r?t          |d         |d         z
  ¦  «        dk     r||d                  dk    rd||d         <   ŒTŒd|S )zJIdentifies boundaries at the peaks of similarity score
        differencesc                 ó   — g | ]}d ‘ŒS r(   rS   ©r$   Úxs     r   r*   z<TextTilingTokenizer._identify_boundaries.<locals>.<listcomp>!  s   € Ð.Ð.Ð.˜A�aÐ.Ð.Ð.r   g       @c                 ó   •— | d         ‰k    S rR   rS   )r¡   Úcutoffs    €r   rV   z:TextTilingTokenizer._identify_boundaries.<locals>.<lambda>-  s   ø€  1 Q¤4¨&¢=€ r   r   é   r   )r[   r-   ry   Ústdr   ÚLCÚsortedÚziprg   Úreverserw   rY   Úabs)
r   rH   Ú
boundariesÚavgÚstdevÚdepth_tuplesÚhpÚdtÚdt2r£   s
            @r   r:   z(TextTilingTokenizer._identify_boundaries  sN  ø€ ð /Ð. Ð.Ñ.Ô.ˆ
å�,ÑÔ¥# lÑ"3Ô"3Ñ3ˆÝ”	˜,Ñ'Ô'ˆàÔ¥Ò#Ð#Ø˜5‘[ˆFˆFà˜5 3™;Ñ&ˆFå�c ,µµc¸,Ñ6GÔ6GÑ0HÔ0HÑIÔIÑJÔJˆØ×ÒÑÔÐÝ•&Ð0Ð0Ð0Ð0°,Ñ?Ô?Ñ@Ô@ˆàð 	*ð 	*ˆBØ !ˆJ�r˜!”uÑØð *ð *�à�q”E˜S œV’O�OÝ˜C œF R¨¤U™NÑ+Ô+¨aÒ/Ð/Ø" 3 q¤6Ô*¨aÒ/Ð/à()�J˜r !œuÑ%øð*ð Ðr   c                 ó"  — d„ |D ¦   «         }t          t          t          |¦  «        dz  d¦  «        d¦  «        }|}||| …         D ]F}|}||dd…         D ]}||k    r|}Œ |}||d…         D ]}||k    r|}Œ ||z   d|z  z
  ||<   |dz  }ŒG|S )zzCalculates the depth of each gap, i.e. the average difference
        between the left and right peaks and the gap's scorec                 ó   — g | ]}d ‘ŒS r(   rS   r    s     r   r*   z5TextTilingTokenizer._depth_scores.<locals>.<listcomp>>  s   € Ð*Ð*Ð*˜a˜Ð*Ð*Ð*r   r   r	   é   NrŒ   r   )ÚminÚmaxr-   )	r   ÚscoresrH   Úcliprc   ÚgapscoreÚlpeakrp   Úrpeaks	            r   r9   z!TextTilingTokenizer._depth_scores:  så   € ð +Ð* 6Ð*Ñ*Ô*ˆõ
 •3•s˜6‘{”{ bÑ(¨!Ñ,Ô,¨aÑ0Ô0ˆØˆà˜t T E˜zÔ*ð 	ð 	ˆHØˆEØ  	 r 	Ô*ð ð �Ø˜E’>�>Ø!�E�EàØˆEØ   œð ð �Ø˜E’>�>Ø!�E�EàØ"'¨%¡-°!°h±,Ñ">ˆL˜ÑØ�Q‰JˆEˆEàÐr   c                 ó’  — g }d\  }}}d}|D ]¹}	|dz  }|	dv r	|rd}|dz  }|	dvr|sd}|t          |¦  «        k     rŠ|t          || j        z  | j        ¦  «        k    ri||         dk    rXt          |¦  «        }
|D ]-}|
t          ||z
  ¦  «        k    rt          ||z
  ¦  «        }
|}Œ- ||vr|                     |¦  «         |dz  }Œº|S )zSNormalize the boundaries identified to the original text's
        paragraph breaks)r   r   r   Fr   z 	
T)r-   r¶   r   rª   r<   )r   r=   r«   r?   Únorm_boundariesÚ
char_countÚ
word_countÚ	gaps_seenÚ	seen_wordÚcharÚbest_fitÚbrÚbestbrs                r   r;   z)TextTilingTokenizer._normalize_boundariesX  s"  € ð ˆØ,3Ñ)ˆ
�J 	Øˆ	àð 	ð 	ˆDØ˜!‰OˆJØ�wˆˆ 9ˆØ!�	Ø˜a‘�
Ø˜7Ð"Ð"¨9Ð"Ø �	Ø�3˜z™?œ?Ò*Ð*¨zÝ�I ¤Ñ&¨¬Ñ/Ô/ò0ð 0ð ˜iÔ(¨AÒ-Ð-å" 4™yœy�HØ.ð "ð "˜Ø#¥c¨"¨z©/Ñ&:Ô&:Ò:Ð:Ý'*¨2°
©?Ñ';Ô';˜HØ%'˜F˜Fà!Ø _Ð4Ð4Ø'×.Ò.¨vÑ6Ô6Ð6Ø˜Q‘�	øàÐr   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r2   r7   ÚHCr   rN   r3   r8   r,   r/   r1   r:   r9   r;   rS   r   r   r   r      sÝ   € € € € € ð%ð %ðR Ø
Ø*ØØ*ØØØØð"ð "ð "ð "ð(Kð Kð KðZ$ð $ð $ðL
ð 
ð 
ðð ð ð$

ð 

ð 

ð0ð 0ð 0ðdð ð ð:ð ð ð<ð ð ð ð r   r   c                   ó"   — e Zd ZdZ	 	 	 	 dd„ZdS )r•   z[A field in the token table holding parameters for each token,
    used later in the processr   r   Nc                 ób   — | j                              t          ¦   «         ¦  «         | j         d= d S ©Nr   )r   r   r   )r   r�   rZ   rŽ   r�   r�   r‘   s          r   r   zTokenTableField.__init__~  s.   € ð 	Œ×Ò�V™XœXÑ&Ô&Ð&ØŒM˜&Ð!Ð!Ð!r   )r   r   r   N©rÆ   rÇ   rÈ   rÉ   r   rS   r   r   r•   r•   z  s@   € € € € € ð!ð !ð ØØØð
"ð 
"ð 
"ð 
"ð 
"ð 
"r   r•   c                   ó   — e Zd ZdZdd„ZdS )rˆ   z3A token list with its original length and its indexNc                 ó„   — |pt          |¦  «        }| j                             t          ¦   «         ¦  «         | j        d= d S rÍ   )r-   r   r   r   )r   rc   r0   Úoriginal_lengths       r   r   zTokenSequence.__init__Ž  s>   € Ø)Ð?­S°Ñ-?Ô-?ˆØŒ×Ò�V™XœXÑ&Ô&Ð&ØŒM˜&Ð!Ð!Ð!r   ©NrÎ   rS   r   r   rˆ   rˆ   ‹  s.   € € € € € Ø9Ð9ð"ð "ð "ð "ð "ð "r   rˆ   é   Úflatc                 óü  — | j         dk    rt          d¦  «        ‚| j        |k     rt          d¦  «        ‚|dk     r| S |dvrt          d¦  «        ‚t          j        d| d         z  | |dd	…         z
  | d| d	         z  | d	| d	…         z
  f         }|d
k    rt          j        |d¦  «        }nt          d|z   dz   ¦  «        }t          j        ||                     ¦   «         z  |d¬¦  «        }||dz
  | dz   …         S )aÈ  smooth the data using a window with requested size.

    This method is based on the convolution of a scaled window with the signal.
    The signal is prepared by introducing reflected copies of the signal
    (with the window size) in both ends so that transient parts are minimized
    in the beginning and end part of the output signal.

    :param x: the input signal
    :param window_len: the dimension of the smoothing window; should be an odd integer
    :param window: the type of window from 'flat', 'hanning', 'hamming', 'bartlett', 'blackman'
        flat window will produce a moving average smoothing.

    :return: the smoothed signal

    example::

        t=linspace(-2,2,0.1)
        x=sin(t)+randn(len(t))*0.1
        y=smooth(x)

    :see also: numpy.hanning, numpy.hamming, numpy.bartlett, numpy.blackman, numpy.convolve,
        scipy.signal.lfilter

    TODO: the window parameter could be the window itself if an array instead of a string
    r   z'smooth only accepts 1 dimension arrays.z1Input vector needs to be bigger than window size.é   )rÔ   ÚhanningÚhammingÚbartlettÚblackmanzDWindow is on of 'flat', 'hanning', 'hamming', 'bartlett', 'blackman'r	   r   rŒ   rÔ   Údznumpy.z(window_len)Úsame)Úmode)	Úndimr6   Úsizery   Úr_ÚonesÚevalÚconvolver[   )r¡   rv   ÚwindowÚsr   Úys         r   rx   rx   •  s%  € ð6 	„v�‚{€{ÝÐBÑCÔCÐCà„v�
ÒÐÝÐLÑMÔMÐMà�A‚~€~ØˆàÐKÐKÐKÝØRñ
ô 
ð 	
õ 	Œ��Q�q”T‘˜A˜j¨¨2˜oÔ.Ñ.°°1°q¸´u±9¸qÀÀZÀKÐPRÐARÔ?SÑ3SÐSÔT€Að �ÒÐÝŒJ�z 3Ñ'Ô'ˆˆå�˜FÑ" ^Ñ3Ñ4Ô4ˆåŒ�q˜1Ÿ5š5™7œ7‘{ A¨FÐ3Ñ3Ô3€AàˆZ˜!‰^˜z˜k¨A™oÐ-Ô.Ð.r   c                 óÞ  — ddl m} ddlm} t	          d¬¦  «        }| €|                     ¦   «         d d…         } |                     | ¦  «        \  }}}}|                     d¦  «         |                     d¦  «         | 	                    t          t          |¦  «        ¦  «        |d¬	¦  «         | 	                    t          t          |¦  «        ¦  «        |d
¬	¦  «         | 	                    t          t          |¦  «        ¦  «        |d¬	¦  «         |                     t          t          |¦  «        ¦  «        |¦  «         |                     ¦   «          |                     ¦   «          d S )Nr   )Úpylab)ÚbrownT)r   i'  zSentence Gap indexz
Gap Scores)ÚlabelzSmoothed Gap scoreszDepth scores)Ú
matplotlibrè   r   ré   r   ÚrawrN   ÚxlabelÚylabelÚplotrg   r-   ÚstemÚlegendÚshow)r=   rè   ré   Úttrå   ÚssrÛ   rM   s           r   Údemorõ   Ë  s:  € Ø Ð Ð Ð Ð Ð à!Ð!Ð!Ð!Ð!Ð!å	 tÐ	,Ñ	,Ô	,€BØ€|Ø�yŠy‰{Œ{˜6˜E˜6Ô"ˆØ—+’+˜dÑ#Ô#�K€A€rˆ1ˆaØ	‡L‚LÐ%Ñ&Ô&Ð&Ø	‡L‚L�ÑÔÐØ	‡J‚J�u•S˜‘V”V‰}Œ}˜a |€JÑ4Ô4Ð4Ø	‡J‚J�u•S˜‘W”W‰~Œ~˜rÐ)>€JÑ?Ô?Ð?Ø	‡J‚J�u•S˜‘V”V‰}Œ}˜a ~€JÑ6Ô6Ð6Ø	‡J‚J�u•S˜‘V”V‰}Œ}˜aÑ Ô Ð Ø	‡L‚L�N„N€NØ	‡J‚J�L„L€L€L€Lr   )rÓ   rÔ   rÒ   )rh   r"   ry   ÚImportErrorÚnltk.tokenize.apir   r2   r4   r¦   rÊ   r7   r   r•   rˆ   rx   rõ   rS   r   r   ú<module>rø      s0  ðð €€€Ø 	€	€	€	ð	Ø€L€L€L€LøØð 	ð 	ð 	Ø€Dð	øøøð )Ð (Ð (Ð (Ð (Ð (à,0Ñ )Ð Ð)Ø	�€€BØ�CÐ ð_ð _ð _ð _ð _˜*ñ _ô _ð _ðD"ð "ð "ð "ð "ñ "ô "ð "ð""ð "ð "ð "ð "ñ "ô "ð "ð3/ð 3/ð 3/ð 3/ðlð ð ð ð ð s   Š �–