§
    'ê[fíp  ã                   ó|  — d Z ddlZddlZddlmZmZmZ ddlmZ ddl	m
Z
 ddlmZ ddlmZ ddlmZ dd	lmZmZ dd
lmZ ddlmZ ddlmZ ddlmZmZ  edg d¢¦  «        Z G d„ d¦  «        Z G d„ d¦  «        Z G d„ d¦  «        Z  G d„ d¦  «        Z! G d„ de!¦  «        Z"d„ Z#e$dk    r
 e#¦   «          g d¢Z%dS )a  
This module brings together a variety of NLTK functionality for
text analysis, and provides simple, interactive interfaces.
Functionality includes: concordancing, collocation discovery,
regular expression search over tokenized strings, and
distributional similarity.
é    N)ÚCounterÚdefaultdictÚ
namedtuple)Úreduce)Úlog)ÚBigramCollocationFinder)ÚMLE)Úpadded_everygram_pipeline)ÚBigramAssocMeasuresÚ	f_measure)ÚConditionalFreqDist)ÚFreqDist)Úsent_tokenize)ÚLazyConcatenationÚ	tokenwrapÚConcordanceLine)ÚleftÚqueryÚrightÚoffsetÚ
left_printÚright_printÚlinec                   óT   — e Zd ZdZed„ ¦   «         Zddd„ fd„Zd„ Zd„ Zdd	„Z	dd„Z
dS )ÚContextIndexa  
    A bidirectional index between words and their 'contexts' in a text.
    The context of a word is usually defined to be the words that occur
    in a fixed window around the word; but other definitions may also
    be used by providing a custom context function.
    c                 ó¾   — |dk    r| |dz
                                 ¦   «         nd}|t          | ¦  «        dz
  k    r| |dz                                  ¦   «         nd}||fS )z;One left token and one right token, normalized to lowercaser   é   ú*START*ú*END*)ÚlowerÚlen)ÚtokensÚir   r   s       ú=/var/www/piapp/venv/lib/python3.11/site-packages/nltk/text.pyÚ_default_contextzContextIndex._default_context.   sf   € ð )*¨Qª¨ˆv�a˜!‘eŒ}×"Ò"Ñ$Ô$Ð$°IˆØ)*­c°&©k¬k¸A©oÒ)=Ð)=��q˜1‘u”×#Ò#Ñ%Ô%Ð%À7ˆØ�eˆ}Ðó    Nc                 ó   — | S ©N© ©Úxs    r$   ú<lambda>zContextIndex.<lambda>5   s   € ÈQ€ r&   c                 ó,  ‡ ‡‡— |‰ _         ‰‰ _        |r|‰ _        n‰ j        ‰ _        ‰rˆfd„‰D ¦   «         Št	          ˆ ˆfd„t          ‰¦  «        D ¦   «         ¦  «        ‰ _        t	          ˆ ˆfd„t          ‰¦  «        D ¦   «         ¦  «        ‰ _        d S )Nc                 ó*   •— g | ]} ‰|¦  «        ¯|‘ŒS r)   r)   )Ú.0ÚtÚfilters     €r$   ú
<listcomp>z)ContextIndex.__init__.<locals>.<listcomp>=   s&   ø€ Ð5Ð5Ð5˜A¨6¨6°!©9¬9Ð5�aÐ5Ð5Ð5r&   c              3   ót   •K  — | ]2\  }}‰                      |¦  «        ‰                     ‰|¦  «        fV — Œ3d S r(   )Ú_keyÚ_context_func©r/   r#   ÚwÚselfr"   s      €€r$   ú	<genexpr>z(ContextIndex.__init__.<locals>.<genexpr>>   sW   øè è € ð %
ð %
Ù>B¸aÀˆT�YŠY�q‰\Œ\˜4×-Ò-¨f°aÑ8Ô8Ð9ð%
ð %
ð %
ð %
ð %
ð %
r&   c              3   ót   •K  — | ]2\  }}‰                      ‰|¦  «        ‰                     |¦  «        fV — Œ3d S r(   )r5   r4   r6   s      €€r$   r9   z(ContextIndex.__init__.<locals>.<genexpr>A   sW   øè è € ð %
ð %
Ù>B¸aÀˆT×Ò ¨Ñ*Ô*¨D¯IªI°a©L¬LÐ9ð%
ð %
ð %
ð %
ð %
ð %
r&   )r4   Ú_tokensr5   r%   ÚCFDÚ	enumerateÚ_word_to_contextsÚ_context_to_words)r8   r"   Úcontext_funcr1   Úkeys   `` ` r$   Ú__init__zContextIndex.__init__5   sæ   øøø€ ØˆŒ	ØˆŒØð 	7Ø!-ˆDÔÐà!%Ô!6ˆDÔØð 	6Ø5Ð5Ð5Ð5 Ð5Ñ5Ô5ˆFÝ!$ð %
ð %
ð %
ð %
ð %
ÝFOÐPVÑFWÔFWð%
ñ %
ô %
ñ "
ô "
ˆÔõ "%ð %
ð %
ð %
ð %
ð %
ÝFOÐPVÑFWÔFWð%
ñ %
ô %
ñ "
ô "
ˆÔÐÐr&   c                 ó   — | j         S )zw
        :rtype: list(str)
        :return: The document that this context index was
            created from.
        ©r;   ©r8   s    r$   r"   zContextIndex.tokensE   ó   € ð Œ|Ðr&   c                 óæ   — |                       |¦  «        }t          | j        |         ¦  «        }i }| j                             ¦   «         D ]%\  }}t	          |t          |¦  «        ¦  «        ||<   Œ&|S )z 
        Return a dictionary mapping from words to 'similarity scores,'
        indicating how often these two words occur in the same
        context.
        )r4   Úsetr>   Úitemsr   )r8   ÚwordÚword_contextsÚscoresr7   Ú
w_contextss         r$   Úword_similarity_dictz!ContextIndex.word_similarity_dictM   sq   € ð �yŠy˜‰ŒˆÝ˜DÔ2°4Ô8Ñ9Ô9ˆàˆØ!Ô3×9Ò9Ñ;Ô;ð 	Bð 	B‰MˆAˆzÝ! -µ°Z±´ÑAÔAˆF�1‰IˆIàˆr&   é   c                 óD  — t          t          ¦  «        }| j        |                      |¦  «                 D ]M}| j        |         D ]=}||k    r5||xx         | j        |         |         | j        |         |         z  z  cc<   Œ>ŒNt          ||j        d¬¦  «        d |…         S )NT)rA   Úreverse)r   Úintr>   r4   r?   ÚsortedÚget)r8   rJ   ÚnrL   Úcr7   s         r$   Úsimilar_wordszContextIndex.similar_words\   s²   € Ý�SÑ!Ô!ˆØÔ'¨¯	ª	°$©¬Ô8ð 	ð 	ˆAØÔ+¨AÔ.ð ð �Ø˜’9�9Ø˜1�I�I”IØÔ.¨qÔ1°$Ô7¸$Ô:PÐQRÔ:SÐTUÔ:VÑVñ�I�I‘Iøðõ
 �f &¤*°dÐ;Ñ;Ô;¸B¸Q¸BÔ?Ð?r&   Fc                 ór  ‡ ‡‡‡— ˆ fd„‰D ¦   «         Šˆ fd„‰D ¦   «         Šˆˆfd„t          t          ‰¦  «        ¦  «        D ¦   «         }t          t          j        ‰¦  «        Š|r%|r#t          dd                     ‰¦  «        ¦  «        ‚‰st          ¦   «         S t          ˆˆ fd„‰D ¦   «         ¦  «        }|S )a§  
        Find contexts where the specified words can all appear; and
        return a frequency distribution mapping each context to the
        number of times that context was used.

        :param words: The words used to seed the similarity search
        :type words: str
        :param fail_on_unknown: If true, then raise a value error if
            any of the given words do not occur at all in the index.
        c                 ó:   •— g | ]}‰                      |¦  «        ‘ŒS r)   )r4   ©r/   r7   r8   s     €r$   r2   z0ContextIndex.common_contexts.<locals>.<listcomp>q   s#   ø€ Ð-Ð-Ð- !�—’˜1‘”Ð-Ð-Ð-r&   c                 óD   •— g | ]}t          ‰j        |         ¦  «        ‘ŒS r)   )rH   r>   rZ   s     €r$   r2   z0ContextIndex.common_contexts.<locals>.<listcomp>r   s)   ø€ ÐBÐBÐB°q•C˜Ô.¨qÔ1Ñ2Ô2ÐBÐBÐBr&   c                 ó0   •— g | ]}‰|         °
‰|         ‘ŒS r)   r)   )r/   r#   ÚcontextsÚwordss     €€r$   r2   z0ContextIndex.common_contexts.<locals>.<listcomp>s   s&   ø€ ÐHÐHÐH˜a¸HÀQ¼KÐH��q”ÐHÐHÐHr&   z%The following word(s) were not found:Ú c              3   óD   •K  — | ]}‰j         |         D ]
}|‰v ¯|V — ŒŒd S r(   )r>   )r/   r7   rV   Úcommonr8   s      €€r$   r9   z/ContextIndex.common_contexts.<locals>.<genexpr>{   sI   øè è € ð ð Ø¨$Ô*@ÀÔ*Cðð Ø%&ÀqÈFÀ{À{�À{À{À{À{À{ðð r&   )Úranger!   r   rH   ÚintersectionÚ
ValueErrorÚjoinr   )r8   r^   Úfail_on_unknownÚemptyÚfdra   r]   s   ``   @@r$   Úcommon_contextszContextIndex.common_contextsf   só   øøøø€ ð .Ð-Ð-Ð- uÐ-Ñ-Ô-ˆØBÐBÐBÐB¸EÐBÑBÔBˆØHÐHÐHÐHÐH¥5­¨U©¬Ñ#4Ô#4ÐHÑHÔHˆÝ�Ô(¨(Ñ3Ô3ˆØð 		�_ð 		ÝÐDÀcÇhÂhÈuÁoÄoÑVÔVÐVØð 	å‘:”:Ðåð ð ð ð ð Ø ðñ ô ñ ô ˆBð ˆIr&   ©rO   )F)Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ústaticmethodr%   rB   r"   rN   rW   ri   r)   r&   r$   r   r   &   s¢   € € € € € ðð ð ðð ñ „\ðð -1¸À;À;ð 
ð 
ð 
ð 
ð ð ð ðð ð ð@ð @ð @ð @ðð ð ð ð ð r&   r   c                   ó@   — e Zd ZdZd„ fd„Zd„ Zd„ Zd„ Zdd„Zdd
„Z	dS )ÚConcordanceIndexzs
    An index that can be used to look up the offset locations at which
    a given word occurs in a document.
    c                 ó   — | S r(   r)   r*   s    r$   r,   zConcordanceIndex.<lambda>‡   s   € ¨Q€ r&   c                 óî   — || _         	 || _        	 t          t          ¦  «        | _        	 t          |¦  «        D ]:\  }}|                      |¦  «        }| j        |                              |¦  «         Œ;dS )aé  
        Construct a new concordance index.

        :param tokens: The document (list of tokens) that this
            concordance index was created from.  This list can be used
            to access the context of a given word occurrence.
        :param key: A function that maps each token to a normalized
            version that will be used as a key in the index.  E.g., if
            you use ``key=lambda s:s.lower()``, then the index will be
            case-insensitive.
        N)r;   r4   r   ÚlistÚ_offsetsr=   Úappend)r8   r"   rA   ÚindexrJ   s        r$   rB   zConcordanceIndex.__init__‡   s€   € ð ˆŒð	 ð ˆŒ	ØDå#¥DÑ)Ô)ˆŒØLå$ VÑ,Ô,ð 	.ð 	.‰KˆE�4Ø—9’9˜T‘?”?ˆDØŒM˜$Ô×&Ò& uÑ-Ô-Ð-Ð-ð	.ð 	.r&   c                 ó   — | j         S )z{
        :rtype: list(str)
        :return: The document that this concordance index was
            created from.
        rD   rE   s    r$   r"   zConcordanceIndex.tokens¡   rF   r&   c                 óF   — |                       |¦  «        }| j        |         S )zä
        :rtype: list(int)
        :return: A list of the offset positions at which the given
            word occurs.  If a key function was specified for the
            index, then given word's key will be looked up.
        )r4   ru   ©r8   rJ   s     r$   ÚoffsetszConcordanceIndex.offsets©   s    € ð �yŠy˜‰ŒˆØŒ}˜TÔ"Ð"r&   c                 óX   — dt          | j        ¦  «        t          | j        ¦  «        fz  S )Nz+<ConcordanceIndex for %d tokens (%d types)>)r!   r;   ru   rE   s    r$   Ú__repr__zConcordanceIndex.__repr__³   s/   € Ø<Ý�”ÑÔÝ�”ÑÔð@
ñ 
ð 	
r&   éP   c           
      óˆ  ‡— t          |t          ¦  «        r|}n|g}|t          d                     |¦  «        ¦  «        z
  dz
  dz  }|dz  }g }|                      |d         ¦  «        }t          |dd…         ¦  «        D ]H\  Š}ˆfd„|                      |¦  «        D ¦   «         }t          |                     |¦  «        ¦  «        }ŒI|rö|D ]óŠd                     | j        ‰‰t          |¦  «        z   …         ¦  «        }	| j        t          d‰|z
  ¦  «        ‰…         }
| j        ‰t          |¦  «        z   ‰|z   …         }d                     |
¦  «        | d…         }d                     |¦  «        d|…         }d                     ||	|g¦  «        }t          |
|	|‰|||¦  «        }|                     |¦  «         Œô|S )z‹
        Find all concordance lines given the query word.

        Provided with a list of words, these will be found as a phrase.
        r_   é   é   r   r   Nc                 ó    •— h | ]
}|‰z
  d z
  ’ŒS )r   r)   )r/   r   r#   s     €r$   ú	<setcomp>z4ConcordanceIndex.find_concordance.<locals>.<setcomp>Ë   s!   ø€ ÐLÐLÐL¨v˜F Q™J¨™NÐLÐLÐLr&   )Ú
isinstancert   r!   re   r{   r=   rS   rc   r;   Úmaxr   rv   )r8   rJ   ÚwidthÚphraseÚ
half_widthÚcontextÚconcordance_listr{   Úword_offsetsÚ
query_wordÚleft_contextÚright_contextr   r   Ú
line_printÚconcordance_liner#   s                   @r$   Úfind_concordancez!ConcordanceIndex.find_concordance¹   sé  ø€ õ �d�DÑ!Ô!ð 	ØˆFˆFà�VˆFà�c #§(¢(¨6Ñ"2Ô"2Ñ3Ô3Ñ3°aÑ7¸AÑ=ˆ
Ø˜1‘*ˆð ÐØ—,’,˜v aœyÑ)Ô)ˆÝ  ¨¨¨¤Ñ,Ô,ð 	Að 	A‰GˆAˆtØLÐLÐLÐL¸¿ºÀdÑ9KÔ9KÐLÑLÔLˆLÝ˜\×6Ò6°wÑ?Ô?Ñ@Ô@ˆGˆGØð 	:Øð :ð :�Ø ŸXšX d¤l°1°q½3¸v¹;¼;±Ð3FÔ&GÑHÔH�
à#œ|­C°°1°w±;Ñ,?Ô,?À!Ð,CÔD�Ø $¤¨Qµ°V±´©_¸qÀ7¹{Ð-JÔ K�à ŸXšX lÑ3Ô3°Z°K°L°LÔA�
Ø!Ÿhšh }Ñ5Ô5°k°z°kÔB�à ŸXšX z°:¸{Ð&KÑLÔL�
å#2Ø ØØ!ØØØØñ$ô $Ð ð !×'Ò'Ð(8Ñ9Ô9Ð9Ð9ØÐr&   é   c                 ó<  — |                       ||¬¦  «        }|st          d¦  «         dS t          |t          |¦  «        ¦  «        }t          d|› dt          |¦  «        › d�¦  «         t	          |d|…         ¦  «        D ]\  }}t          |j        ¦  «         ŒdS )a±  
        Print concordance lines given the query word.
        :param word: The target word or phrase (a list of strings)
        :type word: str or list
        :param lines: The number of lines to display (default=25)
        :type lines: int
        :param width: The width of each line, in characters (default=80)
        :type width: int
        :param save: The option to save the concordance.
        :type save: bool
        )r†   z
no matcheszDisplaying z of z	 matches:N)r‘   ÚprintÚminr!   r=   r   )r8   rJ   r†   ÚlinesrŠ   r#   r�   s          r$   Úprint_concordancez"ConcordanceIndex.print_concordanceå   sÂ   € ð  ×0Ò0°¸UÐ0ÑCÔCÐàð 	-Ý�,ÑÔÐÐÐå˜�sÐ#3Ñ4Ô4Ñ5Ô5ˆEÝÐK ÐKÐK­3Ð/?Ñ+@Ô+@ÐKÐKÐKÑLÔLÐLÝ'0Ð1AÀ&À5À&Ô1IÑ'JÔ'Jð -ð -Ñ#�Ð#ÝÐ&Ô+Ñ,Ô,Ð,Ð,ð-ð -r&   N)r~   )r~   r’   )
rk   rl   rm   rn   rB   r"   r{   r}   r‘   r—   r)   r&   r$   rq   rq   �   s�   € € € € € ðð ð
 $/ ;ð .ð .ð .ð .ð4ð ð ð#ð #ð #ð
ð 
ð 
ð* ð * ð * ð * ðX-ð -ð -ð -ð -ð -r&   rq   c                   ó   — e Zd ZdZd„ Zd„ ZdS )ÚTokenSearcheraâ  
    A class that makes it easier to use regular expressions to search
    over tokenized strings.  The tokenized string is converted to a
    string where tokens are marked with angle brackets -- e.g.,
    ``'<the><window><is><still><open>'``.  The regular expression
    passed to the ``findall()`` method is modified to treat angle
    brackets as non-capturing parentheses, in addition to matching the
    token boundaries; and to have ``'.'`` not match the angle brackets.
    c                 óN   — d                      d„ |D ¦   «         ¦  «        | _        d S )NÚ c              3   ó&   K  — | ]}d |z   dz   V — ŒdS )Ú<Ú>Nr)   )r/   r7   s     r$   r9   z)TokenSearcher.__init__.<locals>.<genexpr>  s*   è è € Ð:Ð:¨a˜C !™G c™MÐ:Ð:Ð:Ð:Ð:Ð:r&   )re   Ú_raw)r8   r"   s     r$   rB   zTokenSearcher.__init__  s(   € Ø—G’GÐ:Ð:°6Ð:Ñ:Ô:Ñ:Ô:ˆŒ	ˆ	ˆ	r&   c                 ó~  — t          j        dd|¦  «        }t          j        dd|¦  «        }t          j        dd|¦  «        }t          j        dd|¦  «        }t          j        || j        ¦  «        }|D ];}|                     d¦  «        s$|                     d¦  «        rt          d	¦  «        ‚Œ<d
„ |D ¦   «         }|S )a  
        Find instances of the regular expression in the text.
        The text is a list of tokens, and a regexp pattern to match
        a single token must be surrounded by angle brackets.  E.g.

        >>> from nltk.text import TokenSearcher
        >>> from nltk.book import text1, text5, text9
        >>> text5.findall("<.*><.*><bro>")
        you rule bro; telling you bro; u twizted bro
        >>> text1.findall("<a>(<.*>)<man>")
        monied; nervous; dangerous; white; white; white; pious; queer; good;
        mature; white; Cape; great; wise; wise; butterless; white; fiendish;
        pale; furious; better; certain; complete; dismasted; younger; brave;
        brave; brave; brave
        >>> text9.findall("<th.*>{3,}")
        thread through those; the thought that; that the thing; the thing
        that; that that thing; through these than through; them that the;
        through the thick; them that they; thought that the

        :param regexp: A regular expression
        :type regexp: str
        z\sr›   r�   z(?:<(?:rž   z)>)z	(?<!\\)\.z[^>]z$Bad regexp for TokenSearcher.findallc                 óH   — g | ]}|d d…                               d¦  «        ‘Œ S )r   éÿÿÿÿz><©Úsplit©r/   Úhs     r$   r2   z)TokenSearcher.findall.<locals>.<listcomp>0  s,   € Ð2Ð2Ð2¨��!�B�$”—’˜dÑ#Ô#Ð2Ð2Ð2r&   )ÚreÚsubÚfindallrŸ   Ú
startswithÚendswithrd   )r8   ÚregexpÚhitsr¦   s       r$   r©   zTokenSearcher.findall
  sÉ   € õ0 ”˜˜r 6Ñ*Ô*ˆÝ”˜˜i¨Ñ0Ô0ˆÝ”˜˜e VÑ,Ô,ˆÝ”˜ f¨fÑ5Ô5ˆõ Œz˜& $¤)Ñ,Ô,ˆð ð 	Ið 	IˆAØ—<’< Ñ$Ô$ð I¨¯ª°C©¬ð IÝ Ð!GÑHÔHÐHøð 3Ð2¨TÐ2Ñ2Ô2ˆØˆr&   N)rk   rl   rm   rn   rB   r©   r)   r&   r$   r™   r™   ü   s<   € € € € € ðð ð;ð ;ð ;ð'ð 'ð 'ð 'ð 'r&   r™   c                   óÆ   — e Zd ZdZdZd!d„Zd„ Zd„ Zd"d	„Zd"d
„Z	d#d„Z
d#d„Zd„ Zd„ Zd„ Zd$d„Zd$d„Zd„ Zd%d„Zd&d„Zd„ Zd„ Zd„ Z ej        d¦  «        Zd„ Zd„ Zd „ ZdS )'ÚTextaÛ  
    A wrapper around a sequence of simple (string) tokens, which is
    intended to support initial exploration of texts (via the
    interactive console).  Its methods perform a variety of analyses
    on the text's contexts (e.g., counting, concordancing, collocation
    discovery), and display the results.  If you wish to write a
    program which makes use of these analyses, then you should bypass
    the ``Text`` class, and use the appropriate analysis function or
    class directly instead.

    A ``Text`` is typically initialized from a given document or
    corpus.  E.g.:

    >>> import nltk.corpus
    >>> from nltk.text import Text
    >>> moby = Text(nltk.corpus.gutenberg.words('melville-moby_dick.txt'))

    TNc                 ób  — | j         rt          |¦  «        }|| _        |r	|| _        dS d|dd…         v rK|dd…                              d¦  «        }d                     d„ |d|…         D ¦   «         ¦  «        | _        dS d                     d„ |dd…         D ¦   «         ¦  «        d	z   | _        dS )
zv
        Create a Text object.

        :param tokens: The source text.
        :type tokens: sequence of str
        ú]NrO   r_   c              3   ó4   K  — | ]}t          |¦  «        V — Œd S r(   ©Ústr©r/   Útoks     r$   r9   z Text.__init__.<locals>.<genexpr>]  s(   è è € Ð CÐ C¨c¥ S¡¤Ð CÐ CÐ CÐ CÐ CÐ Cr&   r   c              3   ó4   K  — | ]}t          |¦  «        V — Œd S r(   r³   rµ   s     r$   r9   z Text.__init__.<locals>.<genexpr>_  s(   è è € Ð @Ð @¨c¥ S¡¤Ð @Ð @Ð @Ð @Ð @Ð @r&   é   z...)Ú_COPY_TOKENSrt   r"   Únamerw   re   )r8   r"   rº   Úends       r$   rB   zText.__init__N  sÄ   € ð Ôð 	"Ý˜&‘\”\ˆFØˆŒàð 	IØˆDŒIˆIˆIØ�F˜3˜B˜3”KÐÐØ˜˜"˜”+×#Ò# CÑ(Ô(ˆCØŸšÐ CÐ C°V¸A¸c¸E´]Ð CÑ CÔ CÑCÔCˆDŒIˆIˆIàŸšÐ @Ð @°V¸B¸Q¸B´ZÐ @Ñ @Ô @Ñ@Ô@À5ÑHˆDŒIˆIˆIr&   c                 ó   — | j         |         S r(   )r"   )r8   r#   s     r$   Ú__getitem__zText.__getitem__e  s   € ØŒ{˜1Œ~Ðr&   c                 ó*   — t          | j        ¦  «        S r(   )r!   r"   rE   s    r$   Ú__len__zText.__len__h  s   € Ý�4”;ÑÔÐr&   éO   r’   c                 ó„   — d| j         vrt          | j        d„ ¬¦  «        | _        | j                             |||¦  «        S )aÌ  
        Prints a concordance for ``word`` with the specified context window.
        Word matching is not case-sensitive.

        :param word: The target word or phrase (a list of strings)
        :type word: str or list
        :param width: The width of each line, in characters (default=80)
        :type width: int
        :param lines: The number of lines to display (default=25)
        :type lines: int

        :seealso: ``ConcordanceIndex``
        Ú_concordance_indexc                 ó*   — |                       ¦   «         S r(   ©r    ©Úss    r$   r,   z"Text.concordance.<locals>.<lambda>  ó   € ¨1¯7ª7©9¬9€ r&   ©rA   )Ú__dict__rq   r"   rÂ   r—   ©r8   rJ   r†   r–   s       r$   ÚconcordancezText.concordanceo  sP   € ð   t¤}Ð4Ð4Ý&6Ø”Ð!4Ð!4ð'ñ 'ô 'ˆDÔ#ð Ô&×8Ò8¸¸uÀeÑLÔLÐLr&   c                 ó’   — d| j         vrt          | j        d„ ¬¦  «        | _        | j                             ||¦  «        d|…         S )aÎ  
        Generate a concordance for ``word`` with the specified context window.
        Word matching is not case-sensitive.

        :param word: The target word or phrase (a list of strings)
        :type word: str or list
        :param width: The width of each line, in characters (default=80)
        :type width: int
        :param lines: The number of lines to display (default=25)
        :type lines: int

        :seealso: ``ConcordanceIndex``
        rÂ   c                 ó*   — |                       ¦   «         S r(   rÄ   rÅ   s    r$   r,   z'Text.concordance_list.<locals>.<lambda>”  rÇ   r&   rÈ   N)rÉ   rq   r"   rÂ   r‘   rÊ   s       r$   rŠ   zText.concordance_list„  sW   € ð   t¤}Ð4Ð4Ý&6Ø”Ð!4Ð!4ð'ñ 'ô 'ˆDÔ#ð Ô&×7Ò7¸¸eÑDÔDÀVÀeÀVÔLÐLr&   rO   r€   c                 ó¦  ‡— d| j         v r| j        |k    r| j        |k    s«|| _        || _        ddlm} |                     d¦  «        Št          j        | j        |¦  «        }| 	                    d¦  «         | 
                    ˆfd„¦  «         t          ¦   «         }t          |                     |j        |¦  «        ¦  «        | _        | j        S )aÚ  
        Return collocations derived from the text, ignoring stopwords.

            >>> from nltk.book import text4
            >>> text4.collocation_list()[:2]
            [('United', 'States'), ('fellow', 'citizens')]

        :param num: The maximum number of collocations to return.
        :type num: int
        :param window_size: The number of tokens spanned by a collocation (default=2)
        :type window_size: int
        :rtype: list(tuple(str, str))
        Ú_collocationsr   )Ú	stopwordsÚenglishr€   c                 óV   •— t          | ¦  «        dk     p|                      ¦   «         ‰v S )Né   )r!   r    )r7   Úignored_wordss    €r$   r,   z'Text.collocation_list.<locals>.<lambda>´  s#   ø€ ­s°1©v¬v¸ªzÐ/W¸Q¿WºW¹Y¼YÈ-Ð=W€ r&   )rÉ   Ú_numÚ_window_sizeÚnltk.corpusrÐ   r^   r   Ú
from_wordsr"   Úapply_freq_filterÚapply_word_filterr   rt   ÚnbestÚlikelihood_ratiorÏ   )r8   ÚnumÚwindow_sizerÐ   ÚfinderÚbigram_measuresrÔ   s         @r$   Úcollocation_listzText.collocation_list˜  sã   ø€ ð ˜tœ}Ð,Ð,Ø”	˜SÒ Ð ØÔ! [Ò0Ð0àˆDŒIØ +ˆDÔð .Ð-Ð-Ð-Ð-Ð-à%ŸOšO¨IÑ6Ô6ˆMÝ,Ô7¸¼À[ÑQÔQˆFØ×$Ò$ QÑ'Ô'Ð'Ø×$Ò$Ð%WÐ%WÐ%WÐ%WÑXÔXÐXÝ1Ñ3Ô3ˆOÝ!%Ø—’˜_Ô=¸sÑCÔCñ"ô "ˆDÔð Ô!Ð!r&   c                 ó‚   — d„ |                       ||¦  «        D ¦   «         }t          t          |d¬¦  «        ¦  «         dS )aý  
        Print collocations derived from the text, ignoring stopwords.

            >>> from nltk.book import text4
            >>> text4.collocations() # doctest: +NORMALIZE_WHITESPACE
            United States; fellow citizens; years ago; four years; Federal
            Government; General Government; American people; Vice President; God
            bless; Chief Justice; one another; fellow Americans; Old World;
            Almighty God; Fellow citizens; Chief Magistrate; every citizen; Indian
            tribes; public debt; foreign nations


        :param num: The maximum number of collocations to print.
        :type num: int
        :param window_size: The number of tokens spanned by a collocation (default=2)
        :type window_size: int
        c                 ó$   — g | ]\  }}|d z   |z   ‘ŒS ©r_   r)   ©r/   Úw1Úw2s      r$   r2   z%Text.collocations.<locals>.<listcomp>Î  s1   € ð 
ð 
ð 
Ù$˜b "ˆB�‰H�r‰Mð
ð 
ð 
r&   ú; )Ú	separatorN)rá   r”   r   )r8   rÝ   rÞ   Úcollocation_stringss       r$   ÚcollocationszText.collocations»  sU   € ð&
ð 
Ø(,×(=Ò(=¸cÀ;Ñ(OÔ(Oð
ñ 
ô 
Ðõ 	�iÐ+°tÐ<Ñ<Ô<Ñ=Ô=Ð=Ð=Ð=r&   c                 ó6   — | j                              |¦  «        S )zJ
        Count the number of times this word appears in the text.
        )r"   Úcountrz   s     r$   rí   z
Text.countÓ  ó   € ð Œ{× Ò  Ñ&Ô&Ð&r&   c                 ó6   — | j                              |¦  «        S )zQ
        Find the index of the first occurrence of the word in the text.
        )r"   rw   rz   s     r$   rw   z
Text.indexÙ  rî   r&   c                 ó   — t           ‚r(   )ÚNotImplementedError)r8   Úmethods     r$   ÚreadabilityzText.readabilityß  s   € å!Ð!r&   c                 óæ  ‡‡‡— d| j         vrt          | j        d„ d„ ¬¦  «        | _        ‰                     ¦   «         Š| j        j        Š‰‰                     ¦   «         v r�t          ‰‰         ¦  «        Št          ˆˆˆfd„‰                     ¦   «         D ¦   «         ¦  «        }d„ | 	                    |¦  «        D ¦   «         }t          t          |¦  «        ¦  «         dS t          d¦  «         dS )	a~  
        Distributional similarity: find other words which appear in the
        same contexts as the specified word; list most similar words first.

        :param word: The word used to seed the similarity search
        :type word: str
        :param num: The number of words to generate (default=20)
        :type num: int
        :seealso: ContextIndex.similar_words()
        Ú_word_context_indexc                 ó*   — |                       ¦   «         S r(   )Úisalphar*   s    r$   r,   zText.similar.<locals>.<lambda>ñ  s   € ¨a¯iªi©k¬k€ r&   c                 ó*   — |                       ¦   «         S r(   rÄ   rÅ   s    r$   r,   zText.similar.<locals>.<lambda>ñ  s   € ÈÏÊÉÌ€ r&   )r1   rA   c              3   óF   •K  — | ]}‰|         D ]}|‰v ¯|‰k    °|V — ŒŒd S r(   r)   )r/   r7   rV   r]   ÚwcirJ   s      €€€r$   r9   zText.similar.<locals>.<genexpr>ú  sW   øè è € ð ð àØ˜Qœðð ð Ø˜�=�=¨¨dª¨ð ð *3¨¨¨¨ð	ð r&   c                 ó   — g | ]\  }}|‘ŒS r)   r)   ©r/   r7   Ú_s      r$   r2   z Text.similar.<locals>.<listcomp>   s   € Ð7Ð7Ð7™4˜1˜a�QÐ7Ð7Ð7r&   z
No matchesN)rÉ   r   r"   rõ   r    r>   Ú
conditionsrH   r   Úmost_commonr”   r   )r8   rJ   rÝ   rh   r^   r]   rú   s    `   @@r$   ÚsimilarzText.similarã  s  øøø€ ð !¨¬Ð5Ð5å'3Ø”Ð$9Ð$9Ð?RÐ?Rð(ñ (ô (ˆDÔ$ð �zŠz‰|Œ|ˆØÔ&Ô8ˆØ�3—>’>Ñ#Ô#Ð#Ð#Ý˜3˜tœ9‘~”~ˆHÝð ð ð ð ð ð àŸšÑ)Ô)ðñ ô ñ ô ˆBð 8Ð7 2§>¢>°#Ñ#6Ô#6Ð7Ñ7Ô7ˆEÝ•)˜EÑ"Ô"Ñ#Ô#Ð#Ð#Ð#å�,ÑÔÐÐÐr&   c                 ó†  — d| j         vrt          | j        d„ ¬¦  «        | _        	 | j                             |d¦  «        }|st          d¦  «         dS d„ |                     |¦  «        D ¦   «         }t          t          d„ |D ¦   «         ¦  «        ¦  «         dS # t          $ r}t          |¦  «         Y d}~dS d}~ww xY w)	aY  
        Find contexts where the specified words appear; list
        most frequent common contexts first.

        :param words: The words used to seed the similarity search
        :type words: str
        :param num: The number of words to generate (default=20)
        :type num: int
        :seealso: ContextIndex.common_contexts()
        rõ   c                 ó*   — |                       ¦   «         S r(   rÄ   rÅ   s    r$   r,   z&Text.common_contexts.<locals>.<lambda>  rÇ   r&   rÈ   TzNo common contexts were foundc                 ó   — g | ]\  }}|‘ŒS r)   r)   rü   s      r$   r2   z(Text.common_contexts.<locals>.<listcomp>  s   € Ð"EÐ"EÐ"E©¨¨A 1Ð"EÐ"EÐ"Er&   c              3   ó,   K  — | ]\  }}|d z   |z   V — ŒdS )rý   Nr)   rå   s      r$   r9   z'Text.common_contexts.<locals>.<genexpr>  s.   è è € ÐLÐL±&°"°b  S¡¨2¡ÐLÐLÐLÐLÐLÐLr&   N)	rÉ   r   r"   rõ   ri   r”   rÿ   r   rd   )r8   r^   rÝ   rh   Úranked_contextsÚes         r$   ri   zText.common_contexts  sî   € ð !¨¬Ð5Ð5å'3Ø”Ð!4Ð!4ð(ñ (ô (ˆDÔ$ð		ØÔ)×9Ò9¸%ÀÑFÔFˆBØð NÝÐ5Ñ6Ô6Ð6Ð6Ð6à"EÐ"E°·²ÀÑ1DÔ1DÐ"EÑ"EÔ"E�Ý•iÐLÐL¸OÐLÑLÔLÑLÔLÑMÔMÐMÐMÐMøåð 	ð 	ð 	Ý�!‰HŒHˆHˆHˆHˆHˆHˆHˆHøøøøð	øøøs   §,B ÁAB Â
C Â&B;Â;C c                 ó*   — ddl m}  || |¦  «         dS )zü
        Produce a plot showing the distribution of the words through the text.
        Requires pylab to be installed.

        :param words: The words to be plotted
        :type words: list(str)
        :seealso: nltk.draw.dispersion_plot()
        r   )Údispersion_plotN)Ú	nltk.drawr  )r8   r^   r  s      r$   r  zText.dispersion_plot!  s.   € ð 	.Ð-Ð-Ð-Ð-Ð-àˆ˜˜eÑ$Ô$Ð$Ð$Ð$r&   rÓ   c                 óx   — t          ||¦  «        \  }}t          |¬¦  «        }|                     ||¦  «         |S )N)Úorder)r
   r	   Úfit)r8   Útokenized_sentsrU   Ú
train_dataÚpadded_sentsÚmodels         r$   Ú_train_default_ngram_lmzText._train_default_ngram_lm.  s<   € Ý#<¸QÀÑ#PÔ#PÑ ˆ
�LÝ˜!�‘”ˆØ�	Š	�*˜lÑ+Ô+Ð+Øˆr&   éd   é*   c                 ó¶  — d„ t          d                     | j        ¦  «        ¦  «        D ¦   «         | _        t	          | d¦  «        s<t          dt          j        ¬¦  «         |                      | j        d¬¦  «        | _	        g }|dk    s
J d	¦   «         ‚t          |¦  «        |k     rlt          | j	                             |||¬
¦  «        ¦  «        D ])\  }}|dk    rŒ|dk    r n|                     |¦  «         Œ*|dz  }t          |¦  «        |k     °l|rd                     |¦  «        dz   nd}|t          |d|…         ¦  «        z   }t          |¦  «         |S )a  
        Print random text, generated using a trigram language model.
        See also `help(nltk.lm)`.

        :param length: The length of text to generate (default=100)
        :type length: int

        :param text_seed: Generation can be conditioned on preceding context.
        :type text_seed: list(str)

        :param random_seed: A random seed or an instance of `random.Random`. If provided,
            makes the random sampling part of generation reproducible. (default=42)
        :type random_seed: int
        c                 ó8   — g | ]}|                      d ¦  «        ‘ŒS rä   r£   )r/   Úsents     r$   r2   z!Text.generate.<locals>.<listcomp>D  s/   € ð !
ð !
ð !
Ø $ˆD�JŠJ�s‰OŒOð!
ð !
ð !
r&   r_   Ú_trigram_modelzBuilding ngram index...)ÚfilerÓ   )rU   r   z!The `length` must be more than 0.)Ú	text_seedÚrandom_seedz<s>z</s>r   r›   N)r   re   r"   Ú_tokenized_sentsÚhasattrr”   ÚsysÚstderrr  r  r!   r=   Úgeneraterv   r   )	r8   Úlengthr  r  Úgenerated_tokensÚidxÚtokenÚprefixÚ
output_strs	            r$   r  zText.generate4  s�  € ð !
ð !
Ý(5°c·h²h¸t¼{Ñ6KÔ6KÑ(LÔ(Lð!
ñ !
ô !
ˆÔõ �tÐ-Ñ.Ô.ð 	ÝÐ+µ#´*Ð=Ñ=Ô=Ð=Ø"&×">Ò">ØÔ%¨ð #?ñ #ô #ˆDÔð Ðà˜ŠzˆzˆzÐ>‰zŒzˆzÝÐ"Ñ#Ô# fÒ,Ð,Ý'ØÔ#×,Ò,Ø i¸[ð -ñ ô ñô ð 	/ð 	/‘
��Uð
 ˜E’>�>ØØ˜F’?�?Ø�EØ ×'Ò'¨Ñ.Ô.Ð.Ð.Ø˜1ÑˆKõ Ð"Ñ#Ô# fÒ,Ð,ð /8Ð?�—’˜)Ñ$Ô$ sÑ*Ð*¸RˆØ�iÐ(8¸¸&¸Ô(AÑBÔBÑBˆ
ÝˆjÑÔÐØÐr&   c                 ó:   —  |                       ¦   «         j        |Ž S )zc
        See documentation for FreqDist.plot()
        :seealso: nltk.prob.FreqDist.plot()
        )ÚvocabÚplot)r8   Úargss     r$   r(  z	Text.plotb  s   € ð
 !ˆt�zŠz‰|Œ|Ô  $Ð'Ð'r&   c                 óJ   — d| j         vrt          | ¦  «        | _        | j        S )z.
        :seealso: nltk.prob.FreqDist
        Ú_vocab)rÉ   r   r+  rE   s    r$   r'  z
Text.vocabi  s&   € ð ˜4œ=Ð(Ð(å" 4™.œ.ˆDŒKØŒ{Ðr&   c                 óÆ   — d| j         vrt          | ¦  «        | _        | j                             |¦  «        }d„ |D ¦   «         }t	          t          |d¦  «        ¦  «         dS )aÓ  
        Find instances of the regular expression in the text.
        The text is a list of tokens, and a regexp pattern to match
        a single token must be surrounded by angle brackets.  E.g.

        >>> from nltk.book import text1, text5, text9
        >>> text5.findall("<.*><.*><bro>")
        you rule bro; telling you bro; u twizted bro
        >>> text1.findall("<a>(<.*>)<man>")
        monied; nervous; dangerous; white; white; white; pious; queer; good;
        mature; white; Cape; great; wise; wise; butterless; white; fiendish;
        pale; furious; better; certain; complete; dismasted; younger; brave;
        brave; brave; brave
        >>> text9.findall("<th.*>{3,}")
        thread through those; the thought that; that the thing; the thing
        that; that that thing; through these than through; them that the;
        through the thick; them that they; thought that the

        :param regexp: A regular expression
        :type regexp: str
        Ú_token_searcherc                 ó8   — g | ]}d                       |¦  «        ‘ŒS rä   )re   r¥   s     r$   r2   z Text.findall.<locals>.<listcomp>�  s"   € Ð*Ð*Ð* �—’˜‘”Ð*Ð*Ð*r&   rè   N)rÉ   r™   r-  r©   r”   r   )r8   r¬   r­   s      r$   r©   zText.findallr  sh   € ð.  D¤MÐ1Ð1Ý#0°Ñ#6Ô#6ˆDÔ àÔ#×+Ò+¨FÑ3Ô3ˆØ*Ð* TÐ*Ñ*Ô*ˆÝ�i˜˜dÑ#Ô#Ñ$Ô$Ð$Ð$Ð$r&   z\w+|[\.\!\?]c                 óð  — |dz
  }|dk    rK| j                              ||         ¦  «        s+|dz  }|dk    r | j                              ||         ¦  «        ¯+|dk    r||         nd}|dz   }|t          |¦  «        k     rX| j                              ||         ¦  «        s8|dz  }|t          |¦  «        k     r | j                              ||         ¦  «        ¯8|t          |¦  «        k    r||         nd}||fS )zÙ
        One left & one right token, both case-normalized.  Skip over
        non-sentence-final punctuation.  Used by the ``ContextIndex``
        that is created for ``similar()`` and ``common_contexts()``.
        r   r   r   r   )Ú_CONTEXT_REÚmatchr!   )r8   r"   r#   Újr   r   s         r$   Ú_contextzText._context–  sþ   € ð �‰EˆØ�1Šfˆf˜TÔ-×3Ò3°F¸1´IÑ>Ô>ˆfØ�‰FˆAð �1Šfˆf˜TÔ-×3Ò3°F¸1´IÑ>Ô>ˆfà šF˜Fˆv�aŒyˆy¨	ˆð �‰EˆØ•#�f‘+”+Šoˆo dÔ&6×&<Ò&<¸VÀA¼YÑ&GÔ&GˆoØ�‰FˆAð •#�f‘+”+Šoˆo dÔ&6×&<Ò&<¸VÀA¼YÑ&GÔ&Gˆoà¥# f¡+¤+Ò-Ð-��q”	�	°7ˆà�eˆ}Ðr&   c                 ó   — d| j         z  S ©Nz
<Text: %s>©rº   rE   s    r$   Ú__str__zText.__str__®  ó   € Ø˜dœiÑ'Ð'r&   c                 ó   — d| j         z  S r5  r6  rE   s    r$   r}   zText.__repr__±  r8  r&   r(   )rÀ   r’   )rO   r€   rj   )rÓ   )r  Nr  )rk   rl   rm   rn   r¹   rB   r½   r¿   rË   rŠ   rá   rë   rí   rw   ró   r   ri   r  r  r  r(  r'  r©   r§   Úcompiler0  r3  r7  r}   r)   r&   r$   r¯   r¯   4  s°  € € € € € ðð ð. €LðIð Ið Ið Ið.ð ð ð ð  ð  ðMð Mð Mð Mð*Mð Mð Mð Mð(!"ð !"ð !"ð !"ðF>ð >ð >ð >ð0'ð 'ð 'ð'ð 'ð 'ð"ð "ð "ð  ð   ð   ð   ðDð ð ð ð8%ð %ð %ðð ð ð ð,ð ,ð ,ð ,ð\(ð (ð (ðð ð ð%ð %ð %ðD �"”*˜_Ñ-Ô-€Kðð ð ð0(ð (ð (ð(ð (ð (ð (ð (r&   r¯   c                   ó*   — e Zd ZdZd„ Zd„ Zd„ Zd„ ZdS )ÚTextCollectiona;  A collection of texts, which can be loaded with list of texts, or
    with a corpus consisting of one or more texts, and which supports
    counting, concordancing, collocation discovery, etc.  Initialize a
    TextCollection as follows:

    >>> import nltk.corpus
    >>> from nltk.text import TextCollection
    >>> from nltk.book import text1, text2, text3
    >>> gutenberg = TextCollection(nltk.corpus.gutenberg)
    >>> mytexts = TextCollection([text1, text2, text3])

    Iterating over a TextCollection produces all the tokens of all the
    texts in order.
    c                 óÔ   ‡— t          ‰d¦  «        r ˆfd„‰                     ¦   «         D ¦   «         Š‰| _        t                               | t          ‰¦  «        ¦  «         i | _        d S )Nr^   c                 ó:   •— g | ]}‰                      |¦  «        ‘ŒS r)   )r^   )r/   ÚfÚsources     €r$   r2   z+TextCollection.__init__.<locals>.<listcomp>È  s#   ø€ Ð@Ð@Ð@¨!�f—l’l 1‘o”oÐ@Ð@Ð@r&   )r  ÚfileidsÚ_textsr¯   rB   r   Ú
_idf_cache)r8   r@  s    `r$   rB   zTextCollection.__init__Æ  sh   ø€ Ý�6˜7Ñ#Ô#ð 	AØ@Ð@Ð@Ð@¨v¯~ª~Ñ/?Ô/?Ð@Ñ@Ô@ˆFàˆŒÝ�Š�dÕ-¨fÑ5Ô5Ñ6Ô6Ð6ØˆŒˆˆr&   c                 óL   — |                      |¦  «        t          |¦  «        z  S )z"The frequency of the term in text.)rí   r!   ©r8   ÚtermÚtexts      r$   ÚtfzTextCollection.tfÎ  s   € à�zŠz˜$ÑÔ¥# d¡)¤)Ñ+Ð+r&   c                 ó2  ‡— | j                              ‰¦  «        }|€yt          ˆfd„| j        D ¦   «         ¦  «        }t          | j        ¦  «        dk    rt	          d¦  «        ‚|r$t          t          | j        ¦  «        |z  ¦  «        nd}|| j         ‰<   |S )z¤The number of texts in the corpus divided by the
        number of texts that the term appears in.
        If a term does not appear in the corpus, 0.0 is returned.Nc                 ó   •— g | ]}‰|v ¯d ‘Œ	S )Tr)   )r/   rG  rF  s     €r$   r2   z&TextCollection.idf.<locals>.<listcomp>Ù  s   ø€ ÐHÐHÐH D¸4À4¸<¸<˜4¸<¸<¸<r&   r   z+IDF undefined for empty document collectiong        )rC  rT   r!   rB  rd   r   )r8   rF  ÚidfÚmatchess    `  r$   rK  zTextCollection.idfÒ  sž   ø€ ð
 Œo×!Ò! $Ñ'Ô'ˆØˆ;ÝÐHÐHÐHÐH¨D¬KÐHÑHÔHÑIÔIˆGÝ�4”;ÑÔ 1Ò$Ð$Ý Ð!NÑOÔOÐOØ5<ÐE•#•c˜$œ+Ñ&Ô&¨Ñ0Ñ1Ô1Ð1À#ˆCØ$'ˆDŒO˜DÑ!Øˆ
r&   c                 óZ   — |                       ||¦  «        |                      |¦  «        z  S r(   )rH  rK  rE  s      r$   Útf_idfzTextCollection.tf_idfà  s%   € Ø�wŠw�t˜TÑ"Ô" T§X¢X¨d¡^¤^Ñ3Ð3r&   N)rk   rl   rm   rn   rB   rH  rK  rN  r)   r&   r$   r<  r<  ¶  sZ   € € € € € ðð ðð ð ð,ð ,ð ,ðð ð ð4ð 4ð 4ð 4ð 4r&   r<  c                  óR  — ddl m}  t          |                      d¬¦  «        ¦  «        }t	          |¦  «         t	          ¦   «          t	          d¦  «         |                     d¦  «         t	          ¦   «          t	          d¦  «         |                     d¦  «         t	          ¦   «          t	          d¦  «         |                     ¦   «          t	          ¦   «          t	          d¦  «         |                     g d	¢¦  «         t	          ¦   «          t	          d
¦  «         | 	                    d¦  «         t	          ¦   «          t	          d¦  «         t	          d|d         ¦  «         t	          d|dd…         ¦  «         t	          d| 
                    ¦   «         d         ¦  «         d S )Nr   )ÚbrownÚnews)Ú
categorieszConcordance:zDistributionally similar words:zCollocations:zDispersion plot:)rQ  ÚreportÚsaidÚ	announcedzVocabulary plot:é2   z	Indexing:ztext[3]:rÓ   z
text[3:5]:é   ztext.vocab()['news']:)r×   rP  r¯   r^   r”   rË   r   rë   r  r(  r'  )rP  rG  s     r$   ÚdemorX  ä  s{  € Ø!Ð!Ð!Ð!Ð!Ð!å�—’ v�Ñ.Ô.Ñ/Ô/€DÝ	ˆ$�K„K€KÝ	�G„G€GÝ	ˆ.ÑÔÐØ×Ò�VÑÔÐÝ	�G„G€GÝ	Ð
+Ñ,Ô,Ð,Ø‡L‚L�ÑÔÐÝ	�G„G€GÝ	ˆ/ÑÔÐØ×ÒÑÔÐÝ	�G„G€Gõ 
Ð
ÑÔÐØ×ÒÐ@Ð@Ð@ÑAÔAÐAÝ	�G„G€GÝ	Ð
ÑÔÐØ‡I‚Iˆb�M„M€MÝ	�G„G€GÝ	ˆ+ÑÔÐÝ	ˆ*�d˜1”gÑÔÐÝ	ˆ,˜˜Q˜q˜Sœ	Ñ"Ô"Ð"Ý	Ð
! 4§:¢:¡<¤<°Ô#7Ñ8Ô8Ð8Ð8Ð8r&   Ú__main__)r   rq   r™   r¯   r<  )&rn   r§   r  Úcollectionsr   r   r   Ú	functoolsr   Úmathr   Únltk.collocationsr   Únltk.lmr	   Únltk.lm.preprocessingr
   Únltk.metricsr   r   Únltk.probabilityr   r<   r   Únltk.tokenizer   Ú	nltk.utilr   r   r   r   rq   r™   r¯   r<  rX  rk   Ú__all__r)   r&   r$   ú<module>re     sA  ððð ð 
€	€	€	Ø 
€
€
€
Ø 8Ð 8Ð 8Ð 8Ð 8Ð 8Ð 8Ð 8Ð 8Ð 8Ø Ð Ð Ð Ð Ð Ø Ð Ð Ð Ð Ð à 5Ð 5Ð 5Ð 5Ð 5Ð 5Ø Ð Ð Ð Ð Ð Ø ;Ð ;Ð ;Ð ;Ð ;Ð ;Ø 7Ð 7Ð 7Ð 7Ð 7Ð 7Ð 7Ð 7Ø 7Ð 7Ð 7Ð 7Ð 7Ð 7Ø %Ð %Ð %Ð %Ð %Ð %Ø 'Ð 'Ð 'Ð 'Ð 'Ð 'Ø 2Ð 2Ð 2Ð 2Ð 2Ð 2Ð 2Ð 2à�*ØØMÐMÐMñô €ðXð Xð Xð Xð Xñ Xô Xð Xðvx-ð x-ð x-ð x-ð x-ñ x-ô x-ð x-ðv5ð 5ð 5ð 5ð 5ñ 5ô 5ð 5ðp~(ð ~(ð ~(ð ~(ð ~(ñ ~(ô ~(ð ~(ðD+4ð +4ð +4ð +4ð +4�Tñ +4ô +4ð +4ð\9ð 9ð 9ð< ˆzÒÐØ€D�F„F€Fðð ð €€€r&   