§
    'ê[f{  ã                   ó*  — d Z ddlZddlmZ ddlmZmZ ddlmZ ddl	m
Z
 ddlmZ ddlmZ dd	lmZmZmZmZmZmZmZ dd
lmZ ddlmZmZ ddlmZmZmZm Z  ddl!m"Z" ddl#m$Z$ ddl%m&Z& ddl'm(Z( ddl)m*Z*m+Z+ ddl,m-Z-m.Z. dd„Z/ e¦   «         Z0dd„Z1dS )a@	  
NLTK Tokenizer Package

Tokenizers divide strings into lists of substrings.  For example,
tokenizers can be used to find the words and punctuation in a string:

    >>> from nltk.tokenize import word_tokenize
    >>> s = '''Good muffins cost $3.88\nin New York.  Please buy me
    ... two of them.\n\nThanks.'''
    >>> word_tokenize(s) # doctest: +NORMALIZE_WHITESPACE
    ['Good', 'muffins', 'cost', '$', '3.88', 'in', 'New', 'York', '.',
    'Please', 'buy', 'me', 'two', 'of', 'them', '.', 'Thanks', '.']

This particular tokenizer requires the Punkt sentence tokenization
models to be installed. NLTK also provides a simpler,
regular-expression based tokenizer, which splits text on whitespace
and punctuation:

    >>> from nltk.tokenize import wordpunct_tokenize
    >>> wordpunct_tokenize(s) # doctest: +NORMALIZE_WHITESPACE
    ['Good', 'muffins', 'cost', '$', '3', '.', '88', 'in', 'New', 'York', '.',
    'Please', 'buy', 'me', 'two', 'of', 'them', '.', 'Thanks', '.']

We can also operate at the level of sentences, using the sentence
tokenizer directly as follows:

    >>> from nltk.tokenize import sent_tokenize, word_tokenize
    >>> sent_tokenize(s)
    ['Good muffins cost $3.88\nin New York.', 'Please buy me\ntwo of them.', 'Thanks.']
    >>> [word_tokenize(t) for t in sent_tokenize(s)] # doctest: +NORMALIZE_WHITESPACE
    [['Good', 'muffins', 'cost', '$', '3.88', 'in', 'New', 'York', '.'],
    ['Please', 'buy', 'me', 'two', 'of', 'them', '.'], ['Thanks', '.']]

Caution: when tokenizing a Unicode string, make sure you are not
using an encoded version of the string (it may be necessary to
decode it first, e.g. with ``s.decode("utf8")``.

NLTK tokenizers can produce token-spans, represented as tuples of integers
having the same semantics as string slices, to support efficient comparison
of tokenizers.  (These methods are implemented as generators.)

    >>> from nltk.tokenize import WhitespaceTokenizer
    >>> list(WhitespaceTokenizer().span_tokenize(s)) # doctest: +NORMALIZE_WHITESPACE
    [(0, 4), (5, 12), (13, 17), (18, 23), (24, 26), (27, 30), (31, 36), (38, 44),
    (45, 48), (49, 51), (52, 55), (56, 58), (59, 64), (66, 73)]

There are numerous ways to tokenize text.  If you need more control over
tokenization, see the other methods provided in this package.

For further information, please see Chapter 3 of the NLTK book.
é    N)Úload)ÚTweetTokenizerÚcasual_tokenize)ÚNLTKWordTokenizer)ÚLegalitySyllableTokenizer)ÚMWETokenizer)ÚPunktSentenceTokenizer)ÚBlanklineTokenizerÚRegexpTokenizerÚWhitespaceTokenizerÚWordPunctTokenizerÚblankline_tokenizeÚregexp_tokenizeÚwordpunct_tokenize)ÚReppTokenizer)ÚSExprTokenizerÚsexpr_tokenize)ÚLineTokenizerÚSpaceTokenizerÚTabTokenizerÚline_tokenize)ÚSyllableTokenizer)ÚStanfordSegmenter)ÚTextTilingTokenizer)ÚToktokTokenizer)ÚTreebankWordDetokenizerÚTreebankWordTokenizer)Úregexp_span_tokenizeÚstring_span_tokenizeÚenglishc                 óR   — t          d|› d�¦  «        }|                     | ¦  «        S )a  
    Return a sentence-tokenized copy of *text*,
    using NLTK's recommended sentence tokenizer
    (currently :class:`.PunktSentenceTokenizer`
    for the specified language).

    :param text: text to split into sentences
    :param language: the model name in the Punkt corpus
    ztokenizers/punkt/z.pickle)r   Útokenize)ÚtextÚlanguageÚ	tokenizers      úJ/var/www/piapp/venv/lib/python3.11/site-packages/nltk/tokenize/__init__.pyÚsent_tokenizer'   `   s1   € õ Ð:¨Ð:Ð:Ð:Ñ;Ô;€IØ×Ò˜dÑ#Ô#Ð#ó    Fc                 óD   — |r| gnt          | |¦  «        }d„ |D ¦   «         S )aê  
    Return a tokenized copy of *text*,
    using NLTK's recommended word tokenizer
    (currently an improved :class:`.TreebankWordTokenizer`
    along with :class:`.PunktSentenceTokenizer`
    for the specified language).

    :param text: text to split into words
    :type text: str
    :param language: the model name in the Punkt corpus
    :type language: str
    :param preserve_line: A flag to decide whether to sentence tokenize the text or not.
    :type preserve_line: bool
    c                 óL   — g | ]!}t                                |¦  «        D ]}|‘ŒŒ"S © )Ú_treebank_word_tokenizerr"   )Ú.0ÚsentÚtokens      r&   ú
<listcomp>z!word_tokenize.<locals>.<listcomp>‚   sI   € ð ð ð ØÕ1I×1RÒ1RÐSWÑ1XÔ1Xðð Ø(-ˆðð ð ð r(   )r'   )r#   r$   Úpreserve_lineÚ	sentencess       r&   Úword_tokenizer3   r   s?   € ð (ÐJ���­]¸4ÀÑ-JÔ-J€Iðð Ø#ðñ ô ð r(   )r    )r    F)2Ú__doc__ÚreÚ	nltk.datar   Únltk.tokenize.casualr   r   Únltk.tokenize.destructiver   Ú nltk.tokenize.legality_principler   Únltk.tokenize.mwer   Únltk.tokenize.punktr	   Únltk.tokenize.regexpr
   r   r   r   r   r   r   Únltk.tokenize.reppr   Únltk.tokenize.sexprr   r   Únltk.tokenize.simpler   r   r   r   Ú!nltk.tokenize.sonority_sequencingr   Ú nltk.tokenize.stanford_segmenterr   Únltk.tokenize.texttilingr   Únltk.tokenize.toktokr   Únltk.tokenize.treebankr   r   Únltk.tokenize.utilr   r   r'   r,   r3   r+   r(   r&   ú<module>rF      s  ðð2ð 2ðh 
€	€	€	à Ð Ð Ð Ð Ð Ø @Ð @Ð @Ð @Ð @Ð @Ð @Ð @Ø 7Ð 7Ð 7Ð 7Ð 7Ð 7Ø FÐ FÐ FÐ FÐ FÐ FØ *Ð *Ð *Ð *Ð *Ð *Ø 6Ð 6Ð 6Ð 6Ð 6Ð 6ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð -Ð ,Ð ,Ð ,Ð ,Ð ,Ø >Ð >Ð >Ð >Ð >Ð >Ð >Ð >ðð ð ð ð ð ð ð ð ð ð ð ð @Ð ?Ð ?Ð ?Ð ?Ð ?Ø >Ð >Ð >Ð >Ð >Ð >Ø 8Ð 8Ð 8Ð 8Ð 8Ð 8Ø 0Ð 0Ð 0Ð 0Ð 0Ð 0Ø QÐ QÐ QÐ QÐ QÐ QÐ QÐ QØ IÐ IÐ IÐ IÐ IÐ IÐ IÐ Ið$ð $ð $ð $ð -Ð,Ñ.Ô.Ð ðð ð ð ð ð r(   