§
    'ê[f  ã                   ó¢   — d Z ddlmZmZ ddlmZmZ  G d„ de¦  «        Z G d„ de¦  «        Z G d„ d	e¦  «        Z	 G d
„ de¦  «        Z
dd„ZdS )aÔ  
Simple Tokenizers

These tokenizers divide strings into substrings using the string
``split()`` method.
When tokenizing using a particular delimiter string, use
the string ``split()`` method directly, as this is more efficient.

The simple tokenizers are *not* available as separate functions;
instead, you should just use the string ``split()`` method directly:

    >>> s = "Good muffins cost $3.88\nin New York.  Please buy me\ntwo of them.\n\nThanks."
    >>> s.split() # doctest: +NORMALIZE_WHITESPACE
    ['Good', 'muffins', 'cost', '$3.88', 'in', 'New', 'York.',
    'Please', 'buy', 'me', 'two', 'of', 'them.', 'Thanks.']
    >>> s.split(' ') # doctest: +NORMALIZE_WHITESPACE
    ['Good', 'muffins', 'cost', '$3.88\nin', 'New', 'York.', '',
    'Please', 'buy', 'me\ntwo', 'of', 'them.\n\nThanks.']
    >>> s.split('\n') # doctest: +NORMALIZE_WHITESPACE
    ['Good muffins cost $3.88', 'in New York.  Please buy me',
    'two of them.', '', 'Thanks.']

The simple tokenizers are mainly useful because they follow the
standard ``TokenizerI`` interface, and so can be used with any code
that expects a tokenizer.  For example, these tokenizers can be used
to specify the tokenization conventions when building a `CorpusReader`.

é    )ÚStringTokenizerÚ
TokenizerI)Úregexp_span_tokenizeÚstring_span_tokenizec                   ó   — e Zd ZdZdZdS )ÚSpaceTokenizeraÎ  Tokenize a string using the space character as a delimiter,
    which is the same as ``s.split(' ')``.

        >>> from nltk.tokenize import SpaceTokenizer
        >>> s = "Good muffins cost $3.88\nin New York.  Please buy me\ntwo of them.\n\nThanks."
        >>> SpaceTokenizer().tokenize(s) # doctest: +NORMALIZE_WHITESPACE
        ['Good', 'muffins', 'cost', '$3.88\nin', 'New', 'York.', '',
        'Please', 'buy', 'me\ntwo', 'of', 'them.\n\nThanks.']
    Ú N©Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú_string© ó    úH/var/www/piapp/venv/lib/python3.11/site-packages/nltk/tokenize/simple.pyr   r   *   s   € € € € € ðð ð €G€G€Gr   r   c                   ó   — e Zd ZdZdZdS )ÚTabTokenizerzäTokenize a string use the tab character as a delimiter,
    the same as ``s.split('\t')``.

        >>> from nltk.tokenize import TabTokenizer
        >>> TabTokenizer().tokenize('a\tb c\n\t d')
        ['a', 'b c\n', ' d']
    ú	Nr
   r   r   r   r   r   8   s   € € € € € ðð ð €G€G€Gr   r   c                   ó   — e Zd ZdZd„ Zd„ ZdS )ÚCharTokenizerz„Tokenize a string into individual characters.  If this functionality
    is ever required directly, use ``for char in string``.
    c                 ó    — t          |¦  «        S ©N)Úlist©ÚselfÚss     r   ÚtokenizezCharTokenizer.tokenizeI   s   € Ý�A‰wŒwˆr   c              #   óp   K  — t          t          dt          |¦  «        dz   ¦  «        ¦  «        E d {V —† d S )Né   )Ú	enumerateÚrangeÚlenr   s     r   Úspan_tokenizezCharTokenizer.span_tokenizeL   s@   è è € Ý�U 1¥c¨!¡f¤f¨q¡jÑ1Ô1Ñ2Ô2Ð2Ð2Ð2Ð2Ð2Ð2Ð2Ð2Ð2r   N)r   r   r   r   r   r$   r   r   r   r   r   D   s<   € € € € € ðð ðð ð ð3ð 3ð 3ð 3ð 3r   r   c                   ó&   — e Zd ZdZdd„Zd„ Zd„ ZdS )ÚLineTokenizera˜  Tokenize a string into its lines, optionally discarding blank lines.
    This is similar to ``s.split('\n')``.

        >>> from nltk.tokenize import LineTokenizer
        >>> s = "Good muffins cost $3.88\nin New York.  Please buy me\ntwo of them.\n\nThanks."
        >>> LineTokenizer(blanklines='keep').tokenize(s) # doctest: +NORMALIZE_WHITESPACE
        ['Good muffins cost $3.88', 'in New York.  Please buy me',
        'two of them.', '', 'Thanks.']
        >>> # same as [l for l in s.split('\n') if l.strip()]:
        >>> LineTokenizer(blanklines='discard').tokenize(s) # doctest: +NORMALIZE_WHITESPACE
        ['Good muffins cost $3.88', 'in New York.  Please buy me',
        'two of them.', 'Thanks.']

    :param blanklines: Indicates how blank lines should be handled.  Valid values are:

        - ``discard``: strip blank lines out of the token list before returning it.
           A line is considered blank if it contains only whitespace characters.
        - ``keep``: leave all blank lines in the token list.
        - ``discard-eof``: if the string ends with a newline, then do not generate
           a corresponding token ``''`` after that newline.
    Údiscardc                 ój   — d}||vr%t          dd                     |¦  «        z  ¦  «        ‚|| _        d S )N)r'   Úkeepúdiscard-eofzBlank lines must be one of: %sr	   )Ú
ValueErrorÚjoinÚ_blanklines)r   Ú
blanklinesÚvalid_blankliness      r   Ú__init__zLineTokenizer.__init__g   sK   € Ø=ÐØÐ-Ð-Ð-ÝØ0°3·8²8Ð<LÑ3MÔ3MÑMñô ð ð &ˆÔÐÐr   c                 óÔ   — |                      ¦   «         }| j        dk    rd„ |D ¦   «         }n;| j        dk    r0|r.|d                              ¦   «         s|                     ¦   «          |S )Nr'   c                 ó:   — g | ]}|                      ¦   «         ¯|‘ŒS r   )Úrstrip)Ú.0Úls     r   ú
<listcomp>z*LineTokenizer.tokenize.<locals>.<listcomp>t   s%   € Ð4Ð4Ð4˜1¨¯ª©¬Ð4�QÐ4Ð4Ð4r   r*   éÿÿÿÿ)Ú
splitlinesr-   ÚstripÚpop)r   r   Úliness      r   r   zLineTokenizer.tokenizep   so   € Ø—’‘”ˆàÔ˜yÒ(Ð(Ø4Ð4 Ð4Ñ4Ô4ˆEˆEØÔ Ò.Ð.Øð ˜U 2œYŸ_š_Ñ.Ô.ð Ø—	’	‘”�Øˆr   c              #   ó|   K  — | j         dk    rt          |d¦  «        E d {V —† d S t          |d¦  «        E d {V —† d S )Nr)   z\nz
\n(\s+\n)*)r-   r   r   r   s     r   r$   zLineTokenizer.span_tokenize{   sd   è è € ØÔ˜vÒ%Ð%Ý+¨A¨uÑ5Ô5Ð5Ð5Ð5Ð5Ð5Ð5Ð5Ð5Ð5å+¨A¨}Ñ=Ô=Ð=Ð=Ð=Ð=Ð=Ð=Ð=Ð=Ð=r   N©r'   )r   r   r   r   r0   r   r$   r   r   r   r&   r&   P   sP   € € € € € ðð ð,&ð &ð &ð &ðð ð ð>ð >ð >ð >ð >r   r&   r'   c                 óF   — t          |¦  «                             | ¦  «        S r   )r&   r   )Útextr.   s     r   Úline_tokenizer@   ˆ   s   € Ý˜Ñ$Ô$×-Ò-¨dÑ3Ô3Ð3r   Nr=   )r   Únltk.tokenize.apir   r   Únltk.tokenize.utilr   r   r   r   r   r&   r@   r   r   r   ú<module>rC      s  ððð ð: :Ð 9Ð 9Ð 9Ð 9Ð 9Ð 9Ð 9Ø IÐ IÐ IÐ IÐ IÐ IÐ IÐ Iðð ð ð ð �_ñ ô ð ð	ð 	ð 	ð 	ð 	�?ñ 	ô 	ð 	ð	3ð 	3ð 	3ð 	3ð 	3�Oñ 	3ô 	3ð 	3ð/>ð />ð />ð />ð />�Jñ />ô />ð />ðp4ð 4ð 4ð 4ð 4ð 4r   