§
    'ê[f¶  ã                   óX   — d Z ddlZddlmZ  G d„ de¦  «        Z e¦   «         j        ZdS )aê  
S-Expression Tokenizer

``SExprTokenizer`` is used to find parenthesized expressions in a
string.  In particular, it divides a string into a sequence of
substrings that are either parenthesized expressions (including any
nested parenthesized expressions), or other whitespace-separated
tokens.

    >>> from nltk.tokenize import SExprTokenizer
    >>> SExprTokenizer().tokenize('(a b (c d)) e f (g)')
    ['(a b (c d))', 'e', 'f', '(g)']

By default, `SExprTokenizer` will raise a ``ValueError`` exception if
used to tokenize an expression with non-matching parentheses:

    >>> SExprTokenizer().tokenize('c) d) e (f (g')
    Traceback (most recent call last):
      ...
    ValueError: Un-matched close paren at char 1

The ``strict`` argument can be set to False to allow for
non-matching parentheses.  Any unmatched close parentheses will be
listed as their own s-expression; and the last partial sexpr with
unmatched open parentheses will be listed as its own sexpr:

    >>> SExprTokenizer(strict=False).tokenize('c) d) e (f (g')
    ['c', ')', 'd', ')', 'e', '(f (g']

The characters used for open and close parentheses may be customized
using the ``parens`` argument to the `SExprTokenizer` constructor:

    >>> SExprTokenizer(parens='{}').tokenize('{a b {c d}} e f {g}')
    ['{a b {c d}}', 'e', 'f', '{g}']

The s-expression tokenizer is also available as a function:

    >>> from nltk.tokenize import sexpr_tokenize
    >>> sexpr_tokenize('(a b (c d)) e f (g)')
    ['(a b (c d))', 'e', 'f', '(g)']

é    N)Ú
TokenizerIc                   ó    — e Zd ZdZdd„Zd„ ZdS )ÚSExprTokenizera\  
    A tokenizer that divides strings into s-expressions.
    An s-expresion can be either:

      - a parenthesized expression, including any nested parenthesized
        expressions, or
      - a sequence of non-whitespace non-parenthesis characters.

    For example, the string ``(a (b c)) d e (f)`` consists of four
    s-expressions: ``(a (b c))``, ``d``, ``e``, and ``(f)``.

    By default, the characters ``(`` and ``)`` are treated as open and
    close parentheses, but alternative strings may be specified.

    :param parens: A two-element sequence specifying the open and close parentheses
        that should be used to find sexprs.  This will typically be either a
        two-character string, or a list of two strings.
    :type parens: str or list
    :param strict: If true, then raise an exception when tokenizing an ill-formed sexpr.
    ú()Tc                 ó(  — t          |¦  «        dk    rt          d¦  «        ‚|| _        |d         | _        |d         | _        t          j        t          j        |d         ¦  «        › dt          j        |d         ¦  «        › �¦  «        | _        d S )Né   z'parens must contain exactly two stringsr   é   Ú|)	ÚlenÚ
ValueErrorÚ_strictÚ_open_parenÚ_close_parenÚreÚcompileÚescapeÚ_paren_regexp)ÚselfÚparensÚstricts      úG/var/www/piapp/venv/lib/python3.11/site-packages/nltk/tokenize/sexpr.pyÚ__init__zSExprTokenizer.__init__O   sˆ   € Ýˆv‰;Œ;˜!ÒÐÝÐFÑGÔGÐGØˆŒØ! !œ9ˆÔØ" 1œIˆÔÝœZÝŒy˜ œÑ#Ô#Ð<Ð<¥b¤i°°q´	Ñ&:Ô&:Ð<Ð<ñ
ô 
ˆÔÐÐó    c                 óü  — g }d}d}| j                              |¦  «        D �]
}|                     ¦   «         }|dk    rE||||                     ¦   «         …                              ¦   «         z  }|                     ¦   «         }|| j        k    r|dz  }|| j        k    r�| j        r*|dk    r$t          d|                     ¦   «         z  ¦  «        ‚t          d|dz
  ¦  «        }|dk    rC| 
                    |||                     ¦   «         …         ¦  «         |                     ¦   «         }�Œ| j        r|dk    rt          d|z  ¦  «        ‚|t          |¦  «        k     r| 
                    ||d…         ¦  «         |S )aQ  
        Return a list of s-expressions extracted from *text*.
        For example:

            >>> SExprTokenizer().tokenize('(a b (c d)) e f (g)')
            ['(a b (c d))', 'e', 'f', '(g)']

        All parentheses are assumed to mark s-expressions.
        (No special processing is done to exclude parentheses that occur
        inside strings, or following backslash characters.)

        If the given expression contains non-matching parentheses,
        then the behavior of the tokenizer depends on the ``strict``
        parameter to the constructor.  If ``strict`` is ``True``, then
        raise a ``ValueError``.  If ``strict`` is ``False``, then any
        unmatched close parentheses will be listed as their own
        s-expression; and the last partial s-expression with unmatched open
        parentheses will be listed as its own s-expression:

            >>> SExprTokenizer(strict=False).tokenize('c) d) e (f (g')
            ['c', ')', 'd', ')', 'e', '(f (g']

        :param text: the string to be tokenized
        :type text: str or iter(str)
        :rtype: iter(str)
        r   r	   z!Un-matched close paren at char %dz Un-matched open paren at char %dN)r   ÚfinditerÚgroupÚstartÚsplitr   r   r   r   ÚmaxÚappendÚendr   )r   ÚtextÚresultÚposÚdepthÚmÚparens          r   ÚtokenizezSExprTokenizer.tokenizeY   ss  € ð6 ˆØˆØˆØÔ#×,Ò,¨TÑ2Ô2ð 	"ñ 	"ˆAØ—G’G‘I”IˆEØ˜ŠzˆzØ˜$˜s Q§W¢W¡Y¤Y˜Ô/×5Ò5Ñ7Ô7Ñ7�Ø—g’g‘i”i�Ø˜Ô(Ò(Ð(Ø˜‘
�Ø˜Ô)Ò)Ð)Ø”<ð V E¨Q¢J JÝ$Ð%HÈ1Ï7Ê7É9Ì9Ñ%TÑUÔUÐUÝ˜A˜u q™yÑ)Ô)�Ø˜A’:�:Ø—M’M $ s¨Q¯UªU©W¬W }Ô"5Ñ6Ô6Ð6ØŸ%š%™'œ'�CùØŒ<ð 	G˜E AšI˜IÝÐ?À#ÑEÑFÔFÐFØ•�T‘”Š?ˆ?Ø�MŠM˜$˜s˜t˜tœ*Ñ%Ô%Ð%Øˆr   N)r   T)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r(   © r   r   r   r   9   sA   € € € € € ðð ð*
ð 
ð 
ð 
ð0ð 0ð 0ð 0ð 0r   r   )r,   r   Únltk.tokenize.apir   r   r(   Úsexpr_tokenizer-   r   r   ú<module>r0      sv   ðð)ð )ðV 
€	€	€	à (Ð (Ð (Ð (Ð (Ð (ðPð Pð Pð Pð P�Zñ Pô Pð Pðf  �Ñ!Ô!Ô*€€€r   