§
    'ê[fA  ã                   óŠ   — d Z ddlZddlZddlmZmZmZ ddlmZ ddl	m
Z
 ddlmZ  G d„ de¦  «        Z G d	„ d
e¦  «        ZdS )a	  

Penn Treebank Tokenizer

The Treebank tokenizer uses regular expressions to tokenize text as in Penn Treebank.
This implementation is a port of the tokenizer sed script written by Robert McIntyre
and available at http://www.cis.upenn.edu/~treebank/tokenizer.sed.
é    N)ÚIteratorÚListÚTuple)Ú
TokenizerI)ÚMacIntyreContractions)Úalign_tokensc            
       óö  — e Zd ZdZ ej        d¦  «        df ej        d¦  «        df ej        d¦  «        dfgZ ej        d¦  «        d	f ej        d
¦  «        df ej        d¦  «        df ej        d¦  «        df ej        d¦  «        df ej        d¦  «        df ej        d¦  «        dfgZ ej        d¦  «        dfZ ej        d¦  «        df ej        d¦  «        df ej        d¦  «        df ej        d¦  «        df ej        d¦  «        df ej        d¦  «        d fgZ	 ej        d!¦  «        d"fZ
 ej        d#¦  «        d$f ej        d%¦  «        d$f ej        d&¦  «        d'f ej        d(¦  «        d'fgZ e¦   «         Z e eej        ej        ¦  «        ¦  «        Z e eej        ej        ¦  «        ¦  «        Z	 d1d*ed+ed,ed-ee         fd.„Zd*ed-eeeef                  fd/„Zd0S )2ÚTreebankWordTokenizera	  
    The Treebank tokenizer uses regular expressions to tokenize text as in Penn Treebank.

    This tokenizer performs the following steps:

    - split standard contractions, e.g. ``don't`` -> ``do n't`` and ``they'll`` -> ``they 'll``
    - treat most punctuation characters as separate tokens
    - split off commas and single quotes, when followed by whitespace
    - separate periods that appear at the end of line

    >>> from nltk.tokenize import TreebankWordTokenizer
    >>> s = '''Good muffins cost $3.88\nin New York.  Please buy me\ntwo of them.\nThanks.'''
    >>> TreebankWordTokenizer().tokenize(s)
    ['Good', 'muffins', 'cost', '$', '3.88', 'in', 'New', 'York.', 'Please', 'buy', 'me', 'two', 'of', 'them.', 'Thanks', '.']
    >>> s = "They'll save and invest more."
    >>> TreebankWordTokenizer().tokenize(s)
    ['They', "'ll", 'save', 'and', 'invest', 'more', '.']
    >>> s = "hi, my name can't hello,"
    >>> TreebankWordTokenizer().tokenize(s)
    ['hi', ',', 'my', 'name', 'ca', "n't", 'hello', ',']
    z^\"ú``z(``)z \1 z([ \(\[{<])(\"|\'{2})z\1 `` z([:,])([^\d])z \1 \2z([:,])$z\.\.\.z ... z[;@#$%&]z \g<0> z([^\.])(\.)([\]\)}>"\']*)\s*$z\1 \2\3 z[?!]z([^'])' z\1 ' z[\]\[\(\)\{\}\<\>]z\(ú-LRB-z\)ú-RRB-z\[ú-LSB-z\]ú-RSB-z\{ú-LCB-z\}ú-RCB-ú--ú -- ú''z '' ú"z([^' ])('[sS]|'[mM]|'[dD]|') z\1 \2 z)([^' ])('ll|'LL|'re|'RE|'ve|'VE|n't|N'T) FÚtextÚconvert_parenthesesÚ
return_strÚreturnc                 ó–  — |durt          j        dt          d¬¦  «         | j        D ]\  }}|                     ||¦  «        }Œ| j        D ]\  }}|                     ||¦  «        }Œ| j        \  }}|                     ||¦  «        }|r#| j        D ]\  }}|                     ||¦  «        }Œ| j        \  }}|                     ||¦  «        }d|z   dz   }| j	        D ]\  }}|                     ||¦  «        }Œ| j
        D ]}|                     d|¦  «        }Œ| j        D ]}|                     d|¦  «        }Œ|                     ¦   «         S )aý  Return a tokenized copy of `text`.

        >>> from nltk.tokenize import TreebankWordTokenizer
        >>> s = '''Good muffins cost $3.88 (roughly 3,36 euros)\nin New York.  Please buy me\ntwo of them.\nThanks.'''
        >>> TreebankWordTokenizer().tokenize(s) # doctest: +NORMALIZE_WHITESPACE
        ['Good', 'muffins', 'cost', '$', '3.88', '(', 'roughly', '3,36',
        'euros', ')', 'in', 'New', 'York.', 'Please', 'buy', 'me', 'two',
        'of', 'them.', 'Thanks', '.']
        >>> TreebankWordTokenizer().tokenize(s, convert_parentheses=True) # doctest: +NORMALIZE_WHITESPACE
        ['Good', 'muffins', 'cost', '$', '3.88', '-LRB-', 'roughly', '3,36',
        'euros', '-RRB-', 'in', 'New', 'York.', 'Please', 'buy', 'me', 'two',
        'of', 'them.', 'Thanks', '.']

        :param text: A string with a sentence or sentences.
        :type text: str
        :param convert_parentheses: if True, replace parentheses to PTB symbols,
            e.g. `(` to `-LRB-`. Defaults to False.
        :type convert_parentheses: bool, optional
        :param return_str: If True, return tokens as space-separated string,
            defaults to False.
        :type return_str: bool, optional
        :return: List of tokens from `text`.
        :rtype: List[str]
        FzHParameter 'return_str' has been deprecated and should no longer be used.é   )ÚcategoryÚ
stacklevelÚ z \1 \2 )ÚwarningsÚwarnÚDeprecationWarningÚSTARTING_QUOTESÚsubÚPUNCTUATIONÚPARENS_BRACKETSÚCONVERT_PARENTHESESÚDOUBLE_DASHESÚENDING_QUOTESÚCONTRACTIONS2ÚCONTRACTIONS3Úsplit)Úselfr   r   r   ÚregexpÚsubstitutions         úJ/var/www/piapp/venv/lib/python3.11/site-packages/nltk/tokenize/treebank.pyÚtokenizezTreebankWordTokenizer.tokenizee   sŸ  € ð6 ˜UÐ"Ð"ÝŒMð"å+Øð	ñ ô ð ð %)Ô$8ð 	2ð 	2Ñ ˆF�LØ—:’:˜l¨DÑ1Ô1ˆDˆDà$(Ô$4ð 	2ð 	2Ñ ˆF�LØ—:’:˜l¨DÑ1Ô1ˆDˆDð  $Ô3Ñˆ�Ø�zŠz˜,¨Ñ-Ô-ˆàð 	6Ø(,Ô(@ð 6ð 6Ñ$�˜Ø—z’z ,°Ñ5Ô5��ð  $Ô1Ñˆ�Ø�zŠz˜,¨Ñ-Ô-ˆð �T‰z˜CÑˆà$(Ô$6ð 	2ð 	2Ñ ˆF�LØ—:’:˜l¨DÑ1Ô1ˆDˆDàÔ(ð 	0ð 	0ˆFØ—:’:˜j¨$Ñ/Ô/ˆDˆDØÔ(ð 	0ð 	0ˆFØ—:’:˜j¨$Ñ/Ô/ˆDˆDð �zŠz‰|Œ|Ðó    c              #   óÒ   ‡K  — |                       |¦  «        }d|v sd|v r.d„ t          j        d|¦  «        D ¦   «         Šˆfd„|D ¦   «         }n|}t          ||¦  «        E d{V —† dS )a‰  
        Returns the spans of the tokens in ``text``.
        Uses the post-hoc nltk.tokens.align_tokens to return the offset spans.

            >>> from nltk.tokenize import TreebankWordTokenizer
            >>> s = '''Good muffins cost $3.88\nin New (York).  Please (buy) me\ntwo of them.\n(Thanks).'''
            >>> expected = [(0, 4), (5, 12), (13, 17), (18, 19), (19, 23),
            ... (24, 26), (27, 30), (31, 32), (32, 36), (36, 37), (37, 38),
            ... (40, 46), (47, 48), (48, 51), (51, 52), (53, 55), (56, 59),
            ... (60, 62), (63, 68), (69, 70), (70, 76), (76, 77), (77, 78)]
            >>> list(TreebankWordTokenizer().span_tokenize(s)) == expected
            True
            >>> expected = ['Good', 'muffins', 'cost', '$', '3.88', 'in',
            ... 'New', '(', 'York', ')', '.', 'Please', '(', 'buy', ')',
            ... 'me', 'two', 'of', 'them.', '(', 'Thanks', ')', '.']
            >>> [s[start:end] for start, end in TreebankWordTokenizer().span_tokenize(s)] == expected
            True

        :param text: A string with a sentence or sentences.
        :type text: str
        :yield: Tuple[int, int]
        r   r   c                 ó6   — g | ]}|                      ¦   «         ‘ŒS © )Úgroup)Ú.0Úms     r/   ú
<listcomp>z7TreebankWordTokenizer.span_tokenize.<locals>.<listcomp>Ë   s    € ÐKÐKÐK Q�q—w’w‘y”yÐKÐKÐKr1   z
``|'{2}|\"c                 óF   •— g | ]}|d v r‰                      d¦  «        n|‘ŒS ))r   r   r   r   )Úpop)r6   ÚtokÚmatcheds     €r/   r8   z7TreebankWordTokenizer.span_tokenize.<locals>.<listcomp>Î   sB   ø€ ð ð ð àð #&Ð):Ð":Ð":�—’˜A‘”�Àðð ð r1   N)r0   ÚreÚfinditerr   )r,   r   Ú
raw_tokensÚtokensr<   s       @r/   Úspan_tokenizez#TreebankWordTokenizer.span_tokenize¬   s¥   øè è € ð. —]’] 4Ñ(Ô(ˆ
ð �4ˆKˆK˜T T˜\˜\àKÐK­"¬+°mÀTÑ*JÔ*JÐKÑKÔKˆGðð ð ð à%ðñ ô ˆFˆFð
  ˆFå ¨Ñ-Ô-Ð-Ð-Ð-Ð-Ð-Ð-Ð-Ð-Ð-r1   N)FF)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r=   Úcompiler"   r$   r%   r&   r'   r(   r   Ú_contractionsÚlistÚmapr)   r*   ÚstrÚboolr   r0   r   r   ÚintrA   r4   r1   r/   r
   r
      sÛ  € € € € € ðð ð0 
ˆŒ�FÑ	Ô	˜UÐ#Ø	ˆŒ�GÑ	Ô	˜gÐ&Ø	ˆŒÐ,Ñ	-Ô	-¨yÐ9ð€Oð 
ˆŒÐ$Ñ	%Ô	% yÐ1Ø	ˆŒ�JÑ	Ô	 Ð)Ø	ˆŒ�IÑ	Ô	 Ð)Ø	ˆŒ�KÑ	 Ô	  *Ð-àˆBŒJÐ7Ñ8Ô8Øð	
ð 
ˆŒ�GÑ	Ô	˜jÐ)Ø	ˆŒ�KÑ	 Ô	  (Ð+ð€Kð "�r”zÐ"7Ñ8Ô8¸*ÐE€Oð 
ˆŒ�EÑ	Ô	˜GÐ$Ø	ˆŒ�EÑ	Ô	˜GÐ$Ø	ˆŒ�EÑ	Ô	˜GÐ$Ø	ˆŒ�EÑ	Ô	˜GÐ$Ø	ˆŒ�EÑ	Ô	˜GÐ$Ø	ˆŒ�EÑ	Ô	˜GÐ$ðÐð  �R”Z Ñ&Ô&¨Ð0€Mð 
ˆŒ�EÑ	Ô	˜FÐ#Ø	ˆŒ�DÑ	Ô	˜6Ð"Ø	ˆŒÐ4Ñ	5Ô	5°yÐAØ	ˆŒÐ@Ñ	AÔ	AÀ9ÐMð	€Mð *Ð)Ñ+Ô+€MØ�D˜˜˜RœZ¨Ô)DÑEÔEÑFÔF€MØ�D˜˜˜RœZ¨Ô)DÑEÔEÑFÔF€Mð PUðEð EØðEØ.2ðEØHLðEà	ˆcŒðEð Eð Eð EðN). #ð ).¨(°5¸¸c¸´?Ô*Cð ).ð ).ð ).ð ).ð ).ð ).r1   r
   c            	       ó  — e Zd ZdZ e¦   «         Zd„ ej        D ¦   «         Zd„ ej        D ¦   «         Z ej	        d¦  «        df ej	        d¦  «        df ej	        d¦  «        df ej	        d	¦  «        df ej	        d
¦  «        dfgZ
 ej	        d¦  «        dfZ ej	        d¦  «        df ej	        d¦  «        df ej	        d¦  «        df ej	        d¦  «        df ej	        d¦  «        df ej	        d¦  «        dfgZ ej	        d¦  «        df ej	        d¦  «        df ej	        d¦  «        dfgZ ej	        d¦  «        df ej	        d ¦  «        df ej	        d!¦  «        d"f ej	        d#¦  «        df ej	        d$¦  «        df ej	        d%¦  «        d&f ej	        d'¦  «        d(fgZ ej	        d)¦  «        d*f ej	        d+¦  «        d(f ej	        d,¦  «        dfgZd4d.ee         d/ed0efd1„Zd4d.ee         d/ed0efd2„Zd3S )5ÚTreebankWordDetokenizera‹  
    The Treebank detokenizer uses the reverse regex operations corresponding to
    the Treebank tokenizer's regexes.

    Note:

    - There're additional assumption mades when undoing the padding of ``[;@#$%&]``
      punctuation symbols that isn't presupposed in the TreebankTokenizer.
    - There're additional regexes added in reversing the parentheses tokenization,
       such as the ``r'([\]\)\}\>])\s([:;,.])'``, which removes the additional right
       padding added to the closing parentheses precedding ``[:;,.]``.
    - It's not possible to return the original whitespaces as they were because
      there wasn't explicit records of where `'\n'`, `'\t'` or `'\s'` were removed at
      the text.split() operation.

    >>> from nltk.tokenize.treebank import TreebankWordTokenizer, TreebankWordDetokenizer
    >>> s = '''Good muffins cost $3.88\nin New York.  Please buy me\ntwo of them.\nThanks.'''
    >>> d = TreebankWordDetokenizer()
    >>> t = TreebankWordTokenizer()
    >>> toks = t.tokenize(s)
    >>> d.detokenize(toks)
    'Good muffins cost $3.88 in New York. Please buy me two of them. Thanks.'

    The MXPOST parentheses substitution can be undone using the ``convert_parentheses``
    parameter:

    >>> s = '''Good muffins cost $3.88\nin New (York).  Please (buy) me\ntwo of them.\n(Thanks).'''
    >>> expected_tokens = ['Good', 'muffins', 'cost', '$', '3.88', 'in',
    ... 'New', '-LRB-', 'York', '-RRB-', '.', 'Please', '-LRB-', 'buy',
    ... '-RRB-', 'me', 'two', 'of', 'them.', '-LRB-', 'Thanks', '-RRB-', '.']
    >>> expected_tokens == t.tokenize(s, convert_parentheses=True)
    True
    >>> expected_detoken = 'Good muffins cost $3.88 in New (York). Please (buy) me two of them. (Thanks).'
    >>> expected_detoken == d.detokenize(t.tokenize(s, convert_parentheses=True), convert_parentheses=True)
    True

    During tokenization it's safe to add more spaces but during detokenization,
    simply undoing the padding doesn't really help.

    - During tokenization, left and right pad is added to ``[!?]``, when
      detokenizing, only left shift the ``[!?]`` is needed.
      Thus ``(re.compile(r'\s([?!])'), r'\g<1>')``.

    - During tokenization ``[:,]`` are left and right padded but when detokenizing,
      only left shift is necessary and we keep right pad after comma/colon
      if the string after is a non-digit.
      Thus ``(re.compile(r'\s([:,])\s([^\d])'), r'\1 \2')``.

    >>> from nltk.tokenize.treebank import TreebankWordDetokenizer
    >>> toks = ['hello', ',', 'i', 'ca', "n't", 'feel', 'my', 'feet', '!', 'Help', '!', '!']
    >>> twd = TreebankWordDetokenizer()
    >>> twd.detokenize(toks)
    "hello, i can't feel my feet! Help!!"

    >>> toks = ['hello', ',', 'i', "can't", 'feel', ';', 'my', 'feet', '!',
    ... 'Help', '!', '!', 'He', 'said', ':', 'Help', ',', 'help', '?', '!']
    >>> twd.detokenize(toks)
    "hello, i can't feel; my feet! Help!! He said: Help, help?!"
    c                 ó^   — g | ]*}t          j        |                     d d¦  «        ¦  «        ‘Œ+S ©z(?#X)z\s©r=   rF   Úreplace©r6   Úpatterns     r/   r8   z"TreebankWordDetokenizer.<listcomp>  ó@   € ð ð ð àõ 	Œ
�7—?’? 7¨EÑ2Ô2Ñ3Ô3ðð ð r1   c                 ó^   — g | ]*}t          j        |                     d d¦  «        ¦  «        ‘Œ+S rP   rQ   rS   s     r/   r8   z"TreebankWordDetokenizer.<listcomp>  rU   r1   z+([^' ])\s('ll|'LL|'re|'RE|'ve|'VE|n't|N'T) z\1\2 z([^' ])\s('[sS]|'[mM]|'[dD]|') z(\S)\s(\'\')ú\1\2z(\'\')\s([.,:)\]>};%])r   r   r   r   r   ú(r   ú)r   ú[r   ú]r   ú{r   ú}z([\[\(\{\<])\sz\g<1>z\s([\]\)\}\>])z([\]\)\}\>])\s([:;,.])z([^'])\s'\sz\1' z\s([?!])z([^\.])\s(\.)([\]\)}>"\']*)\s*$z\1\2\3z([#$])\sz\s([;%])z
\s\.\.\.\sz...z\s([:,])z\1z([ (\[{<])\s``z\1``z(``)\sr   Fr@   r   r   c                 ó®  — d                      |¦  «        }d|z   dz   }| j        D ]}|                     d|¦  «        }Œ| j        D ]}|                     d|¦  «        }Œ| j        D ]\  }}|                     ||¦  «        }Œ|                     ¦   «         }| j        \  }}|                     ||¦  «        }|r#| j        D ]\  }}|                     ||¦  «        }Œ| j        D ]\  }}|                     ||¦  «        }Œ| j	        D ]\  }}|                     ||¦  «        }Œ| j
        D ]\  }}|                     ||¦  «        }Œ|                     ¦   «         S )a¥  
        Treebank detokenizer, created by undoing the regexes from
        the TreebankWordTokenizer.tokenize.

        :param tokens: A list of strings, i.e. tokenized text.
        :type tokens: List[str]
        :param convert_parentheses: if True, replace PTB symbols with parentheses,
            e.g. `-LRB-` to `(`. Defaults to False.
        :type convert_parentheses: bool, optional
        :return: str
        r   rW   )Újoinr*   r#   r)   r(   Ústripr'   r&   r%   r$   r"   )r,   r@   r   r   r-   r.   s         r/   r0   z TreebankWordDetokenizer.tokenize[  s   € ð �xŠx˜ÑÔˆð �T‰z˜CÑˆð Ô(ð 	-ð 	-ˆFØ—:’:˜g tÑ,Ô,ˆDˆDØÔ(ð 	-ð 	-ˆFØ—:’:˜g tÑ,Ô,ˆDˆDð %)Ô$6ð 	2ð 	2Ñ ˆF�LØ—:’:˜l¨DÑ1Ô1ˆDˆDð �zŠz‰|Œ|ˆð  $Ô1Ñˆ�Ø�zŠz˜,¨Ñ-Ô-ˆàð 	6Ø(,Ô(@ð 6ð 6Ñ$�˜Ø—z’z ,°Ñ5Ô5��ð %)Ô$8ð 	2ð 	2Ñ ˆF�LØ—:’:˜l¨DÑ1Ô1ˆDˆDð %)Ô$4ð 	2ð 	2Ñ ˆF�LØ—:’:˜l¨DÑ1Ô1ˆDˆDð %)Ô$8ð 	2ð 	2Ñ ˆF�LØ—:’:˜l¨DÑ1Ô1ˆDˆDà�zŠz‰|Œ|Ðr1   c                 ó.   — |                       ||¦  «        S )z&Duck-typing the abstract *tokenize()*.)r0   )r,   r@   r   s      r/   Ú
detokenizez"TreebankWordDetokenizer.detokenize�  s   € à�}Š}˜VÐ%8Ñ9Ô9Ð9r1   N)F)rB   rC   rD   rE   r   rG   r)   r*   r=   rF   r(   r'   r&   r%   r$   r"   r   rJ   rK   r0   rb   r4   r1   r/   rN   rN   Ø   s  € € € € € ð:ð :ðx *Ð)Ñ+Ô+€Mðð à$Ô2ðñ ô €Mðð à$Ô2ðñ ô €Mð 
ˆŒÐBÑ	CÔ	CÀXÐNØ	ˆŒÐ6Ñ	7Ô	7¸ÐBØ	ˆŒ�OÑ	$Ô	$ gÐ.àˆBŒJÐ0Ñ1Ô1Øð	
ð 
ˆŒ�EÑ	Ô	˜CÐ ð	€Mð  �R”Z Ñ(Ô(¨%Ð0€Mð 
ˆŒ�GÑ	Ô	˜cÐ"Ø	ˆŒ�GÑ	Ô	˜cÐ"Ø	ˆŒ�GÑ	Ô	˜cÐ"Ø	ˆŒ�GÑ	Ô	˜cÐ"Ø	ˆŒ�GÑ	Ô	˜cÐ"Ø	ˆŒ�GÑ	Ô	˜cÐ"ðÐð 
ˆŒÐ%Ñ	&Ô	&¨Ð1Ø	ˆŒÐ%Ñ	&Ô	&¨Ð1Ø	ˆŒÐ-Ñ	.Ô	.°Ð8ð€Oð 
ˆŒ�NÑ	#Ô	# WÐ-Ø	ˆŒ�KÑ	 Ô	  (Ð+à	ˆŒÐ6Ñ	7Ô	7¸ÐCð
 
ˆŒ�KÑ	 Ô	  (Ð+Ø	ˆŒ�KÑ	 Ô	  (Ð+à	ˆŒ�MÑ	"Ô	" FÐ+ð ˆBŒJ�{Ñ#Ô#Øð	
ð€Kð, 
ˆŒÐ%Ñ	&Ô	&¨Ð0Ø	ˆŒ�IÑ	Ô	 Ð&Ø	ˆŒ�EÑ	Ô	˜DÐ!ð€Oð3ð 3˜t Cœyð 3¸tð 3ÐPSð 3ð 3ð 3ð 3ðj:ð :  c¤ð :Àð :ÐRUð :ð :ð :ð :ð :ð :r1   rN   )rE   r=   r   Útypingr   r   r   Únltk.tokenize.apir   Únltk.tokenize.destructiver   Únltk.tokenize.utilr   r
   rN   r4   r1   r/   ú<module>rg      sè   ððð ð 
€	€	€	Ø €€€Ø (Ð (Ð (Ð (Ð (Ð (Ð (Ð (Ð (Ð (à (Ð (Ð (Ð (Ð (Ð (Ø ;Ð ;Ð ;Ð ;Ð ;Ð ;Ø +Ð +Ð +Ð +Ð +Ð +ðx.ð x.ð x.ð x.ð x.˜Jñ x.ô x.ð x.ðvz:ð z:ð z:ð z:ð z:˜jñ z:ô z:ð z:ð z:ð z:r1   