§
    'ê[fU  ã                   ó>   — d Z ddlmZ ddlmZ  G d„ de¦  «        ZdS )a(  
Multi-Word Expression Tokenizer

A ``MWETokenizer`` takes a string which has already been divided into tokens and
retokenizes it, merging multi-word expressions into single tokens, using a lexicon
of MWEs:


    >>> from nltk.tokenize import MWETokenizer

    >>> tokenizer = MWETokenizer([('a', 'little'), ('a', 'little', 'bit'), ('a', 'lot')])
    >>> tokenizer.add_mwe(('in', 'spite', 'of'))

    >>> tokenizer.tokenize('Testing testing testing one two three'.split())
    ['Testing', 'testing', 'testing', 'one', 'two', 'three']

    >>> tokenizer.tokenize('This is a test in spite'.split())
    ['This', 'is', 'a', 'test', 'in', 'spite']

    >>> tokenizer.tokenize('In a little or a little bit or a lot in spite of'.split())
    ['In', 'a_little', 'or', 'a_little_bit', 'or', 'a_lot', 'in_spite_of']

é    )Ú
TokenizerI)ÚTriec                   ó&   — e Zd ZdZdd„Zd„ Zd„ ZdS )ÚMWETokenizerzhA tokenizer that processes tokenized text and merges multi-word expressions
    into single tokens.
    NÚ_c                 óD   — |sg }t          |¦  «        | _        || _        dS )a¥  Initialize the multi-word tokenizer with a list of expressions and a
        separator

        :type mwes: list(list(str))
        :param mwes: A sequence of multi-word expressions to be merged, where
            each MWE is a sequence of strings.
        :type separator: str
        :param separator: String that should be inserted between words in a multi-word
            expression token. (Default is '_')

        N)r   Ú_mwesÚ
_separator)ÚselfÚmwesÚ	separators      úE/var/www/piapp/venv/lib/python3.11/site-packages/nltk/tokenize/mwe.pyÚ__init__zMWETokenizer.__init__(   s)   € ð ð 	ØˆDÝ˜$‘Z”ZˆŒ
Ø#ˆŒˆˆó    c                 ó:   — | j                              |¦  «         dS )a–  Add a multi-word expression to the lexicon (stored as a word trie)

        We use ``util.Trie`` to represent the trie. Its form is a dict of dicts.
        The key True marks the end of a valid MWE.

        :param mwe: The multi-word expression we're adding into the word trie
        :type mwe: tuple(str) or list(str)

        :Example:

        >>> tokenizer = MWETokenizer()
        >>> tokenizer.add_mwe(('a', 'b'))
        >>> tokenizer.add_mwe(('a', 'b', 'c'))
        >>> tokenizer.add_mwe(('a', 'x'))
        >>> expected = {'a': {'x': {True: None}, 'b': {True: None, 'c': {True: None}}}}
        >>> tokenizer._mwes == expected
        True

        N)r	   Úinsert)r   Úmwes     r   Úadd_mwezMWETokenizer.add_mwe9   s    € ð( 	Œ
×Ò˜#ÑÔÐÐÐr   c                 ó(  — d}t          |¦  «        }g }||k     rø||         | j        v rÃ|}| j        }d}||k     r=||         |v r3|||                  }|dz   }t          j        |v r|}||k     r
||         |v °3|dk    r|}t          j        |v s|dk    r8|                     | j                             |||…         ¦  «        ¦  «         |}nA|                     ||         ¦  «         |dz  }n |                     ||         ¦  «         |dz  }||k     °ø|S )a¥  

        :param text: A list containing tokenized text
        :type text: list(str)
        :return: A list of the tokenized text with multi-words merged together
        :rtype: list(str)

        :Example:

        >>> tokenizer = MWETokenizer([('hors', "d'oeuvre")], separator='+')
        >>> tokenizer.tokenize("An hors d'oeuvre tonight, sir?".split())
        ['An', "hors+d'oeuvre", 'tonight,', 'sir?']

        r   éÿÿÿÿé   )Úlenr	   r   ÚLEAFÚappendr
   Újoin)r   ÚtextÚiÚnÚresultÚjÚtrieÚ
last_matchs           r   ÚtokenizezMWETokenizer.tokenizeO   sB  € ð ˆÝ�‰IŒIˆØˆà�!ŠeˆeØ�AŒw˜$œ*Ð$Ð$à�Ø”z�Ø�
Ø˜!’e�e  Q¤¨4  Ø  Q¤œ=�DØ˜A™�AÝ”y DÐ(Ð(Ø%&˜
ð	 ˜!’e�e  Q¤¨4  ð " B’�Ø&˜å”y DÐ(Ð(¨J¸ªO¨OàŸš d¤o×&:Ò&:¸4ÀÀ!À¼9Ñ&EÔ&EÑFÔFÐFØ˜˜ð Ÿš d¨1¤gÑ.Ô.Ð.Ø˜Q™˜˜à—’˜d 1œgÑ&Ô&Ð&Ø�Q‘�ð3 �!Šeˆeð4 ˆr   )Nr   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r#   © r   r   r   r   #   sP   € € € € € ðð ð$ð $ð $ð $ð"ð ð ð,-ð -ð -ð -ð -r   r   N)r'   Únltk.tokenize.apir   Ú	nltk.utilr   r   r(   r   r   ú<module>r+      ss   ððð ð. )Ð (Ð (Ð (Ð (Ð (Ø Ð Ð Ð Ð Ð ðYð Yð Yð Yð Y�:ñ Yô Yð Yð Yð Yr   