§
    'ê[få>  ã            
       óx  — d Z ddlZddlmZ ddlZddlmZ dZdZdZ	dZ
eed	d
dddde	df
Zed         e
gedd…         ¢R Z ej        d¦  «        Z ej        eej        ej        z  ej        z  ¦  «        Z ej        d¦  «        Z ej        d¦  «        Zd d„Zd!d„Z G d„ de¦  «        Zd„ Zd„ Z	 	 	 	 d"d„ZdS )#a�  
Twitter-aware tokenizer, designed to be flexible and easy to adapt to new
domains and tasks. The basic logic is this:

1. The tuple REGEXPS defines a list of regular expression
   strings.

2. The REGEXPS strings are put, in order, into a compiled
   regular expression object called WORD_RE, under the TweetTokenizer
   class.

3. The tokenization is done by WORD_RE.findall(s), where s is the
   user-supplied string, inside the tokenize() method of the class
   TweetTokenizer.

4. When instantiating Tokenizer objects, there are several options:
    * preserve_case. By default, it is set to True. If it is set to
      False, then the tokenizer will downcase everything except for
      emoticons.
    * reduce_len. By default, it is set to False. It specifies whether
      to replace repeated character sequences of length 3 or greater
      with sequences of length 3.
    * strip_handles. By default, it is set to False. It specifies
      whether to remove Twitter handles of text used in the
      `tokenize` method.
    * match_phone_numbers. By default, it is set to True. It indicates
      whether the `tokenize` method should look for phone numbers.
é    N)ÚList)Ú
TokenizerIac  
    (?:
      [<>]?
      [:;=8]                     # eyes
      [\-o\*\']?                 # optional nose
      [\)\]\(\[dDpP/\:\}\{@\|\\] # mouth
      |
      [\)\]\(\[dDpP/\:\}\{@\|\\] # mouth
      [\-o\*\']?                 # optional nose
      [:;=8]                     # eyes
      [<>]?
      |
      </?3                       # heart
    )u  			# Capture 1: entire matched URL
  (?:
  https?:				# URL protocol and colon
    (?:
      /{1,3}				# 1-3 slashes
      |					#   or
      [a-z0-9%]				# Single letter or digit or '%'
                                       # (Trying not to match e.g. "URI::Escape")
    )
    |					#   or
                                       # looks like domain name followed by a slash:
    [a-z0-9.\-]+[.]
    (?:[a-z]{2,13})
    /
  )
  (?:					# One or more:
    [^\s()<>{}\[\]]+			# Run of non-space, non-()<>{}[]
    |					#   or
    \([^\s()]*?\([^\s()]+\)[^\s()]*?\) # balanced parens, one level deep: (...(...)...)
    |
    \([^\s]+?\)				# balanced parens, non-recursive: (...)
  )+
  (?:					# End with:
    \([^\s()]*?\([^\s()]+\)[^\s()]*?\) # balanced parens, one level deep: (...(...)...)
    |
    \([^\s]+?\)				# balanced parens, non-recursive: (...)
    |					#   or
    [^\s`!()\[\]{};:'".,<>?Â«Â»â€œâ€�â€˜â€™]	# not a space or one of these punct chars
  )
  |					# OR, the following to match naked domains:
  (?:
  	(?<!@)			        # not preceded by a @, avoid matching foo@_gmail.com_
    [a-z0-9]+
    (?:[.\-][a-z0-9]+)*
    [.]
    (?:[a-z]{2,13})
    \b
    /?
    (?!@)			        # not succeeded by a @,
                            # avoid matching "foo.na" in "foo.na@example.com"
  )
uÏ  
  (?:
    [\U0001F1E6-\U0001F1FF]{2}  # all enclosed letter pairs
    |
    # English flag
    \U0001F3F4\U000E0067\U000E0062\U000E0065\U000E006e\U000E0067\U000E007F
    |
    # Scottish flag
    \U0001F3F4\U000E0067\U000E0062\U000E0073\U000E0063\U000E0074\U000E007F
    |
    # For Wales? Why Richard, it profit a man nothing to give his soul for the whole world â€¦ but for Wales!
    \U0001F3F4\U000E0067\U000E0062\U000E0077\U000E006C\U000E0073\U000E007F
  )
a	  
    (?:
      (?:            # (international)
        \+?[01]
        [ *\-.\)]*
      )?
      (?:            # (area code)
        [\(]?
        \d{3}
        [ *\-.\)]*
      )?
      \d{3}          # exchange
      [ *\-.\)]*
      \d{4}          # base
    )z	<[^>\s]+>z[\-]+>|<[\-]+z(?:@[\w_]+)z(?:\#+[\w_]+[\w\'_\-]*[\w_]+)z#[\w.+-]+@[\w-]+\.(?:[\w-]\.?)+[\w-]uR   .(?:
        [ðŸ�»-ðŸ�¿]?(?:â€�.[ðŸ�»-ðŸ�¿]?)+
        |
        [ðŸ�»-ðŸ�¿]
    )a…  
    (?:[^\W\d_](?:[^\W\d_]|['\-_])+[^\W\d_]) # Words with apostrophes or dashes.
    |
    (?:[+\-]?\d+[,/.:-]\d+[+\-]?)  # Numbers, including fractions, decimals.
    |
    (?:[\w_]+)                     # Words without apostrophes or dashes.
    |
    (?:\.(?:\s*\.){1,})            # Ellipsis dots.
    |
    (?:\S)                         # Everything else that isn't whitespace.
    é   z([^a-zA-Z0-9])\1{3,}z&(#?(x?))([^&;\s]+);zZ(?<![A-Za-z0-9_!@#\$%&*])@(([A-Za-z0-9_]){15}(?!@)|([A-Za-z0-9_]){1,14}(?![A-Za-z0-9_]*@))Ústrictc                 ód   — |€d}t          | t          ¦  «        r|                      ||¦  «        S | S )Núutf-8)Ú
isinstanceÚbytesÚdecode)ÚtextÚencodingÚerrorss      úH/var/www/piapp/venv/lib/python3.11/site-packages/nltk/tokenize/casual.pyÚ_str_to_unicoder   ì   s8   € ØÐØˆÝ�$�ÑÔð -Ø�{Š{˜8 VÑ,Ô,Ð,Ø€Kó    © Tr   c                 ód   ‡‡— ˆˆfd„}t                                |t          | |¦  «        ¦  «        S )u·  
    Remove entities from text by converting them to their
    corresponding unicode character.

    :param text: a unicode string or a byte string encoded in the given
    `encoding` (which defaults to 'utf-8').

    :param list keep:  list of entity names which should not be replaced.    This supports both numeric entities (``&#nnnn;`` and ``&#hhhh;``)
    and named entities (such as ``&nbsp;`` or ``&gt;``).

    :param bool remove_illegal: If `True`, entities that can't be converted are    removed. Otherwise, entities that can't be converted are kept "as
    is".

    :returns: A unicode string with the entities removed.

    See https://github.com/scrapy/w3lib/blob/master/w3lib/html.py

        >>> from nltk.tokenize.casual import _replace_html_entities
        >>> _replace_html_entities(b'Price: &pound;100')
        'Price: \xa3100'
        >>> print(_replace_html_entities(b'Price: &pound;100'))
        Price: Â£100
        >>>
    c                 óP  •— |                       d¦  «        }|                       d¦  «        r}	 |                       d¦  «        rt          |d¦  «        }nt          |d¦  «        }d|cxk    rdk    r&n n#t          |f¦  «                             d¦  «        S nO# t          $ r d }Y nAw xY w|‰v r|                       d	¦  «        S t
          j        j                             |¦  «        }|�'	 t          |¦  «        S # t          t          f$ r Y nw xY w‰rd
n|                       d	¦  «        S )Né   r   é   é   é
   é€   éŸ   Úcp1252r   Ú )ÚgroupÚintr
   r   Ú
ValueErrorÚhtmlÚentitiesÚname2codepointÚgetÚchrÚOverflowError)ÚmatchÚentity_bodyÚnumberÚkeepÚremove_illegals      €€r   Ú_convert_entityz/_replace_html_entities.<locals>._convert_entity  sO  ø€ Ø—k’k !‘n”nˆØ�;Š;�q‰>Œ>ð 	CðØ—;’;˜q‘>”>ð 2Ý  ¨bÑ1Ô1�F�Få  ¨bÑ1Ô1�Fð
 ˜6Ð)Ð)Ò)Ð) TÒ)Ð)Ð)Ð)Ð)Ý  & Ñ+Ô+×2Ò2°8Ñ<Ô<Ð<øøÝð ð ð Ø���ðøøøð ˜dÐ"Ð"Ø—{’{ 1‘~”~Ð%Ý”]Ô1×5Ò5°kÑBÔBˆFØÐðÝ˜6‘{”{Ð"øÝ¥Ð.ð ð ð Ø�ðøøøð $Ð7ˆrˆr¨¯ª°Q©¬Ð7s$   ­A(B ÂB&Â%B&Ã)C8 Ã8DÄD)ÚENT_REÚsubr   )r   r)   r*   r   r+   s    ``  r   Ú_replace_html_entitiesr.   ô   sB   øø€ ð88ð 8ð 8ð 8ð 8ð 8õ8 �:Š:�o¥°t¸XÑ'FÔ'FÑGÔGÐGr   c                   óv   — e Zd ZdZdZdZ	 	 	 	 dd„Zdedee         fd„Z	e
dd
„¦   «         Ze
dd„¦   «         ZdS )ÚTweetTokenizeraÚ  
    Tokenizer for tweets.

        >>> from nltk.tokenize import TweetTokenizer
        >>> tknzr = TweetTokenizer()
        >>> s0 = "This is a cooool #dummysmiley: :-) :-P <3 and some arrows < > -> <--"
        >>> tknzr.tokenize(s0) # doctest: +NORMALIZE_WHITESPACE
        ['This', 'is', 'a', 'cooool', '#dummysmiley', ':', ':-)', ':-P', '<3', 'and', 'some', 'arrows', '<', '>', '->',
         '<--']

    Examples using `strip_handles` and `reduce_len parameters`:

        >>> tknzr = TweetTokenizer(strip_handles=True, reduce_len=True)
        >>> s1 = '@remy: This is waaaaayyyy too much for you!!!!!!'
        >>> tknzr.tokenize(s1)
        [':', 'This', 'is', 'waaayyy', 'too', 'much', 'for', 'you', '!', '!', '!']
    NTFc                 ó>   — || _         || _        || _        || _        dS )ae  
        Create a `TweetTokenizer` instance with settings for use in the `tokenize` method.

        :param preserve_case: Flag indicating whether to preserve the casing (capitalisation)
            of text used in the `tokenize` method. Defaults to True.
        :type preserve_case: bool
        :param reduce_len: Flag indicating whether to replace repeated character sequences
            of length 3 or greater with sequences of length 3. Defaults to False.
        :type reduce_len: bool
        :param strip_handles: Flag indicating whether to remove Twitter handles of text used
            in the `tokenize` method. Defaults to False.
        :type strip_handles: bool
        :param match_phone_numbers: Flag indicating whether the `tokenize` method should look
            for phone numbers. Defaults to True.
        :type match_phone_numbers: bool
        N©Úpreserve_caseÚ
reduce_lenÚstrip_handlesÚmatch_phone_numbers)Úselfr3   r4   r5   r6   s        r   Ú__init__zTweetTokenizer.__init__L  s)   € ð. +ˆÔØ$ˆŒØ*ˆÔØ#6ˆÔ Ð Ð r   r   Úreturnc                 ót  — t          |¦  «        }| j        rt          |¦  «        }| j        rt	          |¦  «        }t
                               d|¦  «        }| j        r| j         	                    |¦  «        }n| j
         	                    |¦  «        }| j        st          t          d„ |¦  «        ¦  «        }|S )zÒTokenize the input text.

        :param text: str
        :rtype: list(str)
        :return: a tokenized list of strings; joining this list returns        the original string if `preserve_case=False`.
        ú\1\1\1c                 ób   — t                                | ¦  «        r| n|                      ¦   «         S )N)ÚEMOTICON_REÚsearchÚlower)Úxs    r   ú<lambda>z)TweetTokenizer.tokenize.<locals>.<lambda>‚  s%   € ¥K×$6Ò$6°qÑ$9Ô$9ÐH˜q˜q¸q¿wºw¹y¼y€ r   )r.   r5   Úremove_handlesr4   Úreduce_lengtheningÚHANG_REr-   r6   ÚPHONE_WORD_REÚfindallÚWORD_REr3   ÚlistÚmap)r7   r   Ú	safe_textÚwordss       r   ÚtokenizezTweetTokenizer.tokenizeh  s½   € õ & dÑ+Ô+ˆàÔð 	(Ý! $Ñ'Ô'ˆDàŒ?ð 	,Ý% dÑ+Ô+ˆDå—K’K 	¨4Ñ0Ô0ˆ	àÔ#ð 	4ØÔ&×.Ò.¨yÑ9Ô9ˆEˆEà”L×(Ò(¨Ñ3Ô3ˆEàÔ!ð 	ÝÝÐHÐHÈ5ÑQÔQñô ˆEð ˆr   úregex.Patternc                 ó   — t          | ¦  «        j        sgt          j        dd                     t
          ¦  «        › d�t          j        t          j        z  t          j        z  ¦  «        t          | ¦  «        _        t          | ¦  «        j        S )zCore TweetTokenizer regexú(Ú|ú))	ÚtypeÚ_WORD_REÚregexÚcompileÚjoinÚREGEXPSÚVERBOSEÚIÚUNICODE©r7   s    r   rG   zTweetTokenizer.WORD_RE†  sm   € õ �D‰zŒzÔ"ð 	Ý"'¤-Ø(�C—H’H�WÑ%Ô%Ð(Ð(Ð(Ý”¥¤Ñ'­%¬-Ñ7ñ#ô #�D�‰JŒJÔõ �D‰zŒzÔ"Ð"r   c                 ó   — t          | ¦  «        j        sgt          j        dd                     t
          ¦  «        › d�t          j        t          j        z  t          j        z  ¦  «        t          | ¦  «        _        t          | ¦  «        j        S )z#Secondary core TweetTokenizer regexrO   rP   rQ   )	rR   Ú_PHONE_WORD_RErT   rU   rV   ÚREGEXPS_PHONErX   rY   rZ   r[   s    r   rE   zTweetTokenizer.PHONE_WORD_RE‘  sm   € õ �D‰zŒzÔ(ð 	Ý(-¬Ø.�C—H’H�]Ñ+Ô+Ð.Ð.Ð.Ý”¥¤Ñ'­%¬-Ñ7ñ)ô )�D�‰JŒJÔ%õ �D‰zŒzÔ(Ð(r   ©TFFT)r9   rM   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__rS   r]   r8   Ústrr   rL   ÚpropertyrG   rE   r   r   r   r0   r0   2  s²   € € € € € ðð ð( €HØ€Nð ØØØ ð7ð 7ð 7ð 7ð8˜Sð  T¨#¤Yð ð ð ð ð< ð#ð #ð #ñ „Xð#ð ð)ð )ð )ñ „Xð)ð )ð )r   r0   c                 óV   — t          j        d¦  «        }|                     d| ¦  «        S )ze
    Replace repeated character sequences of length 3 or greater with sequences
    of length 3.
    z	(.)\1{2,}r;   )rT   rU   r-   )r   Úpatterns     r   rC   rC   ¢  s'   € õ
 Œm˜LÑ)Ô)€GØ�;Š;�y $Ñ'Ô'Ð'r   c                 ó8   — t                                d| ¦  «        S )z4
    Remove Twitter username handles from text.
    Ú )Ú
HANDLES_REr-   )r   s    r   rB   rB   «  s   € õ
 �>Š>˜#˜tÑ$Ô$Ð$r   Fc                 óN   — t          ||||¬¦  «                             | ¦  «        S )z:
    Convenience function for wrapping the tokenizer.
    r2   )r0   rL   )r   r3   r4   r5   r6   s        r   Úcasual_tokenizerl   ¸  s4   € õ Ø#ØØ#Ø/ð	ñ ô ÷
 ‚hˆt�n„nðr   )Nr   )r   Tr   r_   )rc   r    Útypingr   rT   Únltk.tokenize.apir   Ú	EMOTICONSÚURLSÚFLAGSÚPHONE_REGEXrW   r^   rU   rD   rX   rY   rZ   r=   r,   rj   r   r.   r0   rC   rB   rl   r   r   r   ú<module>rs      sÅ  ððð ð@ €€€Ø Ð Ð Ð Ð Ð à €€€à (Ð (Ð (Ð (Ð (Ð (ð$	€	ð$)€ðf	€ð 	€ð$ 	ààààà(à.ð	ð 
ð
ð/"€ðJ ˜”˜[Ð7¨7°1°2°2¬;Ð7Ð7€ð ˆ%Œ-Ð/Ñ
0Ô
0€ð ˆeŒm˜I u¤}°u´wÑ'>ÀÄÑ'NÑOÔO€ð 
ˆŒÐ.Ñ	/Ô	/€ð ˆUŒ]ðHñô €
ðð ð ð ð8Hð 8Hð 8Hð 8Hð|h)ð h)ð h)ð h)ð h)�Zñ h)ô h)ð h)ð`(ð (ð (ð%ð %ð %ð ØØØðð ð ð ð ð r   