§
    'ê[f   ã                   ób   — d Z ddlZddlZddlmZ ddlmZmZmZ ddl	m
Z
  G d„ de¦  «        ZdS )z{
A reader for corpora that consist of Tweets. It is assumed that the Tweets
have been serialised into line-delimited JSON.
é    N)ÚCorpusReader)ÚStreamBackedCorpusViewÚZipFilePathPointerÚconcat)ÚTweetTokenizerc                   óT   — e Zd ZdZeZ	 d e¦   «         dfd„Zd	d„Zd	d„Z	d	d„Z
d„ ZdS )
ÚTwitterCorpusReadera7  
    Reader for corpora that consist of Tweets represented as a list of line-delimited JSON.

    Individual Tweets can be tokenized using the default tokenizer, or by a
    custom tokenizer specified as a parameter to the constructor.

    Construct a new Tweet corpus reader for a set of documents
    located at the given root directory.

    If you made your own tweet collection in a directory called
    `twitter-files`, then you can initialise the reader as::

        from nltk.corpus import TwitterCorpusReader
        reader = TwitterCorpusReader(root='/path/to/twitter-files', '.*\.json')

    However, the recommended approach is to set the relevant directory as the
    value of the environmental variable `TWITTER`, and then invoke the reader
    as follows::

       root = os.environ['TWITTER']
       reader = TwitterCorpusReader(root, '.*\.json')

    If you want to work directly with the raw Tweets, the `json` library can
    be used::

       import json
       for tweet in reader.docs():
           print(json.dumps(tweet, indent=1, sort_keys=True))

    NÚutf8c                 ó  — t          j        | |||¦  «         |                      | j        ¦  «        D ]N}t	          |t
          ¦  «        rŒt          j                             |¦  «        dk    rt          d|› d�¦  «        ‚ŒO	 || _
        dS )a  
        :param root: The root directory for this corpus.
        :param fileids: A list or regexp specifying the fileids in this corpus.
        :param word_tokenizer: Tokenizer for breaking the text of Tweets into
            smaller units, including but not limited to words.
        r   zFile z	 is emptyN)r   Ú__init__ÚabspathsÚ_fileidsÚ
isinstancer   ÚosÚpathÚgetsizeÚ
ValueErrorÚ_word_tokenizer)ÚselfÚrootÚfileidsÚword_tokenizerÚencodingr   s         úN/var/www/piapp/venv/lib/python3.11/site-packages/nltk/corpus/reader/twitter.pyr   zTwitterCorpusReader.__init__:   s—   € õ 	Ô˜d D¨'°8Ñ<Ô<Ð<à—M’M $¤-Ñ0Ô0ð 	:ð 	:ˆDÝ˜$Õ 2Ñ3Ô3ð :ØÝ”—’ Ñ&Ô&¨!Ò+Ð+Ý Ð!8¨Ð!8Ð!8Ð!8Ñ9Ô9Ð9ð ,àEà-ˆÔÐÐó    c                 ód   ‡ — t          ˆ fd„‰                      |dd¦  «        D ¦   «         ¦  «        S )a(  
        Returns the full Tweet objects, as specified by `Twitter
        documentation on Tweets
        <https://dev.twitter.com/docs/platform-objects/tweets>`_

        :return: the given file(s) as a list of dictionaries deserialised
            from JSON.
        :rtype: list(dict)
        c                 óR   •— g | ]#\  }}}‰                      |‰j        |¬ ¦  «        ‘Œ$S ))r   )Ú
CorpusViewÚ_read_tweets)Ú.0r   ÚencÚfileidr   s       €r   ú
<listcomp>z,TwitterCorpusReader.docs.<locals>.<listcomp>Y   sD   ø€ ð ð ð á'�T˜3 ð —’  dÔ&7À#�ÑFÔFðð ð r   T)r   r   )r   r   s   ` r   ÚdocszTwitterCorpusReader.docsN   sM   ø€ õ ðð ð ð à+/¯=ª=¸À$ÈÑ+MÔ+Mðñ ô ñ
ô 
ð 	
r   c                 óø   — |                       |¦  «        }g }|D ]_}	 |d         }t          |t          ¦  «        r|                     | j        ¦  «        }|                     |¦  «         ŒP# t          $ r Y Œ\w xY w|S )z›
        Returns only the text content of Tweets in the file(s)

        :return: the given file(s) as a list of Tweets.
        :rtype: list(str)
        Útext)r$   r   ÚbytesÚdecoder   ÚappendÚKeyError)r   r   Ú
fulltweetsÚtweetsÚjsonor&   s         r   ÚstringszTwitterCorpusReader.strings_   s–   € ð —Y’Y˜wÑ'Ô'ˆ
ØˆØð 	ð 	ˆEðØ˜V”}�Ý˜d¥EÑ*Ô*ð 6ØŸ;š; t¤}Ñ5Ô5�DØ—’˜dÑ#Ô#Ð#Ð#øÝð ð ð Ø�ðøøøàˆs   �AA*Á*
A7Á6A7c                 óX   ‡— |                       |¦  «        }| j        Šˆfd„|D ¦   «         S )zÎ
        :return: the given file(s) as a list of the text content of Tweets as
            as a list of words, screenanames, hashtags, URLs and punctuation symbols.

        :rtype: list(list(str))
        c                 ó:   •— g | ]}‰                      |¦  «        ‘ŒS © )Útokenize)r    ÚtÚ	tokenizers     €r   r#   z1TwitterCorpusReader.tokenized.<locals>.<listcomp>{   s'   ø€ Ð6Ð6Ð6¨!�	×"Ò" 1Ñ%Ô%Ð6Ð6Ð6r   )r.   r   )r   r   r,   r4   s      @r   Ú	tokenizedzTwitterCorpusReader.tokenizedr   s8   ø€ ð —’˜gÑ&Ô&ˆØÔ(ˆ	Ø6Ð6Ð6Ð6¨vÐ6Ñ6Ô6Ð6r   c                 ó´   — g }t          d¦  «        D ]E}|                     ¦   «         }|s|c S t          j        |¦  «        }|                     |¦  «         ŒF|S )zS
        Assumes that each line in ``stream`` is a JSON-serialised object.
        é
   )ÚrangeÚreadlineÚjsonÚloadsr)   )r   Ústreamr,   ÚiÚlineÚtweets         r   r   z TwitterCorpusReader._read_tweets}   sg   € ð ˆÝ�r‘”ð 	!ð 	!ˆAØ—?’?Ñ$Ô$ˆDØð Ø���Ý”J˜tÑ$Ô$ˆEØ�MŠM˜%Ñ Ô Ð Ð Øˆr   )N)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r   r   r$   r.   r5   r   r1   r   r   r	   r	      sš   € € € € € ðð ð> (€Jðð
 !°°Ñ1AÔ1AÈFð.ð .ð .ð .ð(
ð 
ð 
ð 
ð"ð ð ð ð&	7ð 	7ð 	7ð 	7ðð ð ð ð r   r	   )rC   r:   r   Únltk.corpus.reader.apir   Únltk.corpus.reader.utilr   r   r   Únltk.tokenizer   r	   r1   r   r   ú<module>rG      s£   ððð ð
 €€€Ø 	€	€	€	à /Ð /Ð /Ð /Ð /Ð /Ø VÐ VÐ VÐ VÐ VÐ VÐ VÐ VÐ VÐ VØ (Ð (Ð (Ð (Ð (Ð (ðsð sð sð sð s˜,ñ sô sð sð sð sr   