§
    'ê[f/!  ã                   ó¶   — d Z ddlZddlZddlmZmZ ddlmZ ddlmZ ddl	m
Z
 ddlmZ ddlmZ  G d	„ d
e¬¦  «        Zd„ Zd„ Zdd„Z G d„ de¬¦  «        ZdS )zLanguage Model Interface.é    N)ÚABCMetaÚabstractmethod)Úbisect)Ú
accumulate)ÚNgramCounter)Ú	log_base2)Ú
Vocabularyc                   óD   — e Zd ZdZd„ Zed„ ¦   «         Zed„ ¦   «         ZdS )Ú	SmoothingzìNgram Smoothing Interface

    Implements Chen & Goodman 1995's idea that all smoothing algorithms have
    certain features in common. This should ideally allow smoothing algorithms to
    work both with Backoff and Interpolation.
    c                 ó"   — || _         || _        dS )zä
        :param vocabulary: The Ngram vocabulary object.
        :type vocabulary: nltk.lm.vocab.Vocabulary
        :param counter: The counts of the vocabulary items.
        :type counter: nltk.lm.counter.NgramCounter
        N)ÚvocabÚcounts)ÚselfÚ
vocabularyÚcounters      ú?/var/www/piapp/venv/lib/python3.11/site-packages/nltk/lm/api.pyÚ__init__zSmoothing.__init__   s   € ð  ˆŒ
ØˆŒˆˆó    c                 ó   — t          ¦   «         ‚©N©ÚNotImplementedError)r   Úwords     r   Úunigram_scorezSmoothing.unigram_score&   ó   € å!Ñ#Ô#Ð#r   c                 ó   — t          ¦   «         ‚r   r   ©r   r   Úcontexts      r   Úalpha_gammazSmoothing.alpha_gamma*   r   r   N)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r   r   © r   r   r   r      sc   € € € € € ðð ðð ð ð ð$ð $ñ „^ð$ð ð$ð $ñ „^ð$ð $ð $r   r   )Ú	metaclassc                 ó@   — t          | ¦  «        t          | ¦  «        z  S )z0Return average (aka mean) for sequence of items.)ÚsumÚlen)Úitemss    r   Ú_meanr*   /   s   € åˆu‰:Œ:�˜E™
œ
Ñ"Ð"r   c                 ób   — t          | t          j        ¦  «        r| S t          j        | ¦  «        S r   )Ú
isinstanceÚrandomÚRandom)Úseed_or_generators    r   Ú_random_generatorr0   4   s.   € ÝÐ#¥V¤]Ñ3Ô3ð !Ø Ð ÝŒ=Ð*Ñ+Ô+Ð+r   c                 ó$  — | st          d¦  «        ‚t          | ¦  «        t          |¦  «        k    rt          d¦  «        ‚t          t          |¦  «        ¦  «        }|d         }|                     ¦   «         }| t          |||z  ¦  «                 S )z`Like random.choice, but with weights.

    Heavily inspired by python 3.6 `random.choices`.
    z"Can't choose from empty populationz3The number of weights does not match the populationéÿÿÿÿ)Ú
ValueErrorr(   Úlistr   r-   r   )Ú
populationÚweightsÚrandom_generatorÚcum_weightsÚtotalÚ	thresholds         r   Ú_weighted_choicer;   :   sŠ   € ð
 ð ?ÝÐ=Ñ>Ô>Ð>Ý
ˆ:�„�#˜g™,œ,Ò&Ð&ÝÐNÑOÔOÐOÝ•z 'Ñ*Ô*Ñ+Ô+€KØ˜ŒO€EØ ×'Ò'Ñ)Ô)€IØ•f˜[¨%°)Ñ*;Ñ<Ô<Ô=Ð=r   c                   ód   — e Zd ZdZdd„Zdd„Zdd„Zedd„¦   «         Zdd„Z	d„ Z
d	„ Zd
„ Zdd„ZdS )ÚLanguageModelzKABC for Language Models.

    Cannot be directly instantiated itself.

    Nc                 óæ   — || _         |r9t          |t          ¦  «        s$t          j        d| j        j        ›d�d¬¦  «         |€t          ¦   «         n|| _        |€t          ¦   «         n|| _	        dS )ap  Creates new LanguageModel.

        :param vocabulary: If provided, this vocabulary will be used instead
            of creating a new one when training.
        :type vocabulary: `nltk.lm.Vocabulary` or None
        :param counter: If provided, use this object to count ngrams.
        :type counter: `nltk.lm.NgramCounter` or None
        :param ngrams_fn: If given, defines how sentences in training text are turned to ngram
            sequences.
        :type ngrams_fn: function or None
        :param pad_fn: If given, defines how sentences in training text are padded.
        :type pad_fn: function or None
        z$The `vocabulary` argument passed to z- must be an instance of `nltk.lm.Vocabulary`.é   )Ú
stacklevelN)
Úorderr,   r	   ÚwarningsÚwarnÚ	__class__r    r   r   r   )r   rA   r   r   s       r   r   zLanguageModel.__init__P   sˆ   € ð ˆŒ
Øð 	�j¨µZÑ@Ô@ð 	ÝŒMð?°t´~Ô7Nð ?ð ?ð ?àðñ ô ð ð
 &0Ð%7•Z‘\”\�\¸ZˆŒ
Ø(/¨•l‘n”n�n¸WˆŒˆˆr   c                 ó¸   ‡ — ‰ j         s+|€t          d¦  «        ‚‰ j                              |¦  «         ‰ j                             ˆ fd„|D ¦   «         ¦  «         dS )zeTrains the model on a text.

        :param text: Training text as a sequence of sentences.

        Nz:Cannot fit without a vocabulary or text to create it from.c              3   óL   •K  — | ]}‰j                              |¦  «        V — Œd S r   )r   Úlookup)Ú.0Úsentr   s     €r   ú	<genexpr>z$LanguageModel.fit.<locals>.<genexpr>t   s3   øè è € ÐDÐD°t˜4œ:×,Ò,¨TÑ2Ô2ÐDÐDÐDÐDÐDÐDr   )r   r3   Úupdater   )r   ÚtextÚvocabulary_texts   `  r   ÚfitzLanguageModel.fith   ss   ø€ ð Œzð 	/ØÐ&Ý ØPñô ð ð ŒJ×Ò˜oÑ.Ô.Ð.ØŒ×ÒÐDÐDÐDÐD¸tÐDÑDÔDÑDÔDÐDÐDÐDr   c                 ó–   — |                       | j                             |¦  «        |r| j                             |¦  «        nd¦  «        S )z©Masks out of vocab (OOV) words and computes their model score.

        For model-specific logic of calculating scores, see the `unmasked_score`
        method.
        N)Úunmasked_scorer   rG   r   s      r   ÚscorezLanguageModel.scorev   sL   € ð ×"Ò"ØŒJ×Ò˜dÑ#Ô#À7Ð%T T¤Z×%6Ò%6°wÑ%?Ô%?Ð%?ÐPTñ
ô 
ð 	
r   c                 ó   — t          ¦   «         ‚)aÒ  Score a word given some optional context.

        Concrete models are expected to provide an implementation.
        Note that this method does not mask its arguments with the OOV label.
        Use the `score` method for that.

        :param str word: Word for which we want the score
        :param tuple(str) context: Context the word is in.
            If `None`, compute unigram score.
        :param context: tuple(str) or None
        :rtype: float
        r   r   s      r   rP   zLanguageModel.unmasked_score€   s   € õ "Ñ#Ô#Ð#r   c                 óH   — t          |                      ||¦  «        ¦  «        S )z‡Evaluate the log score of this word in this context.

        The arguments are the same as for `score` and `unmasked_score`.

        )r   rQ   r   s      r   ÚlogscorezLanguageModel.logscore�   s    € õ ˜Ÿš D¨'Ñ2Ô2Ñ3Ô3Ð3r   c                 ód   — |r#| j         t          |¦  «        dz            |         n| j         j        S )z²Helper method for retrieving counts for a given context.

        Assumes context has been checked and oov words in it masked.
        :type context: tuple(str) or None

        é   )r   r(   Úunigrams)r   r   s     r   Úcontext_countszLanguageModel.context_counts˜   s2   € ð 7>ÐWˆDŒK�˜G™œ qÑ(Ô)¨'Ô2Ð2À4Ä;ÔCWð	
r   c                 ó@   ‡ — dt          ˆ fd„|D ¦   «         ¦  «        z  S )z©Calculate cross-entropy of model for given evaluation text.

        :param Iterable(tuple(str)) text_ngrams: A sequence of ngram tuples.
        :rtype: float

        r2   c                 óX   •— g | ]&}‰                      |d          |dd …         ¦  «        ‘Œ'S )r2   N)rT   )rH   Úngramr   s     €r   ú
<listcomp>z)LanguageModel.entropy.<locals>.<listcomp>«   s3   ø€ ÐKÐKÐK°eˆT�]Š]˜5 œ9 e¨C¨R¨C¤jÑ1Ô1ÐKÐKÐKr   )r*   ©r   Útext_ngramss   ` r   ÚentropyzLanguageModel.entropy£   s5   ø€ ð •EØKÐKÐKÐK¸{ÐKÑKÔKñ
ô 
ñ 
ð 	
r   c                 óH   — t          d|                      |¦  «        ¦  «        S )zŽCalculates the perplexity of the given text.

        This is simply 2 ** cross-entropy for the text, so the arguments are the same.

        g       @)Úpowr_   r]   s     r   Ú
perplexityzLanguageModel.perplexity®   s    € õ �3˜Ÿš [Ñ1Ô1Ñ2Ô2Ð2r   rV   c                 ó®  ‡ ‡— |€g nt          |¦  «        }t          |¦  «        }|dk    rèt          |¦  «        ‰ j        k    r|‰ j         dz   d…         n|Š‰                      ‰ j                             ‰¦  «        ¦  «        }‰rR|sPt          ‰¦  «        dk    r
‰dd…         ng Š‰                      ‰ j                             ‰¦  «        ¦  «        }‰r|¯Pt          |¦  «        }t          |t          ˆˆ fd„|D ¦   «         ¦  «        |¦  «        S g }t          |¦  «        D ]0}|                     ‰                      d||z   |¬¦  «        ¦  «         Œ1|S )aà  Generate words from the model.

        :param int num_words: How many words to generate. By default 1.
        :param text_seed: Generation can be conditioned on preceding context.
        :param random_seed: A random seed or an instance of `random.Random`. If provided,
            makes the random sampling part of generation reproducible.
        :return: One (str) word or a list of words generated from model.

        Examples:

        >>> from nltk.lm import MLE
        >>> lm = MLE(2)
        >>> lm.fit([[("a", "b"), ("b", "c")]], vocabulary_text=['a', 'b', 'c'])
        >>> lm.fit([[("a",), ("b",), ("c",)]])
        >>> lm.generate(random_seed=3)
        'a'
        >>> lm.generate(text_seed=['a'])
        'b'

        NrV   c              3   óD   •K  — | ]}‰                      |‰¦  «        V — Œd S r   )rQ   )rH   Úwr   r   s     €€r   rJ   z)LanguageModel.generate.<locals>.<genexpr>Þ   s1   øè è € Ð>Ð>°�d—j’j  GÑ,Ô,Ð>Ð>Ð>Ð>Ð>Ð>r   )Ú	num_wordsÚ	text_seedÚrandom_seed)r4   r0   r(   rA   rX   r   rG   Úsortedr;   ÚtupleÚrangeÚappendÚgenerate)	r   rf   rg   rh   r7   ÚsamplesÚ	generatedÚ_r   s	   `       @r   rm   zLanguageModel.generate¶   s¥  øø€ ð* $Ð+�B�Bµ°i±´ˆ	Ý,¨[Ñ9Ô9Ðà˜Š>ˆ>õ �y‘>”> T¤ZÒ/Ð/ð ˜4œ:˜+¨™/Ð+Ð+Ô,Ð,àð ð
 ×)Ò)¨$¬*×*;Ò*;¸GÑ*DÔ*DÑEÔEˆGØð J 'ð JÝ),¨W©¬¸Ò)9Ð)9˜' ! " "œ+˜+¸r�Ø×-Ò-¨d¬j×.?Ò.?ÀÑ.HÔ.HÑIÔI�ð ð J 'ð Jõ ˜W‘o”oˆGÝ#ØÝÐ>Ð>Ð>Ð>Ð>°gÐ>Ñ>Ô>Ñ>Ô>Ø ñô ð ð ˆ	Ý�yÑ!Ô!ð 	ð 	ˆAØ×ÒØ—’ØØ'¨)Ñ3Ø 0ð ñ ô ñô ð ð ð Ðr   )NNr   )rV   NN)r    r!   r"   r#   r   rN   rQ   r   rP   rT   rX   r_   rb   rm   r$   r   r   r=   r=   I   s×   € € € € € ðð ðEð Eð Eð Eð0Eð Eð Eð Eð
ð 
ð 
ð 
ð ð$ð $ð $ñ „^ð$ð4ð 4ð 4ð 4ð	
ð 	
ð 	
ð	
ð 	
ð 	
ð3ð 3ð 3ð5ð 5ð 5ð 5ð 5ð 5r   r=   r   )r#   r-   rB   Úabcr   r   r   Ú	itertoolsr   Únltk.lm.counterr   Únltk.lm.utilr   Únltk.lm.vocabularyr	   r   r*   r0   r;   r=   r$   r   r   ú<module>rv      s6  ðð  Ð à €€€Ø €€€Ø 'Ð 'Ð 'Ð 'Ð 'Ð 'Ð 'Ð 'Ø Ð Ð Ð Ð Ð Ø  Ð  Ð  Ð  Ð  Ð  à (Ð (Ð (Ð (Ð (Ð (Ø "Ð "Ð "Ð "Ð "Ð "Ø )Ð )Ð )Ð )Ð )Ð )ð$ð $ð $ð $ð $˜'ð $ñ $ô $ð $ð6#ð #ð #ð
,ð ,ð ,ð>ð >ð >ð >ðbð bð bð bð b˜gð bñ bô bð bð bð br   