§
    'ê[f»  ã                   óê   — d Z ddlZddlmZ ddlmZ ddlmZ ddlm	Z	 ed„ ¦   «         Z
e
                     e¦  «        d„ ¦   «         Ze
                     e¦  «        d	„ ¦   «         Z G d
„ d¦  «        ZdS )zLanguage Model Vocabularyé    N)ÚCounter)ÚIterable)Úsingledispatch)Úchainc                 ó@   — t          dt          | ¦  «        › �¦  «        ‚)Nz/Unsupported type for looking up in vocabulary: )Ú	TypeErrorÚtype©ÚwordsÚvocabs     úF/var/www/piapp/venv/lib/python3.11/site-packages/nltk/lm/vocabulary.pyÚ_dispatched_lookupr      s   € å
ÐSÅdÈ5ÁkÄkÐSÐSÑ
TÔ
TÐTó    c                 ó:   ‡— t          ˆfd„| D ¦   «         ¦  «        S )zcLook up a sequence of words in the vocabulary.

    Returns an iterator over looked up words.

    c              3   ó8   •K  — | ]}t          |‰¦  «        V — Œd S ©N©r   )Ú.0Úwr   s     €r   ú	<genexpr>z_.<locals>.<genexpr>   s.   øè è € Ð=Ð=°!Õ# A uÑ-Ô-Ð=Ð=Ð=Ð=Ð=Ð=r   )Útupler
   s    `r   Ú_r      s(   ø€ õ Ð=Ð=Ð=Ð=°uÐ=Ñ=Ô=Ñ=Ô=Ð=r   c                 ó   — | |v r| n|j         S )z$Looks up one word in the vocabulary.)Ú	unk_label)Úwordr   s     r   Ú_string_lookupr      s   € ð ˜5�=�=ˆ4ˆ4 e¤oÐ5r   c                   ó`   — e Zd ZdZdd„Zed„ ¦   «         Zd„ Zd„ Zd	„ Z	d
„ Z
d„ Zd„ Zd„ Zd„ ZdS )Ú
VocabularyaÈ
  Stores language model vocabulary.

    Satisfies two common language modeling requirements for a vocabulary:

    - When checking membership and calculating its size, filters items
      by comparing their counts to a cutoff value.
    - Adds a special "unknown" token which unseen words are mapped to.

    >>> words = ['a', 'c', '-', 'd', 'c', 'a', 'b', 'r', 'a', 'c', 'd']
    >>> from nltk.lm import Vocabulary
    >>> vocab = Vocabulary(words, unk_cutoff=2)

    Tokens with counts greater than or equal to the cutoff value will
    be considered part of the vocabulary.

    >>> vocab['c']
    3
    >>> 'c' in vocab
    True
    >>> vocab['d']
    2
    >>> 'd' in vocab
    True

    Tokens with frequency counts less than the cutoff value will be considered not
    part of the vocabulary even though their entries in the count dictionary are
    preserved.

    >>> vocab['b']
    1
    >>> 'b' in vocab
    False
    >>> vocab['aliens']
    0
    >>> 'aliens' in vocab
    False

    Keeping the count entries for seen words allows us to change the cutoff value
    without having to recalculate the counts.

    >>> vocab2 = Vocabulary(vocab.counts, unk_cutoff=1)
    >>> "b" in vocab2
    True

    The cutoff value influences not only membership checking but also the result of
    getting the size of the vocabulary using the built-in `len`.
    Note that while the number of keys in the vocabulary's counter stays the same,
    the items in the vocabulary differ depending on the cutoff.
    We use `sorted` to demonstrate because it keeps the order consistent.

    >>> sorted(vocab2.counts)
    ['-', 'a', 'b', 'c', 'd', 'r']
    >>> sorted(vocab2)
    ['-', '<UNK>', 'a', 'b', 'c', 'd', 'r']
    >>> sorted(vocab.counts)
    ['-', 'a', 'b', 'c', 'd', 'r']
    >>> sorted(vocab)
    ['<UNK>', 'a', 'c', 'd']

    In addition to items it gets populated with, the vocabulary stores a special
    token that stands in for so-called "unknown" items. By default it's "<UNK>".

    >>> "<UNK>" in vocab
    True

    We can look up words in a vocabulary using its `lookup` method.
    "Unseen" words (with counts less than cutoff) are looked up as the unknown label.
    If given one word (a string) as an input, this method will return a string.

    >>> vocab.lookup("a")
    'a'
    >>> vocab.lookup("aliens")
    '<UNK>'

    If given a sequence, it will return an tuple of the looked up words.

    >>> vocab.lookup(["p", 'a', 'r', 'd', 'b', 'c'])
    ('<UNK>', 'a', '<UNK>', 'd', '<UNK>', 'c')

    It's possible to update the counts after the vocabulary has been created.
    In general, the interface is the same as that of `collections.Counter`.

    >>> vocab['b']
    1
    >>> vocab.update(["b", "b", "c"])
    >>> vocab['b']
    3
    Né   ú<UNK>c                 óª   — || _         |dk     rt          d|› �¦  «        ‚|| _        t          ¦   «         | _        |                      |�|nd¦  «         dS )aË  Create a new Vocabulary.

        :param counts: Optional iterable or `collections.Counter` instance to
                       pre-seed the Vocabulary. In case it is iterable, counts
                       are calculated.
        :param int unk_cutoff: Words that occur less frequently than this value
                               are not considered part of the vocabulary.
        :param unk_label: Label for marking words not part of vocabulary.

        r   z)Cutoff value cannot be less than 1. Got: NÚ )r   Ú
ValueErrorÚ_cutoffr   ÚcountsÚupdate)Úselfr%   Ú
unk_cutoffr   s       r   Ú__init__zVocabulary.__init__   s^   € ð #ˆŒØ˜Š>ˆ>ÝÐUÈÐUÐUÑVÔVÐVØ!ˆŒå‘i”iˆŒØ�Š˜fÐ0�F�F°bÑ9Ô9Ð9Ð9Ð9r   c                 ó   — | j         S )ziCutoff value.

        Items with count below this value are not considered part of vocabulary.

        )r$   ©r'   s    r   ÚcutoffzVocabulary.cutoff’   s   € ð Œ|Ðr   c                 óf   —  | j         j        |i |¤Ž t          d„ | D ¦   «         ¦  «        | _        dS )zWUpdate vocabulary counts.

        Wraps `collections.Counter.update` method.

        c              3   ó   K  — | ]}d V — ŒdS )r   N© )r   r   s     r   r   z$Vocabulary.update.<locals>.<genexpr>¢   s"   è è € Ð(Ð(˜a˜Ð(Ð(Ð(Ð(Ð(Ð(r   N)r%   r&   ÚsumÚ_len)r'   Úcounter_argsÚcounter_kwargss      r   r&   zVocabulary.update›   s@   € ð 	ˆŒÔ˜LÐ;¨NÐ;Ð;Ð;ÝÐ(Ð( 4Ð(Ñ(Ô(Ñ(Ô(ˆŒ	ˆ	ˆ	r   c                 ó"   — t          || ¦  «        S )a  Look up one or more words in the vocabulary.

        If passed one word as a string will return that word or `self.unk_label`.
        Otherwise will assume it was passed a sequence of words, will try to look
        each of them up and return an iterator over the looked up words.

        :param words: Word(s) to look up.
        :type words: Iterable(str) or str
        :rtype: generator(str) or str
        :raises: TypeError for types other than strings or iterables

        >>> from nltk.lm import Vocabulary
        >>> vocab = Vocabulary(["a", "b", "c", "a", "b"], unk_cutoff=2)
        >>> vocab.lookup("a")
        'a'
        >>> vocab.lookup("aliens")
        '<UNK>'
        >>> vocab.lookup(["a", "b", "c", ["x", "b"]])
        ('a', 'b', '<UNK>', ('<UNK>', 'b'))

        r   )r'   r   s     r   ÚlookupzVocabulary.lookup¤   s   € õ, " %¨Ñ.Ô.Ð.r   c                 ó@   — || j         k    r| j        n| j        |         S r   )r   r$   r%   ©r'   Úitems     r   Ú__getitem__zVocabulary.__getitem__¼   s!   € Ø# t¤~Ò5Ð5ˆtŒ|ˆ|¸4¼;ÀtÔ;LÐLr   c                 ó$   — | |         | j         k    S )zPOnly consider items with counts GE to cutoff as being in the
        vocabulary.)r,   r7   s     r   Ú__contains__zVocabulary.__contains__¿   s   € ð �DŒz˜Tœ[Ò(Ð(r   c                 ód   ‡ — t          ˆ fd„‰ j        D ¦   «         ‰ j        r‰ j        gng ¦  «        S )zKBuilding on membership check define how to iterate over
        vocabulary.c              3   ó$   •K  — | ]
}|‰v ¯|V — Œd S r   r/   )r   r8   r'   s     €r   r   z&Vocabulary.__iter__.<locals>.<genexpr>È   s'   øè è € Ð:Ð:�d¨T°T¨\¨\ˆT¨\¨\¨\¨\Ð:Ð:r   )r   r%   r   r+   s   `r   Ú__iter__zVocabulary.__iter__Ä   sD   ø€ õ Ø:Ð:Ð:Ð:˜dœkÐ:Ñ:Ô:Ø $¤Ð3ˆTŒ^ÐÐ°ñ
ô 
ð 	
r   c                 ó   — | j         S )z1Computing size of vocabulary reflects the cutoff.)r1   r+   s    r   Ú__len__zVocabulary.__len__Ì   s
   € àŒyÐr   c                 ób   — | j         |j         k    o| j        |j        k    o| j        |j        k    S r   )r   r,   r%   )r'   Úothers     r   Ú__eq__zVocabulary.__eq__Ð   s5   € àŒN˜eœoÒ-ð ,Ø”˜uœ|Ò+ð,à”˜uœ|Ò+ð	
r   c                 ót   — d                      | j        j        | j        | j        t          | ¦  «        ¦  «        S )Nz/<{} with cutoff={} unk_label='{}' and {} items>)ÚformatÚ	__class__Ú__name__r,   r   Úlenr+   s    r   Ú__str__zVocabulary.__str__×   s2   € Ø@×GÒGØŒNÔ# T¤[°$´.Å#ÀdÁ)Ä)ñ
ô 
ð 	
r   )Nr   r    )rG   Ú
__module__Ú__qualname__Ú__doc__r)   Úpropertyr,   r&   r5   r9   r;   r>   r@   rC   rI   r/   r   r   r   r   %   sË   € € € € € ðWð Wðr:ð :ð :ð :ð& ðð ñ „Xðð)ð )ð )ð/ð /ð /ð0Mð Mð Mð)ð )ð )ð

ð 
ð 
ðð ð ð
ð 
ð 
ð
ð 
ð 
ð 
ð 
r   r   )rL   ÚsysÚcollectionsr   Úcollections.abcr   Ú	functoolsr   Ú	itertoolsr   r   Úregisterr   Ústrr   r   r/   r   r   ú<module>rU      s  ðð  Ð à 
€
€
€
Ø Ð Ð Ð Ð Ð Ø $Ð $Ð $Ð $Ð $Ð $Ø $Ð $Ð $Ð $Ð $Ð $Ø Ð Ð Ð Ð Ð ð ðUð Uñ „ðUð ×Ò˜XÑ&Ô&ð>ð >ñ 'Ô&ð>ð ×Ò˜SÑ!Ô!ð6ð 6ñ "Ô!ð6ð
u
ð u
ð u
ð u
ð u
ñ u
ô u
ð u
ð u
ð u
r   