§
    'ê[fÛ!  ã                   óŽ   — d Z ddlmZ ddlmZmZmZ ddlmZm	Z	 ddl
mZmZ ddlmZ 	 dd„Zd	„ Zd
„ Z G d„ d¦  «        Zdd„ZdS )z 
Utility functions for parsers.
é    )Úload)ÚCFGÚPCFGÚFeatureGrammar)ÚChartÚChartParser)ÚFeatureChartÚFeatureChartParser)ÚInsideChartParserNc                 óf  — t          | fi |¤Ž}t          |t          ¦  «        st          d¦  «        ‚t          |t          ¦  «        r|€t
          } ||||¬¦  «        S t          |t          ¦  «        r |€t          }|€t          } ||||¬¦  «        S |€t          }|€t          } ||||¬¦  «        S )a¦  
    Load a grammar from a file, and build a parser based on that grammar.
    The parser depends on the grammar format, and might also depend
    on properties of the grammar itself.

    The following grammar formats are currently supported:
      - ``'cfg'``  (CFGs: ``CFG``)
      - ``'pcfg'`` (probabilistic CFGs: ``PCFG``)
      - ``'fcfg'`` (feature-based CFGs: ``FeatureGrammar``)

    :type grammar_url: str
    :param grammar_url: A URL specifying where the grammar is located.
        The default protocol is ``"nltk:"``, which searches for the file
        in the the NLTK data package.
    :type trace: int
    :param trace: The level of tracing that should be used when
        parsing a text.  ``0`` will generate no tracing output;
        and higher numbers will produce more verbose tracing output.
    :param parser: The class used for parsing; should be ``ChartParser``
        or a subclass.
        If None, the class depends on the grammar format.
    :param chart_class: The class used for storing the chart;
        should be ``Chart`` or a subclass.
        Only used for CFGs and feature CFGs.
        If None, the chart class depends on the grammar format.
    :type beam_size: int
    :param beam_size: The maximum length for the parser's edge queue.
        Only used for probabilistic CFGs.
    :param load_args: Keyword parameters used when loading the grammar.
        See ``data.load`` for more information.
    z1The grammar must be a CFG, or a subclass thereof.N)ÚtraceÚ	beam_size)r   Úchart_class)r   Ú
isinstancer   Ú
ValueErrorr   r   r   r
   r	   r   r   )Úgrammar_urlr   Úparserr   r   Ú	load_argsÚgrammars          úC/var/www/piapp/venv/lib/python3.11/site-packages/nltk/parse/util.pyÚload_parserr      sÝ   € õD �;Ð,Ð, )Ð,Ð,€GÝ�g�sÑ#Ô#ð QÝÐOÑPÔPÐPÝ�'�4Ñ Ô ð EØˆ>Ý&ˆFØˆv�g U°iÐ@Ñ@Ô@Ð@å	�G�^Ñ	,Ô	,ð EØˆ>Ý'ˆFØÐÝ&ˆKØˆv�g U¸ÐDÑDÔDÐDð ˆ>Ý ˆFØÐÝˆKØˆv�g U¸ÐDÑDÔDÐDó    c              #   ó¨   K  — t          | d¬¦  «        D ]=\  }\  }}t          |¦  «        |d||dddddg
}d                     |¦  «        dz   }|V — Œ>dS )	a°  
    A module to convert a single POS tagged sentence into CONLL format.

    >>> from nltk import word_tokenize, pos_tag
    >>> text = "This is a foobar sentence."
    >>> for line in taggedsent_to_conll(pos_tag(word_tokenize(text))): # doctest: +NORMALIZE_WHITESPACE
    ... 	print(line, end="")
        1	This	_	DT	DT	_	0	a	_	_
        2	is	_	VBZ	VBZ	_	0	a	_	_
        3	a	_	DT	DT	_	0	a	_	_
        4	foobar	_	JJ	JJ	_	0	a	_	_
        5	sentence	_	NN	NN	_	0	a	_	_
        6	.		_	.	.	_	0	a	_	_

    :param sentence: A single input sentence to parse
    :type sentence: list(tuple(str, str))
    :rtype: iter(str)
    :return: a generator yielding a single sentence in CONLL format.
    é   )ÚstartÚ_Ú0Úaú	Ú
N)Ú	enumerateÚstrÚjoin)ÚsentenceÚiÚwordÚtagÚ	input_strs        r   Útaggedsent_to_conllr)   O   sx   è è € õ( & h°aÐ8Ñ8Ô8ð ð Ñˆ‰KˆT�3Ý˜‘V”V˜T 3¨¨S°#°s¸CÀÀcÐJˆ	Ø—I’I˜iÑ(Ô(¨4Ñ/ˆ	Øˆˆˆˆðð r   c              #   óF   K  — | D ]}t          |¦  «        E d{V —† dV — ŒdS )aV  
    A module to convert the a POS tagged document stream
    (i.e. list of list of tuples, a list of sentences) and yield lines
    in CONLL format. This module yields one line per word and two newlines
    for end of sentence.

    >>> from nltk import word_tokenize, sent_tokenize, pos_tag
    >>> text = "This is a foobar sentence. Is that right?"
    >>> sentences = [pos_tag(word_tokenize(sent)) for sent in sent_tokenize(text)]
    >>> for line in taggedsents_to_conll(sentences): # doctest: +NORMALIZE_WHITESPACE
    ...     if line:
    ...         print(line, end="")
    1	This	_	DT	DT	_	0	a	_	_
    2	is	_	VBZ	VBZ	_	0	a	_	_
    3	a	_	DT	DT	_	0	a	_	_
    4	foobar	_	JJ	JJ	_	0	a	_	_
    5	sentence	_	NN	NN	_	0	a	_	_
    6	.		_	.	.	_	0	a	_	_
    <BLANKLINE>
    <BLANKLINE>
    1	Is	_	VBZ	VBZ	_	0	a	_	_
    2	that	_	IN	IN	_	0	a	_	_
    3	right	_	NN	NN	_	0	a	_	_
    4	?	_	.	.	_	0	a	_	_
    <BLANKLINE>
    <BLANKLINE>

    :param sentences: Input sentences to parse
    :type sentence: list(list(tuple(str, str)))
    :rtype: iter(str)
    :return: a generator yielding sentences in CONLL format.
    Nz

)r)   )Ú	sentencesr$   s     r   Útaggedsents_to_conllr,   i   sM   è è € ðB ð ð ˆÝ& xÑ0Ô0Ð0Ð0Ð0Ð0Ð0Ð0Ð0Øˆˆˆˆðð r   c                   ó"   — e Zd ZdZdd„Zdd„ZdS )ÚTestGrammarz
    Unit tests for  CFG.
    Nc                 ój   — || _         t          |d¬¦  «        | _        || _        || _        || _        d S )Nr   )r   )Útest_grammarr   ÚcpÚsuiteÚ_acceptÚ_reject)Úselfr   r2   ÚacceptÚrejects        r   Ú__init__zTestGrammar.__init__™   s7   € Ø#ˆÔå˜g¨QÐ/Ñ/Ô/ˆŒØˆŒ
ØˆŒØˆŒˆˆr   Fc                 óâ  — | j         D ]æ}t          |d         dz   d¬¦  «         dD ]´}||         D ]©}|                     ¦   «         }t          | j                             |¦  «        ¦  «        }|r3|r1t          ¦   «          t          |¦  «         |D ]}t          |¦  «         Œ|dk    r|g k    rt          d|z  ¦  «        ‚d}Œ“|rt          d	|z  ¦  «        ‚d}	ŒªŒµ|r|	rt          d
¦  «         ŒçdS )a}  
        Sentences in the test suite are divided into two classes:

        - grammatical (``accept``) and
        - ungrammatical (``reject``).

        If a sentence should parse according to the grammar, the value of
        ``trees`` will be a non-empty list. If a sentence should be rejected
        according to the grammar, then the value of ``trees`` will be None.
        Údocú:Ú )Úend)r6   r7   r6   zSentence '%s' failed to parse'TzSentence '%s' received a parse'zAll tests passed!N)r2   ÚprintÚsplitÚlistr1   Úparser   )
r5   Ú
show_treesÚtestÚkeyÚsentÚtokensÚtreesÚtreeÚacceptedÚrejecteds
             r   ÚrunzTestGrammar.run¡   s<  € ð ”Jð 	+ð 	+ˆDÝ�$�u”+ Ñ#¨Ð-Ñ-Ô-Ð-Ø+ð ,ð ,�Ø  œIð ,ð ,�DØ!ŸZšZ™\œ\�FÝ  ¤§¢¨vÑ!6Ô!6Ñ7Ô7�EØ!ð ( eð (Ý™œ˜Ý˜d™œ˜Ø$)ð (ð (˜DÝ! $™KœK˜K˜KØ˜h’�Ø  Bš;˜;Ý",Ð-MÐPTÑ-TÑ"UÔ"UÐUà'+˜H˜Hà ð ,Ý",Ð-NÐQUÑ-UÑ"VÔ"VÐVà'+˜H˜Hð#,ð$ ð +˜Hð +ÝÐ)Ñ*Ô*Ð*øð-	+ð 	+r   )NN)F)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r8   rK   © r   r   r.   r.   ”   sF   € € € € € ðð ðð ð ð ð!+ð !+ð !+ð !+ð !+ð !+r   r.   ú#%;c                 óš  — |�|                       |¦  «        } g }|                      d¦  «        D ]›}|dk    s
|d         |v rŒ|                     dd¦  «        }d}t          |¦  «        dk    r:|d         dv r|d         d	v }|d         }nt          |d         ¦  «        }|d         }|                     ¦   «         }|g k    rŒ“|||fgz  }Œœ|S )
aŒ  
    Parses a string with one test sentence per line.
    Lines can optionally begin with:

    - a bool, saying if the sentence is grammatical or not, or
    - an int, giving the number of parse trees is should have,

    The result information is followed by a colon, and then the sentence.
    Empty lines and lines beginning with a comment char are ignored.

    :return: a list of tuple of sentences and expected results,
        where a sentence is a list of str,
        and a result is None, or bool, or int

    :param comment_chars: ``str`` of possible comment characters.
    :param encoding: the encoding of the string, if it is binary
    Nr    Ú r   r;   r   é   )ÚTrueÚtrueÚFalseÚfalse)rU   rV   )Údecoder?   ÚlenÚint)ÚstringÚcomment_charsÚencodingr+   r$   Ú
split_infoÚresultrF   s           r   Úextract_test_sentencesra   Å   s÷   € ð$ ÐØ—’˜xÑ(Ô(ˆØ€IØ—L’L Ñ&Ô&ð (ð (ˆØ�rŠ>ˆ>˜X aœ[¨MÐ9Ð9ØØ—^’^ C¨Ñ+Ô+ˆ
ØˆÝˆz‰?Œ?˜aÒÐØ˜!Œ}Ð BÐBÐBØ# AœÐ*:Ð:�Ø% aœ=��å˜Z¨œ]Ñ+Ô+�Ø% aœ=�Ø—’Ñ!Ô!ˆØ�RŠ<ˆ<ØØ�v˜vÐ&Ð'Ñ'ˆ	ˆ	ØÐr   )r   NNr   )rQ   N)rO   Ú	nltk.datar   Únltk.grammarr   r   r   Únltk.parse.chartr   r   Únltk.parse.featurechartr	   r
   Únltk.parse.pchartr   r   r)   r,   r.   ra   rP   r   r   ú<module>rg      s	  ððð ð Ð Ð Ð Ð Ð Ø 2Ð 2Ð 2Ð 2Ð 2Ð 2Ð 2Ð 2Ð 2Ð 2Ø /Ð /Ð /Ð /Ð /Ð /Ð /Ð /Ø DÐ DÐ DÐ DÐ DÐ DÐ DÐ DØ /Ð /Ð /Ð /Ð /Ð /ð DEð6Eð 6Eð 6Eð 6Eðrð ð ð4#ð #ð #ðV.+ð .+ð .+ð .+ð .+ñ .+ô .+ð .+ðb%ð %ð %ð %ð %ð %r   