§
    'ê[fô%  ã                   ó†   — d Z ddlmZ ddlmZmZmZ  G d„ de¦  «        Zdd„Z G d„ d	e	¦  «        Z
 G d
„ de¦  «        ZdS )zACorpus reader for the XML version of the British National Corpus.é    )Úconcat)ÚElementTreeÚXMLCorpusReaderÚXMLCorpusViewc                   óH   — e Zd ZdZdd„Zdd„Zdd„Zdd„Zdd	„Zdd
„Z	d„ Z
dS )ÚBNCCorpusReadera7  Corpus reader for the XML version of the British National Corpus.

    For access to the complete XML data structure, use the ``xml()``
    method.  For access to simple word lists and tagged word lists, use
    ``words()``, ``sents()``, ``tagged_words()``, and ``tagged_sents()``.

    You can obtain the full version of the BNC corpus at
    https://www.ota.ox.ac.uk/desc/2554

    If you extracted the archive to a directory called `BNC`, then you can
    instantiate the reader as::

        BNCCorpusReader(root='BNC/Texts/', fileids=r'[A-K]/\w*/\w*\.xml')

    Tc                 ó@   — t          j        | ||¦  «         || _        d S ©N)r   Ú__init__Ú_lazy)ÚselfÚrootÚfileidsÚlazys       úJ/var/www/piapp/venv/lib/python3.11/site-packages/nltk/corpus/reader/bnc.pyr   zBNCCorpusReader.__init__   s"   € ÝÔ   t¨WÑ5Ô5Ð5ØˆŒ
ˆ
ˆ
ó    NFc                 ó4   — |                       |dd||¦  «        S )aT  
        :return: the given file(s) as a list of words
            and punctuation symbols.
        :rtype: list(str)

        :param strip_space: If true, then strip trailing spaces from
            word tokens.  Otherwise, leave the spaces on the tokens.
        :param stem: If true, then use word stems instead of word strings.
        FN©Ú_views©r   r   Ústrip_spaceÚstems       r   ÚwordszBNCCorpusReader.words#   s   € ð �{Š{˜7 E¨4°¸dÑCÔCÐCr   c                 ó@   — |rdnd}|                       |d|||¦  «        S )a   
        :return: the given file(s) as a list of tagged
            words and punctuation symbols, encoded as tuples
            ``(word,tag)``.
        :rtype: list(tuple(str,str))

        :param c5: If true, then the tags used will be the more detailed
            c5 tags.  Otherwise, the simplified tags will be used.
        :param strip_space: If true, then strip trailing spaces from
            word tokens.  Otherwise, leave the spaces on the tokens.
        :param stem: If true, then use word stems instead of word strings.
        Úc5ÚposFr   ©r   r   r   r   r   Útags         r   Útagged_wordszBNCCorpusReader.tagged_words/   s,   € ð Ð#ˆdˆd˜eˆØ�{Š{˜7 E¨3°¸TÑBÔBÐBr   c                 ó4   — |                       |dd||¦  «        S )aˆ  
        :return: the given file(s) as a list of
            sentences or utterances, each encoded as a list of word
            strings.
        :rtype: list(list(str))

        :param strip_space: If true, then strip trailing spaces from
            word tokens.  Otherwise, leave the spaces on the tokens.
        :param stem: If true, then use word stems instead of word strings.
        TNr   r   s       r   ÚsentszBNCCorpusReader.sents?   s   € ð �{Š{˜7 D¨$°¸TÑBÔBÐBr   c                 óB   — |rdnd}|                       |d|||¬¦  «        S )a  
        :return: the given file(s) as a list of
            sentences, each encoded as a list of ``(word,tag)`` tuples.
        :rtype: list(list(tuple(str,str)))

        :param c5: If true, then the tags used will be the more detailed
            c5 tags.  Otherwise, the simplified tags will be used.
        :param strip_space: If true, then strip trailing spaces from
            word tokens.  Otherwise, leave the spaces on the tokens.
        :param stem: If true, then use word stems instead of word strings.
        r   r   T)Úsentr   r   r   r   r   s         r   Útagged_sentszBNCCorpusReader.tagged_sentsL   s8   € ð Ð#ˆdˆd˜eˆØ�{Š{Ø˜$ C°[Àtð ñ 
ô 
ð 	
r   c                 óš   ‡‡‡‡‡— | j         rt          n| j        Št          ˆˆˆˆˆfd„|                      |¦  «        D ¦   «         ¦  «        S )zPA helper function that instantiates BNCWordViews or the list of words/sentences.c           	      ó.   •— g | ]} ‰|‰‰‰‰¦  «        ‘ŒS © r'   )Ú.0ÚfileidÚfr#   r   r   r   s     €€€€€r   ú
<listcomp>z*BNCCorpusReader._views.<locals>.<listcomp>a   s;   ø€ ð ð ð àð ��&˜$  [°$Ñ7Ô7ðð ð r   )r   ÚBNCWordViewÚ_wordsr   Úabspaths)r   r   r#   r   r   r   r*   s     ````@r   r   zBNCCorpusReader._views]   sq   øøøøø€ àœ:Ð6�KˆK¨4¬;ˆÝðð ð ð ð ð ð ð à"Ÿmšm¨GÑ4Ô4ðñ ô ñ
ô 
ð 	
r   c           	      ó„  — g }t          j        |¦  «                             ¦   «         }|                     d¦  «        D ]û}g }	t	          |¦  «        D ]¡}
|
j        }|sd}|s|r|                     ¦   «         }|r|
                     d|¦  «        }|dk    r||
                     d¦  «        f}n1|dk    r+||
                     d|
                     d¦  «        ¦  «        f}|	                     |¦  «         Œ¢|r/|                     t          |j
        d         |	¦  «        ¦  «         Œæ|                     |	¦  «         Œüd|vsJ ‚|S )aÐ  
        Helper used to implement the view methods -- returns a list of
        words or a list of sentences, optionally tagged.

        :param fileid: The name of the underlying file.
        :param bracket_sent: If true, include sentence bracketing.
        :param tag: The name of the tagset to use, or None for no tags.
        :param strip_space: If true, strip spaces from word tokens.
        :param stem: If true, then substitute stems for words.
        z.//sÚ Úhwr   r   ÚnN)r   ÚparseÚgetrootÚfindallÚ_all_xmlwords_inÚtextÚstripÚgetÚappendÚBNCSentenceÚattribÚextend)r   r)   Úbracket_sentr   r   r   ÚresultÚxmldocÚxmlsentr#   ÚxmlwordÚwords               r   r-   zBNCCorpusReader._wordsg   sc  € ð ˆåÔ" 6Ñ*Ô*×2Ò2Ñ4Ô4ˆØ—~’~ fÑ-Ô-ð 	$ð 	$ˆGØˆDÝ+¨GÑ4Ô4ð "ð "�Ø”|�Øð Ø�DØð ( $ð (ØŸ:š:™<œ<�DØð 3Ø"Ÿ;š; t¨TÑ2Ô2�DØ˜$’;�;Ø  '§+¢+¨dÑ"3Ô"3Ð4�D�DØ˜E’\�\Ø  '§+¢+¨e°W·[²[ÀÑ5FÔ5FÑ"GÔ"GÐH�DØ—’˜DÑ!Ô!Ð!Ð!Øð $Ø—’�k¨'¬.¸Ô*=¸tÑDÔDÑEÔEÐEÐEà—’˜dÑ#Ô#Ð#Ð#à˜6Ð!Ð!Ð!Ð!Øˆr   )T)NTF)NFTF)NFFTF)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r   r!   r$   r   r-   r'   r   r   r   r      s±   € € € € € ðð ð ð ð ð ð
Dð 
Dð 
Dð 
DðCð Cð Cð Cð Cð Cð Cð Cð
ð 
ð 
ð 
ð"
ð 
ð 
ð 
ð#ð #ð #ð #ð #r   r   Nc                 óv   — |€g }| D ]1}|j         dv r|                     |¦  «         Œ!t          ||¦  «         Œ2|S )N)ÚcÚw)r   r:   r6   )Úeltr?   Úchilds      r   r6   r6   �   sU   € Ø€~ØˆØð ,ð ,ˆØŒ9˜
Ð"Ð"Ø�MŠM˜%Ñ Ô Ð Ð å˜U FÑ+Ô+Ð+Ð+Ø€Mr   c                   ó   — e Zd ZdZd„ ZdS )r;   z‹
    A list of words, augmented by an attribute ``num`` used to record
    the sentence identifier (the ``n`` attribute from the XML).
    c                 óJ   — || _         t                               | |¦  «         d S r
   )ÚnumÚlistr   )r   rO   Úitemss      r   r   zBNCSentence.__init__ž   s#   € ØˆŒÝ�Š�d˜EÑ"Ô"Ð"Ð"Ð"r   N)rD   rE   rF   rG   r   r'   r   r   r;   r;   ˜   s-   € € € € € ðð ð
#ð #ð #ð #ð #r   r;   c                   ó:   — e Zd ZdZh d£Z	 d„ Zd„ Zd„ Zd„ Zd„ Z	dS )	r,   zN
    A stream backed corpus view specialized for use with the BNC corpus.
    >   ÚpbÚgapÚalignÚeventÚpauseÚshiftÚvocalÚunclearc                 óT  — |rd}nd}|| _         || _        || _        || _        d| _        d| _        d| _        d| _        t          j	        | ||¦  «         |  
                    ¦   «          |                      | j        d| j        ¦  «         |                      ¦   «          ddi| _        dS )aG  
        :param fileid: The name of the underlying file.
        :param sent: If true, include sentence bracketing.
        :param tag: The name of the tagset to use, or None for no tags.
        :param strip_space: If true, strip spaces from word tokens.
        :param stem: If true, then substitute stems for words.
        z.*/sz.*/s/(.*/)?(c|w)Nz.*/teiHeader$r   r'   )Ú_sentÚ_tagÚ_strip_spaceÚ_stemÚtitleÚauthorÚeditorÚrespsr   r   Ú_openÚ
read_blockÚ_streamÚhandle_headerÚcloseÚ_tag_context)r   r)   r#   r   r   r   Útagspecs          r   r   zBNCWordView.__init__¸   s±   € ð ð 	)ØˆGˆGà(ˆGØˆŒ
ØˆŒ	Ø'ˆÔØˆŒ
àˆŒ
ØˆŒØˆŒØˆŒ
åÔ˜t V¨WÑ5Ô5Ð5ð 	�
Š
‰ŒˆØ�Š˜œ o°tÔ7IÑJÔJÐJØ�
Š
‰Œˆð  ˜GˆÔÐÐr   c                 óâ  — |                      d¦  «        }|r$d                     d„ |D ¦   «         ¦  «        | _        |                      d¦  «        }|r$d                     d„ |D ¦   «         ¦  «        | _        |                      d¦  «        }|r$d                     d„ |D ¦   «         ¦  «        | _        |                      d¦  «        }|r&d	                     d
„ |D ¦   «         ¦  «        | _        d S d S )NztitleStmt/titleÚ
c              3   óH   K  — | ]}|j                              ¦   «         V — Œd S r
   ©r7   r8   )r(   r`   s     r   ú	<genexpr>z,BNCWordView.handle_header.<locals>.<genexpr>Ü   s0   è è € Ð"JÐ"J¸% 5¤:×#3Ò#3Ñ#5Ô#5Ð"JÐ"JÐ"JÐ"JÐ"JÐ"Jr   ztitleStmt/authorc              3   óH   K  — | ]}|j                              ¦   «         V — Œd S r
   rn   )r(   ra   s     r   ro   z,BNCWordView.handle_header.<locals>.<genexpr>à   ó0   è è € Ð#NÐ#N¸F F¤K×$5Ò$5Ñ$7Ô$7Ð#NÐ#NÐ#NÐ#NÐ#NÐ#Nr   ztitleStmt/editorc              3   óH   K  — | ]}|j                              ¦   «         V — Œd S r
   rn   )r(   rb   s     r   ro   z,BNCWordView.handle_header.<locals>.<genexpr>ä   rq   r   ztitleStmt/respStmtz

c              3   óT   K  — | ]#}d                       d„ |D ¦   «         ¦  «        V — Œ$dS )rl   c              3   óH   K  — | ]}|j                              ¦   «         V — Œd S r
   rn   )r(   Úresp_elts     r   ro   z6BNCWordView.handle_header.<locals>.<genexpr>.<genexpr>é   s0   è è € ÐEÐE°H˜(œ-×-Ò-Ñ/Ô/ÐEÐEÐEÐEÐEÐEr   N)Újoin)r(   Úresps     r   ro   z,BNCWordView.handle_header.<locals>.<genexpr>è   sN   è è € ð %ð %ØJN�—	’	ÐEÐEÀÐEÑEÔEÑEÔEð%ð %ð %ð %ð %ð %r   )r5   rv   r`   ra   rb   rc   )r   rK   ÚcontextÚtitlesÚauthorsÚeditorsrc   s          r   rg   zBNCWordView.handle_headerØ   s  € à—’Ð.Ñ/Ô/ˆØð 	KØŸšÐ"JÐ"JÀ6Ð"JÑ"JÔ"JÑJÔJˆDŒJà—+’+Ð0Ñ1Ô1ˆØð 	OØŸ)š)Ð#NÐ#NÀgÐ#NÑ#NÔ#NÑNÔNˆDŒKà—+’+Ð0Ñ1Ô1ˆØð 	OØŸ)š)Ð#NÐ#NÀgÐ#NÑ#NÔ#NÑNÔNˆDŒKà—’Ð0Ñ1Ô1ˆØð 	ØŸšð %ð %ØRWð%ñ %ô %ñ ô ˆDŒJˆJˆJð	ð 	r   c                 ód   — | j         r|                      |¦  «        S |                      |¦  «        S r
   )r\   Úhandle_sentÚhandle_word)r   rK   rx   s      r   Ú
handle_eltzBNCWordView.handle_eltì   s4   € ØŒ:ð 	)Ø×#Ò# CÑ(Ô(Ð(à×#Ò# CÑ(Ô(Ð(r   c                 óL  — |j         }|sd}| j        s| j        r|                     ¦   «         }| j        r|                     d|¦  «        }| j        dk    r||                     d¦  «        f}n6| j        dk    r+||                     d|                     d¦  «        ¦  «        f}|S )Nr0   r1   r   r   )r7   r^   r_   r8   r9   r]   )r   rK   rC   s      r   r~   zBNCWordView.handle_wordò   s«   € ØŒxˆØð 	ØˆDØÔð 	  ¤
ð 	 Ø—:’:‘<”<ˆDØŒ:ð 	'Ø—7’7˜4 Ñ&Ô&ˆDØŒ9˜ÒÐØ˜#Ÿ'š' $™-œ-Ð(ˆDˆDØŒY˜%ÒÐØ˜#Ÿ'š' %¨¯ª°©¬Ñ7Ô7Ð8ˆDØˆr   c                 ó,  ‡ — g }|D ]t}|j         dv r|ˆ fd„|D ¦   «         z  }Œ|j         dv r)|                     ‰                      |¦  «        ¦  «         ŒO|j         ‰ j        vrt	          d|j         z  ¦  «        ‚Œut          |j        d         |¦  «        S )N)ÚmwÚhiÚcorrÚtruncc                 ó:   •— g | ]}‰                      |¦  «        ‘ŒS r'   )r~   )r(   rJ   r   s     €r   r+   z+BNCWordView.handle_sent.<locals>.<listcomp>  s'   ø€ Ð<Ð<Ð<°˜×)Ò)¨!Ñ,Ô,Ð<Ð<Ð<r   )rJ   rI   zUnexpected element %sr2   )r   r:   r~   Útags_to_ignoreÚ
ValueErrorr;   r<   )r   rK   r#   rL   s   `   r   r}   zBNCWordView.handle_sent   s¹   ø€ ØˆØð 	Fð 	FˆEØŒyÐ9Ð9Ð9ØÐ<Ð<Ð<Ð<°eÐ<Ñ<Ô<Ñ<��Ø”˜jÐ(Ð(Ø—’˜D×,Ò,¨UÑ3Ô3Ñ4Ô4Ð4Ð4Ø” $Ô"5Ð5Ð5Ý Ð!8¸5¼9Ñ!DÑEÔEÐEð 6å˜3œ: cœ?¨DÑ1Ô1Ð1r   N)
rD   rE   rF   rG   r‡   r   rg   r   r~   r}   r'   r   r   r,   r,   £   s€   € € € € € ðð ð	ð 	ð 	€Nðð$ð $ð $ð@ð ð ð()ð )ð )ðð ð ð	2ð 	2ð 	2ð 	2ð 	2r   r,   r
   )rG   Únltk.corpus.reader.utilr   Únltk.corpus.reader.xmldocsr   r   r   r   r6   rP   r;   r,   r'   r   r   ú<module>r‹      sí   ðð HÐ Gà *Ð *Ð *Ð *Ð *Ð *Ø RÐ RÐ RÐ RÐ RÐ RÐ RÐ RÐ RÐ Rð|ð |ð |ð |ð |�oñ |ô |ð |ð~ð ð ð ð#ð #ð #ð #ð #�$ñ #ô #ð #ðf2ð f2ð f2ð f2ð f2�-ñ f2ô f2ð f2ð f2ð f2r   