§
    'ê[f5   ã                   ón   — d dl Z d dlZd dlZd dlZd dlZd dlmZ d dlmZ d dl	m
Z
  G d„ de
¦  «        ZdS )é    N)ÚZipFilePathPointer)Úfind_dir)Ú
TokenizerIc                   ó`   — e Zd ZdZdd„Zd„ Zdd„Zd„ Zed„ ¦   «         Z	ed	„ ¦   «         Z
d
„ ZdS )ÚReppTokenizeraé  
    A class for word tokenization using the REPP parser described in
    Rebecca Dridan and Stephan Oepen (2012) Tokenization: Returning to a
    Long Solved Problem - A Survey, Contrastive  Experiment, Recommendations,
    and Toolkit. In ACL. http://anthology.aclweb.org/P/P12/P12-2.pdf#page=406

    >>> sents = ['Tokenization is widely regarded as a solved problem due to the high accuracy that rulebased tokenizers achieve.' ,
    ... 'But rule-based tokenizers are hard to maintain and their rules language specific.' ,
    ... 'We evaluated our method on three languages and obtained error rates of 0.27% (English), 0.35% (Dutch) and 0.76% (Italian) for our best models.'
    ... ]
    >>> tokenizer = ReppTokenizer('/home/alvas/repp/') # doctest: +SKIP
    >>> for sent in sents:                             # doctest: +SKIP
    ...     tokenizer.tokenize(sent)                   # doctest: +SKIP
    ...
    (u'Tokenization', u'is', u'widely', u'regarded', u'as', u'a', u'solved', u'problem', u'due', u'to', u'the', u'high', u'accuracy', u'that', u'rulebased', u'tokenizers', u'achieve', u'.')
    (u'But', u'rule-based', u'tokenizers', u'are', u'hard', u'to', u'maintain', u'and', u'their', u'rules', u'language', u'specific', u'.')
    (u'We', u'evaluated', u'our', u'method', u'on', u'three', u'languages', u'and', u'obtained', u'error', u'rates', u'of', u'0.27', u'%', u'(', u'English', u')', u',', u'0.35', u'%', u'(', u'Dutch', u')', u'and', u'0.76', u'%', u'(', u'Italian', u')', u'for', u'our', u'best', u'models', u'.')

    >>> for sent in tokenizer.tokenize_sents(sents): # doctest: +SKIP
    ...     print(sent)                              # doctest: +SKIP
    ...
    (u'Tokenization', u'is', u'widely', u'regarded', u'as', u'a', u'solved', u'problem', u'due', u'to', u'the', u'high', u'accuracy', u'that', u'rulebased', u'tokenizers', u'achieve', u'.')
    (u'But', u'rule-based', u'tokenizers', u'are', u'hard', u'to', u'maintain', u'and', u'their', u'rules', u'language', u'specific', u'.')
    (u'We', u'evaluated', u'our', u'method', u'on', u'three', u'languages', u'and', u'obtained', u'error', u'rates', u'of', u'0.27', u'%', u'(', u'English', u')', u',', u'0.35', u'%', u'(', u'Dutch', u')', u'and', u'0.76', u'%', u'(', u'Italian', u')', u'for', u'our', u'best', u'models', u'.')
    >>> for sent in tokenizer.tokenize_sents(sents, keep_token_positions=True): # doctest: +SKIP
    ...     print(sent)                                                         # doctest: +SKIP
    ...
    [(u'Tokenization', 0, 12), (u'is', 13, 15), (u'widely', 16, 22), (u'regarded', 23, 31), (u'as', 32, 34), (u'a', 35, 36), (u'solved', 37, 43), (u'problem', 44, 51), (u'due', 52, 55), (u'to', 56, 58), (u'the', 59, 62), (u'high', 63, 67), (u'accuracy', 68, 76), (u'that', 77, 81), (u'rulebased', 82, 91), (u'tokenizers', 92, 102), (u'achieve', 103, 110), (u'.', 110, 111)]
    [(u'But', 0, 3), (u'rule-based', 4, 14), (u'tokenizers', 15, 25), (u'are', 26, 29), (u'hard', 30, 34), (u'to', 35, 37), (u'maintain', 38, 46), (u'and', 47, 50), (u'their', 51, 56), (u'rules', 57, 62), (u'language', 63, 71), (u'specific', 72, 80), (u'.', 80, 81)]
    [(u'We', 0, 2), (u'evaluated', 3, 12), (u'our', 13, 16), (u'method', 17, 23), (u'on', 24, 26), (u'three', 27, 32), (u'languages', 33, 42), (u'and', 43, 46), (u'obtained', 47, 55), (u'error', 56, 61), (u'rates', 62, 67), (u'of', 68, 70), (u'0.27', 71, 75), (u'%', 75, 76), (u'(', 77, 78), (u'English', 78, 85), (u')', 85, 86), (u',', 86, 87), (u'0.35', 88, 92), (u'%', 92, 93), (u'(', 94, 95), (u'Dutch', 95, 100), (u')', 100, 101), (u'and', 102, 105), (u'0.76', 106, 110), (u'%', 110, 111), (u'(', 112, 113), (u'Italian', 113, 120), (u')', 120, 121), (u'for', 122, 125), (u'our', 126, 129), (u'best', 130, 134), (u'models', 135, 141), (u'.', 141, 142)]
    Úutf8c                 óx   — |                       |¦  «        | _        t          j        ¦   «         | _        || _        d S )N)Úfind_repptokenizerÚrepp_dirÚtempfileÚ
gettempdirÚworking_dirÚencoding)Úselfr   r   s      úF/var/www/piapp/venv/lib/python3.11/site-packages/nltk/tokenize/repp.pyÚ__init__zReppTokenizer.__init__6   s3   € Ø×/Ò/°Ñ9Ô9ˆŒå#Ô.Ñ0Ô0ˆÔà ˆŒˆˆó    c                 óH   — t          |                      |g¦  «        ¦  «        S )zÈ
        Use Repp to tokenize a single sentence.

        :param sentence: A single sentence string.
        :type sentence: str
        :return: A tuple of tokens.
        :rtype: tuple(str)
        )ÚnextÚtokenize_sents)r   Úsentences     r   ÚtokenizezReppTokenizer.tokenize=   s"   € õ �D×'Ò'¨¨
Ñ3Ô3Ñ4Ô4Ð4r   Fc              #   óü  K  — t          j        d| j        dd¬¦  «        5 }|D ]'}|                     t	          |¦  «        dz   ¦  «         Œ(|                     ¦   «          |                      |j        ¦  «        }|                      |¦  «         	                    | j
        ¦  «                             ¦   «         }|                      |¦  «        D ]}|st          |Ž \  }}}	|V — Œ	 ddd¦  «         dS # 1 swxY w Y   dS )zà
        Tokenize multiple sentences using Repp.

        :param sentences: A list of sentence strings.
        :type sentences: list(str)
        :return: A list of tuples of tokens
        :rtype: iter(tuple(str))
        zrepp_input.ÚwF)ÚprefixÚdirÚmodeÚdeleteÚ
N)r   ÚNamedTemporaryFiler   ÚwriteÚstrÚcloseÚgenerate_repp_commandÚnameÚ_executeÚdecoder   ÚstripÚparse_repp_outputsÚzip)
r   Ú	sentencesÚkeep_token_positionsÚ
input_fileÚsentÚcmdÚrepp_outputÚtokenized_sentÚstartsÚendss
             r   r   zReppTokenizer.tokenize_sentsH   s^  è è € õ Ô(Ø  dÔ&6¸SÈð
ñ 
ô 
ð 	%àà!ð 3ð 3�Ø× Ò ¥ T¡¤¨TÑ!1Ñ2Ô2Ð2Ð2Ø×ÒÑÔÐà×,Ò,¨Z¬_Ñ=Ô=ˆCàŸ-š-¨Ñ,Ô,×3Ò3°D´MÑBÔB×HÒHÑJÔJˆKØ"&×"9Ò"9¸+Ñ"FÔ"Fð %ð %�Ø+ð Hå36¸Ð3GÑ0�N F¨DØ$Ð$Ð$Ð$Ð$ð	%ð	%ð 	%ð 	%ñ 	%ô 	%ð 	%ð 	%ð 	%ð 	%ð 	%ð 	%ð 	%øøøð 	%ð 	%ð 	%ð 	%ð 	%ð 	%s    CC1Ã1C5Ã8C5c                 óT   — | j         dz   g}|d| j         dz   gz  }|ddgz  }||gz  }|S )z«
        This module generates the REPP command to be used at the terminal.

        :param inputfilename: path to the input file
        :type inputfilename: str
        ú	/src/reppz-cú/erg/repp.setz--formatÚtriple)r   )r   Úinputfilenamer/   s      r   r$   z#ReppTokenizer.generate_repp_commandb   sI   € ð Œ}˜{Ñ*Ð+ˆØ��d”m oÑ5Ð6Ñ6ˆØ�
˜HÐ%Ñ%ˆØ�ˆÑˆØˆ
r   c                 óŠ   — t          j        | t           j        t           j        ¬¦  «        }|                     ¦   «         \  }}|S )N)ÚstdoutÚstderr)Ú
subprocessÚPopenÚPIPEÚcommunicate)r/   Úpr:   r;   s       r   r&   zReppTokenizer._executeo   s3   € åÔ˜S­¬ÅÄÐQÑQÔQˆØŸš™œ‰ˆ�Øˆr   c              #   óð   K  — t          j        dt           j        ¦  «        }|                      d¦  «        D ]>}d„ |                     |¦  «        D ¦   «         }t          d„ |D ¦   «         ¦  «        }|V — Œ?dS )aZ  
        This module parses the tri-tuple format that REPP outputs using the
        "--format triple" option and returns an generator with tuple of string
        tokens.

        :param repp_output:
        :type repp_output: type
        :return: an iterable of the tokenized sentences as tuples of strings
        :rtype: iter(tuple)
        z^\((\d+), (\d+), (.+)\)$z

c                 óT   — g | ]%\  }}}|t          |¦  «        t          |¦  «        f‘Œ&S © )Úint)Ú.0ÚstartÚendÚtokens       r   ú
<listcomp>z4ReppTokenizer.parse_repp_outputs.<locals>.<listcomp>ƒ   sA   € ð $ð $ð $á%�E˜3 ð �˜E™
œ
¥C¨¡H¤HÐ-ð$ð $ð $r   c              3   ó&   K  — | ]}|d          V — ŒdS )é   NrC   )rE   Úts     r   ú	<genexpr>z3ReppTokenizer.parse_repp_outputs.<locals>.<genexpr>‡   s&   è è € Ð=Ð= 1˜!˜Aœ$Ð=Ð=Ð=Ð=Ð=Ð=r   N)ÚreÚcompileÚ	MULTILINEÚsplitÚfindallÚtuple)r0   Ú
line_regexÚsectionÚwords_with_positionsÚwordss        r   r)   z ReppTokenizer.parse_repp_outputsu   sž   è è € õ ”ZÐ ;½R¼\ÑJÔJˆ
Ø"×(Ò(¨Ñ0Ô0ð 	'ð 	'ˆGð$ð $à)3×);Ò);¸GÑ)DÔ)Dð$ñ $ô $Ð õ Ð=Ð=Ð(<Ð=Ñ=Ô=Ñ=Ô=ˆEØ&Ð&Ð&Ð&Ð&ð	'ð 	'r   c                 óü   — t           j                             |¦  «        r|}nt          |d¬¦  «        }t           j                             |dz   ¦  «        sJ ‚t           j                             |dz   ¦  «        sJ ‚|S )zX
        A module to find REPP tokenizer binary and its *repp.set* config file.
        )ÚREPP_TOKENIZER)Úenv_varsr5   r6   )ÚosÚpathÚexistsr   )r   Úrepp_dirnameÚ	_repp_dirs      r   r
   z ReppTokenizer.find_repptokenizerŠ   sy   € õ Œ7�>Š>˜,Ñ'Ô'ð 	MØ$ˆIˆIå  Ð8KÐLÑLÔLˆIåŒw�~Š~˜i¨+Ñ5Ñ6Ô6Ð6Ð6Ð6ÝŒw�~Š~˜i¨/Ñ9Ñ:Ô:Ð:Ð:Ð:ØÐr   N)r   )F)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r   r$   Ústaticmethodr&   r)   r
   rC   r   r   r   r      sª   € € € € € ðð ð@!ð !ð !ð !ð	5ð 	5ð 	5ð%ð %ð %ð %ð4ð ð ð ðð ñ „\ðð
 ð'ð 'ñ „\ð'ð(ð ð ð ð r   r   )r[   rN   r<   Úsysr   Ú	nltk.datar   Únltk.internalsr   Únltk.tokenize.apir   r   rC   r   r   ú<module>ri      s«   ðð 
€	€	€	Ø 	€	€	€	Ø Ð Ð Ð Ø 
€
€
€
Ø €€€à (Ð (Ð (Ð (Ð (Ð (Ø #Ð #Ð #Ð #Ð #Ð #Ø (Ð (Ð (Ð (Ð (Ð (ð@ð @ð @ð @ð @�Jñ @ô @ð @ð @ð @r   