Ë
    †\;j15  ã                   ó.  — d Z ddlZddlZddlmZ ddlZddlmZ g ZdZ	dZ
dZdZd	Zd
ZdZd„ Zdd„Zd„ Zd„ Z edddd¬«      dd„«       Z edddd¬«      dd„«       Z edddd¬«      dd„«       Z edddd¬«      dd„«       Z edddd¬«      d„ «       Zy)aW  
ACL2016 Multimodal Machine Translation. Please see this website for more
details: http://www.statmt.org/wmt16/multimodal-task.html#task1

If you use the dataset created for your task, please cite the following paper:
Multi30K: Multilingual English-German Image Descriptions.

@article{elliott-EtAl:2016:VL16,
 author    = {{Elliott}, D. and {Frank}, S. and {Sima"an}, K. and {Specia}, L.},
 title     = {Multi30K: Multilingual English-German Image Descriptions},
 booktitle = {Proceedings of the 6th Workshop on Vision and Language},
 year      = {2016},
 pages     = {70--74},
 year      = 2016
}
é    N)Údefaultdict)Ú
deprecatedz2http://paddlemodels.bj.bcebos.com/wmt/wmt16.tar.gzÚ 0c38be43600334966403524a40dcd81eiò+  iK  z<s>z<e>z<unk>c           	      ó  — t        t        «      }t        j                  | d¬«      5 }|j	                  d«      D ]q  }|j                  «       }|j                  «       j                  d«      }t        |«      dk7  rŒA|dk(  r|d   n|d   }|j                  «       D ]  }	||	xx   dz  cc<   Œ Œs 	 d d d «       t        |d	«      5 }
|
j                  t        › d
t        › d
t        › d
�j                  «       «       t        t        |j!                  «       d„ d¬«      «      D ]B  \  }}|dz   |k(  r n5|
j                  |d   j                  «       «       |
j                  d«       ŒD d d d «       y # 1 sw Y   Œ¾xY w# 1 sw Y   y xY w)NÚr©Úmodeúwmt16/trainÚ	é   Úenr   é   ÚwbÚ
c                 ó   — | d   S )Nr   © )Úxs    ú]G:\00. PROJECTS\API\Inventory\templateJSON\kerjaOCR\Lib\site-packages\paddle/dataset/wmt16.pyÚ<lambda>z__build_dict.<locals>.<lambda>B   s   € °A°a²Dó    T)ÚkeyÚreverseé   ó   
)r   ÚintÚtarfileÚopenÚextractfileÚdecodeÚstripÚsplitÚlenÚwriteÚ
START_MARKÚEND_MARKÚUNK_MARKÚencodeÚ	enumerateÚsortedÚitems)Útar_fileÚ	dict_sizeÚ	save_pathÚlangÚ	word_dictÚfÚlineÚ
line_splitÚsenÚwÚfoutÚidxÚwords                r   Ú__build_dictr8   3   sA  € ÜœCÓ €IÜ	�‰�h SÕ	)¨QØ—M‘M -Ö0ˆDØ—;‘;“=ˆDØŸ™›×+Ñ+¨DÓ1ˆJÜ�:‹ !Ò#ØØ#'¨4¢<�*˜Q’-°ZÀ±]ˆCØ—Y‘Y–[�Ø˜!“ Ñ!”ñ !ñ 1÷ 
*ô 
ˆi˜Ô	 $Ø�
‰
”z�l "¤X J¨b´°
¸"Ð=×EÑEÓGÔHÜ"Ü�9—?‘?Ó$©.À$ÔGö
‰IˆC�ð �Q‰w˜)Ò#ÙØ�J‰J�t˜A‘w—~‘~Ó'Ô(Ø�J‰J�uÕð
÷ 
Ð	÷ 
*Ð	)ú÷ 
Ð	ús   §BE+ÃB E7Å+E4Å7F c                 ó4  — t         j                  j                  t        j                  j
                  j                  d||fz  «      }t         j                  j                  |«      r&t        t        |d«      j                  «       «      |k7  rt        | |||«       i }t        |d«      5 }t        |«      D ]J  \  }}|r"|j                  «       j                  «       ||<   Œ*|||j                  «       j                  «       <   ŒL 	 d d d «       |S # 1 sw Y   |S xY w)Núwmt16/%s_%d.dictÚrb)ÚosÚpathÚjoinÚpaddleÚdatasetÚcommonÚ	DATA_HOMEÚexistsr"   r   Ú	readlinesr8   r(   r    r   )	r+   r,   r.   r   Ú	dict_pathr/   Úfdictr6   r1   s	            r   Ú__load_dictrG   J   sä   € Ü—‘—‘Ü�‰×Ñ×'Ñ'Ð);¸tÀYÐ>OÑ)Oó€Iô �7‰7�>‰>˜)Ô$ÜŒD�˜DÓ!×+Ñ+Ó-Ó.°)Ò;ä�X˜y¨)°TÔ:à€IÜ	ˆi˜Ô	 %Ü" 5Ö)‰IˆC�ÙØ!%§¡£×!4Ñ!4Ó!6�	˜#’à36�	˜$Ÿ*™*›,×-Ñ-Ó/Ò0ñ	 *÷ 
ð Ð÷ 
ð Ðús   Â)ADÄDc                 óv   — t        | |dk(  rt        nt        «      } t        ||dk(  rt        nt        «      }| |fS )Nr   )ÚminÚTOTAL_EN_WORDSÚTOTAL_DE_WORDS©Úsrc_dict_sizeÚtrg_dict_sizeÚsrc_langs      r   Ú__get_dict_sizerP   ]   sA   € ÜØ¨(°dÒ*:�Äó€Mô Ø¨(°dÒ*:�Äó€Mð ˜-Ð'Ð'r   c                 ó"   ‡ ‡‡‡‡— ˆˆˆˆ ˆfd„}|S )Nc            
   3   ó¸  •K  — t        ‰‰‰«      } t        ‰‰‰dk(  rdnd«      }| t           }| t           }| t           }‰dk(  rdnd}d|z
  }t	        j
                  ‰d¬«      5 }|j                  ‰«      D ]À  }|j                  «       }|j                  «       j                  d«      }	t        |	«      dk7  rŒA|	|   j                  «       }
|g|
D �cg c]  }| j                  ||«      ‘Œ c}z   |gz   }|	|   j                  «       }|D �cg c]  }|j                  ||«      ‘Œ }}||gz   }|g|z   }|||f–— ŒÂ 	 d d d «       y c c}w c c}w # 1 sw Y   y xY w­w)	Nr   Úder   r   r   r   r   r   )rG   r$   r%   r&   r   r   r   r   r    r!   r"   Úget)Úsrc_dictÚtrg_dictÚstart_idÚend_idÚunk_idÚsrc_colÚtrg_colr0   r1   r2   Ú	src_wordsr4   Úsrc_idsÚ	trg_wordsÚtrg_idsÚtrg_ids_nextÚ	file_namerM   rO   r+   rN   s                   €€€€€r   Úreaderzreader_creator.<locals>.readerh   sx  øè ø€ Ü˜x¨¸ÓAˆÜØ�m¨h¸$Ò.>¡dÀDó
ˆð œJÑ'ˆØœ(Ñ#ˆØœ(Ñ#ˆà 4Ò'‘!¨QˆØ�g‘+ˆä�\‰\˜(¨Õ-°ØŸ™ iÖ0�Ø—{‘{“}�Ø!ŸZ™Z›\×/Ñ/°Ó5�
Ü�z“? aÒ'ØØ& wÑ/×5Ñ5Ó7�	à�JÙ8AÓB¹	°1�x—|‘| A vÕ.¸	ÑBñCà�hñð ð ' wÑ/×5Ñ5Ó7�	Ù<EÓF¹I°q˜8Ÿ<™<¨¨6Õ2¸I�ÐFà&¨&¨Ñ1�Ø#˜* wÑ.�à˜w¨Ð4Ó4ñ% 1÷ .Ð-ùò Cùò
 G÷ .Ð-üs=   ƒA"EÁ%A,EÃE
Ã* EÄ
E	Ä#EÄ;	EÅ
EÅEÅEr   )r+   ra   rM   rN   rO   rb   s   ````` r   Úreader_creatorrc   g   s   ü€ ÷#5ð #5ðJ €Mr   z2.0.0zpaddle.text.datasets.WMT16r   z>Please use new dataset API which supports paddle.io.DataLoader)ÚsinceÚ	update_toÚlevelÚreasonc                 óÄ   — |dvrt        d«      ‚t        | ||«      \  } }t        t        j                  j
                  j                  t        dt        d«      d| ||¬«      S )a}  
    WMT16 train set reader.

    This function returns the reader for train data. Each sample the reader
    returns is made up of three fields: the source language word index sequence,
    target language word index sequence and next word index sequence.


    NOTE:
    The original like for training data is:
    http://www.quest.dcs.shef.ac.uk/wmt16_files_mmt/training.tar.gz

    paddle.dataset.wmt16 provides a tokenized version of the original dataset by
    using moses's tokenization script:
    https://github.com/moses-smt/mosesdecoder/blob/master/scripts/tokenizer/tokenizer.perl

    Args:
        src_dict_size(int): Size of the source language dictionary. Three
                            special tokens will be added into the dictionary:
                            <s> for start mark, <e> for end mark, and <unk> for
                            unknown word.
        trg_dict_size(int): Size of the target language dictionary. Three
                            special tokens will be added into the dictionary:
                            <s> for start mark, <e> for end mark, and <unk> for
                            unknown word.
        src_lang(string): A string indicating which language is the source
                          language. Available options are: "en" for English
                          and "de" for Germany.

    Returns:
        callable: The train reader.
    ©r   rS   zIAn error language type.  Only support: en (for English); de(for Germany).Úwmt16úwmt16.tar.gzr
   ©r+   ra   rM   rN   rO   ©	Ú
ValueErrorrP   rc   r?   r@   rA   ÚdownloadÚDATA_URLÚDATA_MD5rL   s      r   Útrainrr   �   su   € ðP �|Ñ#Üð1ó
ð 	
ô $3Ø�} hó$Ñ €M�=ô Ü—‘×&Ñ&×/Ñ/Ü�gœx¨ó
ð  Ø#Ø#Øôð r   c                 óÄ   — |dvrt        d«      ‚t        | ||«      \  } }t        t        j                  j
                  j                  t        dt        d«      d| ||¬«      S )a}  
    WMT16 test set reader.

    This function returns the reader for test data. Each sample the reader
    returns is made up of three fields: the source language word index sequence,
    target language word index sequence and next word index sequence.

    NOTE:
    The original like for test data is:
    http://www.quest.dcs.shef.ac.uk/wmt16_files_mmt/mmt16_task1_test.tar.gz

    paddle.dataset.wmt16 provides a tokenized version of the original dataset by
    using moses's tokenization script:
    https://github.com/moses-smt/mosesdecoder/blob/master/scripts/tokenizer/tokenizer.perl

    Args:
        src_dict_size(int): Size of the source language dictionary. Three
                            special tokens will be added into the dictionary:
                            <s> for start mark, <e> for end mark, and <unk> for
                            unknown word.
        trg_dict_size(int): Size of the target language dictionary. Three
                            special tokens will be added into the dictionary:
                            <s> for start mark, <e> for end mark, and <unk> for
                            unknown word.
        src_lang(string): A string indicating which language is the source
                          language. Available options are: "en" for English
                          and "de" for Germany.

    Returns:
        callable: The test reader.
    ri   úHAn error language type. Only support: en (for English); de(for Germany).rj   rk   z
wmt16/testrl   rm   rL   s      r   Útestru   Ì   su   € ðN �|Ñ#Üð?ó
ð 	
ô
 $3Ø�} hó$Ñ €M�=ô Ü—‘×&Ñ&×/Ñ/Ü�gœx¨ó
ð Ø#Ø#Øôð r   c                 óÄ   — |dvrt        d«      ‚t        | ||«      \  } }t        t        j                  j
                  j                  t        dt        d«      d| ||¬«      S )a�  
    WMT16 validation set reader.

    This function returns the reader for validation data. Each sample the reader
    returns is made up of three fields: the source language word index sequence,
    target language word index sequence and next word index sequence.

    NOTE:
    The original like for validation data is:
    http://www.quest.dcs.shef.ac.uk/wmt16_files_mmt/validation.tar.gz

    paddle.dataset.wmt16 provides a tokenized version of the original dataset by
    using moses's tokenization script:
    https://github.com/moses-smt/mosesdecoder/blob/master/scripts/tokenizer/tokenizer.perl

    Args:
        src_dict_size(int): Size of the source language dictionary. Three
                            special tokens will be added into the dictionary:
                            <s> for start mark, <e> for end mark, and <unk> for
                            unknown word.
        trg_dict_size(int): Size of the target language dictionary. Three
                            special tokens will be added into the dictionary:
                            <s> for start mark, <e> for end mark, and <unk> for
                            unknown word.
        src_lang(string): A string indicating which language is the source
                          language. Available options are: "en" for English
                          and "de" for Germany.

    Returns:
        callable: The validation reader.
    ri   rt   rj   rk   z	wmt16/valrl   rm   rL   s      r   Ú
validationrw     su   € ðL �|Ñ#Üð?ó
ð 	
ô $3Ø�} hó$Ñ €M�=ô Ü—‘×&Ñ&×/Ñ/Ü�gœx¨ó
ð Ø#Ø#Øôð r   c                 óÌ  — | dk(  rt        |t        «      }nt        |t        «      }t        j                  j                  t        j                  j                  j                  d| |fz  «      }t        j                  j                  |«      sJ d«       ‚	 	 t        j                  j                  t        j                  j                  j                  d«      }t        ||| |«      S )a»  
    return the word dictionary for the specified language.

    Args:
        lang(string): A string indicating which language is the source
                      language. Available options are: "en" for English
                      and "de" for Germany.
        dict_size(int): Size of the specified language dictionary.
        reverse(bool): If reverse is set to False, the returned python
                       dictionary will use word as key and use index as value.
                       If reverse is set to True, the returned python
                       dictionary will use index as key and word as value.

    Returns:
        dict: The word dictionary for the specific language.
    r   r:   z Word dictionary does not exist. rk   )rI   rJ   rK   r<   r=   r>   r?   r@   rA   rB   rC   rG   )r.   r,   r   rE   r+   s        r   Úget_dictry   B  s®   € ð0 ˆt‚|Ü˜	¤>Ó2‰	ä˜	¤>Ó2ˆ	ä—‘—‘Ü�‰×Ñ×'Ñ'Ð);¸tÀYÐ>OÑ)Oó€Iô �7‰7�>‰>˜)Ô$ÐHÐ&HÓHÐ$ØEØÜ�w‰w�|‰|œFŸN™N×1Ñ1×;Ñ;¸^ÓL€HÜ�x ¨D°'Ó:Ð:r   c                  ó€   — t         j                  j                  j                  j	                  t
        dt        d«       y)zdownload the entire dataset.rj   rk   N)r?   Úv4r@   rA   ro   rp   rq   r   r   r   Úfetchr|   i  s+   € ô ‡I�I×Ñ×Ñ×%Ñ%Ü�'œ8 ^õr   )F)r   )Ú__doc__r<   r   Úcollectionsr   r?   Úpaddle.utilsr   Ú__all__rp   rq   rJ   rK   r$   r%   r&   r8   rG   rP   rc   rr   ru   rw   ry   r|   r   r   r   Ú<module>r�      s  ðñó" 
Û Ý #ã Ý #à
€à?€Ø-€à€Ø€à€
Ø€Ø€òó.ò&(ò&ñR Ø
Ø*Ø
ØKô	ò3óð3ñl Ø
Ø*Ø
ØKô	ò3óð3ñl Ø
Ø*Ø
ØKô	ò1óð1ñh Ø
Ø*Ø
ØKô	ò;óð;ñB Ø
Ø*Ø
ØKô	ñóñr   