/usr/local/lib/python3.6/site-packages/sacremoses/__pycache__
NameSizeModeActions
chinese.cpython-36.pyc147100644editdlrm
cli.cpython-36.pyc65190644editdlrm
corpus.cpython-36.pyc49150644editdlrm
indic.cpython-36.pyc9490644editdlrm
normalize.cpython-36.pyc48290644editdlrm
sent_tokenize.cpython-36.pyc7210644editdlrm
subwords.cpython-36.pyc46500644editdlrm
tokenize.cpython-36.pyc165930644editdlrm
truecase.cpython-36.pyc117780644editdlrm
util.cpython-36.pyc51300644editdlrm
__init__.cpython-36.pyc2990644editdlrm
__main__.cpython-36.pyc2240644editdlrm
Edit: /usr/local/lib/python3.6/site-packages/sacremoses/__pycache__/cli.cpython-36.pyc (6519B)
3 Eg# @sddlZddlmZddlmZddlmZddlZddlmZm Z ddl m Z m Z ddl mZddlmZddlZddlZejdd krddlZejZejed ed d gd Zejdedejdddddejdddddejdddddejddddd d!ejd"d#Zeejj d$dd%krd2ddd?d0e$d@dAZ(ej&dBejdCddddDd0ejdEdFdddGd0ejdHd5dddId0ejdJd9dddKd0e$dLdMZ)ej&dNejdOdPddQdRejdSd.dddTd0ejdUd5dddVd0e$dWdXZ*ej&dYejdOdPddZdRejdSd.dddTd0ejdUd5ddd[d0e$d\d]Z+ej&d^ejd_d.ddd`d0e$dadbZ,dS)cN)deepcopy)partial)update_wrapper)MosesTokenizerMosesDetokenizer)MosesTruecaserMosesDetruecaser)MosesPunctNormalizer)parallelize_preprocesszTYou should really be using Python3!!! Tick tock, tick tock, https://pythonclock.org/z-hz--help)Zhelp_option_namesT)chainZcontext_settingsz --languagez-lenz+Use language specific rules when tokenizing)defaulthelpz --processesz-jzNo. of processes.z --encodingz-eutf8zSpecify encoding of file.z--quietz-qFzDisable progress bar.)is_flagrrcCsdS)N)languageencoding processesquietrr8/usr/local/lib/python3.6/site-packages/sacremoses/cli.pycli#s r.c Ks\tjd|dD}|}x|D]}|t|f|}qW|rNx|D]}tj|q.new_func..processor)r)rr')r&rrnew_funcEs zprocessor..new_func)r)r&rr(r)r&rr'@s r'ccsH|dkr"x:|D]}||VqWn"x t|||| dD] }|Vq6WdS)Nr) progress_bar)r )r!funcrrlineoutlinerrrparallel_or_notNs  r-tokenizez--aggressive-dash-splitsz-azTriggers dash split rules.)rrrz --xml-escapez-xz"Escape special characters for XML.z--protected-patternsz-pzXSpecify file with patters to be protected in tokenisation. Special values: :basic: :web:)rz--custom-nb-prefixesz-czjSpecify a custom non-breaking prefixes file, add prefixes to the default ones from the specified language.c Cs|t||d}|rZ|dkr |j}n:|dkr0|j}n*t|dd} dd| jD}WdQRXt|jd|||d } t|| ||S) N)langZ custom_nonbreaking_prefixes_filez:basic:z:web:r)rcSsg|] }|jqSr)strip).0patternrrr sz!tokenize_file..T) return_straggressive_dash_splitsescapeprotected_patterns)rZBASIC_PROTECTED_PATTERNSZWEB_PROTECTED_PATTERNSopen readlinesrr.r-) r!rrrZ xml_escaper5r7Zcustom_nb_prefixesmosesr Zmoses_tokenizerrr tokenize_file^s $ r; detokenizez--xml-unescapez$Unescape special characters for XML.cCs4t|d}t|jd|d}ttttj||||S)N)r/T)r4unescape)rrr<r-rmapstrsplit)r!rrrZ xml_unescaper:Zmoses_detokenizerrrdetokenize_files rA normalizez--normalize-quote-commasz Normalize quotations and commas.z--normalize-numbersz-dzNormalize number.z--replace-unicode-punctsz2Replace unicode punctuations BEFORE normalization.z--remove-control-charsz.Remove control characters AFTER normalization.c Cs*t|||||d}t|j} t|| ||S)N)Znorm_quote_commasZ norm_numbersZpre_replace_unicode_punctZpost_remove_control_chars)r rrBr-) r!rrrZnormalize_quote_commasZnormalize_numbersZreplace_unicode_punctsZremove_control_charsr:Zmoses_normalizerrrnormalize_files$ rCztrain-truecasez --modelfilez-mzFilename to save the modelfile.)requiredrz--is-asrz)A flag to indicate that model is for ASR.z--possibly-use-first-tokenz*Use the first token as part of truecasing.c Cs,t|d}|j|||| d}|j|dS)N)is_asr)possibly_use_first_tokenrr))rtrain save_model) r!rrr modelfilerErFr:modelrrrtrain_truecasers  rKtruecasez$Filename to save/load the modelfile.z1Use the first token as part of truecase training.c Csdtjj|ss       & '