U
    Á½„bOO  ã                   @   sŠ  d dl Z d dlmZmZ d dlmZmZmZmZ zd dl	m
Z
 W n ek
rX   eZ
Y nX ddlmZmZmZmZ ddlmZmZmZmZ ddlmZ dd	lmZmZ dd
lmZmZmZm Z m!Z!m"Z" e  #d¡Z$e  %¡ Z&e& 'e  (d¡¡ de)e*e*e+ee ee e,e,edœ	dd„Z-dee*e*e+ee ee e,e,edœ	dd„Z.d e
e*e*e+ee ee e,e,edœ	dd„Z/d!e
e*e*e+ee ee e,edœdd„Z0dS )"é    N)ÚbasenameÚsplitext)ÚBinaryIOÚListÚOptionalÚSet)ÚPathLikeé   )Úcoherence_ratioÚencoding_languagesÚmb_encoding_languagesÚmerge_coherence_ratios)ÚIANA_SUPPORTEDÚTOO_BIG_SEQUENCEÚTOO_SMALL_SEQUENCEÚTRACE)Ú
mess_ratio)ÚCharsetMatchÚCharsetMatches)Úany_specified_encodingÚ	iana_nameÚidentify_sig_or_bomÚis_cp_similarÚis_multi_byte_encodingÚshould_strip_sig_or_bomZcharset_normalizerz)%(asctime)s | %(levelname)s | %(message)sé   é   çš™™™™™É?TF)	Ú	sequencesÚstepsÚ
chunk_sizeÚ	thresholdÚcp_isolationÚcp_exclusionÚpreemptive_behaviourÚexplainÚreturnc           1      C   s°	  t | ttfƒs td t| ƒ¡ƒ‚|r>tj}t t	¡ t 
t¡ t| ƒ}	|	dkrŽt d¡ |rvt t	¡ t 
|prtj¡ tt| dddg dƒgƒS |dk	rºt td	d
 |¡¡ dd„ |D ƒ}ng }|dk	rêt tdd
 |¡¡ dd„ |D ƒ}ng }|	|| k�rt td|||	¡ d}|	}|dk�r:|	| |k �r:t|	| ƒ}t| ƒtk }
t| ƒtk}|
�rlt td |	¡¡ n|�r„t td |	¡¡ g }|�r–t| ƒnd}|dk	�r¼| |¡ t td|¡ tƒ }g }g }d}d}d}tƒ }t| ƒ\}}|dk	�r| |¡ t tdt|ƒ|¡ | d¡ d|k�r.| d¡ |t D �]z}|�rP||k�rP�q6|�rd||k�rd�q6||k�rr�q6| |¡ d}||k}|�o”t|ƒ}|dk�r¸|�s¸t td|¡ �q6zt|ƒ}W n. t t!fk
�rò   t td|¡ Y �q6Y nX zr|�r>|dk�r>t"|dk�r"| dtdƒ… n| t|ƒtdƒ… |d� n&t"|dk�rN| n| t|ƒd… |d�}W n\ t#t$fk
�rÂ } z8t |t$ƒ�sžt td|t"|ƒ¡ | |¡ W Y ¢�q6W 5 d}~X Y nX d}|D ]}t%||ƒ�rÌd} �qê�qÌ|�rt td||¡ �q6t&|�sdnt|ƒ|	t|	| ƒƒ}|�o@|dk	�o@t|ƒ|	k } | �rVt td|¡ tt|ƒd ƒ}!t'|!d ƒ}!d}"d}#g }$g }%|D �]¤}&|&| |	d! k�r �q„| |&|&| … }'|�rÈ|dk�rÈ||' }'z|'j(||�rÚd"nd#d$�}(W nL t#k
�r0 } z,t td%|t"|ƒ¡ |!}"d}#W Y ¢
 �q,W 5 d}~X Y nX |�rØ|&dk�rØ| |& d&k�rØt)|d'ƒ})|�rØ|(d|)… |k�rØt&|&|&d d(ƒD ]T}*| |*|&| … }'|�r®|dk�r®||' }'|'j(|d"d$�}(|(d|)… |k�r‚ �qØ�q‚|$ |(¡ |% t*|(|ƒ¡ |%d( |k�r|"d7 }"|"|!k�s"|�r„|dk�r„ �q,�q„|#�sª|�rª|�sªz| td)ƒd… j(|d#d$� W nL t#k
�r¨ } z,t td*|t"|ƒ¡ | |¡ W Y ¢�q6W 5 d}~X Y nX |%�rÀt+|%ƒt|%ƒ nd}+|+|k�sØ|"|!k�rP| |¡ t td+||"t,|+d, d-d.�¡ |dd|fk�r6|#�s6t| ||dg |ƒ},||k�r8|,}n|dk�rH|,}n|,}�q6t td/|t,|+d, d-d.�¡ |�s|t-|ƒ}-nt.|ƒ}-|-�r¢t td0 |t"|-ƒ¡¡ g }.|dk�râ|$D ],}(t/|(d1|-�rÎd2 |-¡ndƒ}/|. |/¡ �q´t0|.ƒ}0|0�rt td3 |0|¡¡ | t| ||+||0|ƒ¡ ||ddfk�rn|+d1k �rnt d4|¡ |�r\t t	¡ t 
|¡ t|| gƒ  S ||k�r6t d5|¡ |�ržt t	¡ t 
|¡ t|| gƒ  S �q6t|ƒdk�	rd|�sÔ|�sÔ|�ràt td6¡ |�	r t d7|j1¡ | |¡ nd|�	r|dk�	s4|�	r*|�	r*|j2|j2k�	s4|dk	�	rJt d8¡ | |¡ n|�	rdt d9¡ | |¡ |�	rˆt d:| 3¡ j1t|ƒd ¡ n
t d;¡ |�	r¬t t	¡ t 
|¡ |S )<ae  
    Given a raw bytes sequence, return the best possibles charset usable to render str objects.
    If there is no results, it is a strong indicator that the source is binary/not text.
    By default, the process will extract 5 blocs of 512o each to assess the mess and coherence of a given sequence.
    And will give up a particular code page after 20% of measured mess. Those criteria are customizable at will.

    The preemptive behavior DOES NOT replace the traditional detection workflow, it prioritize a particular code page
    but never take it for granted. Can improve the performance.

    You may want to focus your attention to some code page or/and not others, use cp_isolation and cp_exclusion for that
    purpose.

    This function will strip the SIG in the payload/sequence every time except on UTF-16, UTF-32.
    By default the library does not setup any handler other than the NullHandler, if you choose to set the 'explain'
    toggle to True it will alter the logger configuration to add a StreamHandler that is suitable for debugging.
    Custom logging format and handler can be set manually.
    z4Expected object of type bytes or bytearray, got: {0}r   z<Encoding detection on empty bytes, assuming utf_8 intention.Úutf_8g        FÚ Nz`cp_isolation is set. use this flag for debugging purpose. limited list of encoding allowed : %s.z, c                 S   s   g | ]}t |d ƒ‘qS ©F©r   ©Ú.0Úcp© r.   ú:/tmp/pip-unpacked-wheel-2ta4nrol/charset_normalizer/api.pyÚ
<listcomp>]   s     zfrom_bytes.<locals>.<listcomp>zacp_exclusion is set. use this flag for debugging purpose. limited list of encoding excluded : %s.c                 S   s   g | ]}t |d ƒ‘qS r)   r*   r+   r.   r.   r/   r0   h   s     z^override steps (%i) and chunk_size (%i) as content does not fit (%i byte(s) given) parameters.r	   z>Trying to detect encoding from a tiny portion of ({}) byte(s).zIUsing lazy str decoding because the payload is quite large, ({}) byte(s).z@Detected declarative mark in sequence. Priority +1 given for %s.zIDetected a SIG or BOM mark on first %i byte(s). Priority +1 given for %s.Úascii>   Úutf_32Úutf_16z[Encoding %s wont be tested as-is because it require a BOM. Will try some sub-encoder LE/BE.z2Encoding %s does not provide an IncrementalDecoderg    €„A)Úencodingz9Code page %s does not fit given bytes sequence at ALL. %sTzW%s is deemed too similar to code page %s and was consider unsuited already. Continuing!zpCode page %s is a multi byte encoding table and it appear that at least one character was encoded using n-bytes.é   é   é   ÚignoreÚstrict)ÚerrorszaLazyStr Loading: After MD chunk decode, code page %s does not fit given bytes sequence at ALL. %sé€   é   éÿÿÿÿg     jè@z^LazyStr Loading: After final lookup, code page %s does not fit given bytes sequence at ALL. %szc%s was excluded because of initial chaos probing. Gave up %i time(s). Computed mean chaos is %f %%.éd   é   )Úndigitsz=%s passed initial chaos probing. Mean measured chaos is %f %%z&{} should target any language(s) of {}gš™™™™™¹?ú,z We detected language {} using {}z.Encoding detection: %s is most likely the one.zoEncoding detection: %s is most likely the one as we detected a BOM or SIG within the beginning of the sequence.zONothing got out of the detection process. Using ASCII/UTF-8/Specified fallback.z7Encoding detection: %s will be used as a fallback matchz:Encoding detection: utf_8 will be used as a fallback matchz:Encoding detection: ascii will be used as a fallback matchz]Encoding detection: Found %s as plausible (best-candidate) for content. With %i alternatives.z=Encoding detection: Unable to determine any suitable charset.)4Ú
isinstanceÚ	bytearrayÚbytesÚ	TypeErrorÚformatÚtypeÚloggerÚlevelÚ
addHandlerÚexplain_handlerÚsetLevelr   ÚlenÚdebugÚremoveHandlerÚloggingÚWARNINGr   r   ÚlogÚjoinÚintr   r   r   ÚappendÚsetr   r   Úaddr   r   ÚModuleNotFoundErrorÚImportErrorÚstrÚUnicodeDecodeErrorÚLookupErrorr   ÚrangeÚmaxÚdecodeÚminr   ÚsumÚroundr   r   r
   r   r4   ÚfingerprintÚbest)1r   r   r    r!   r"   r#   r$   r%   Zprevious_logger_levelÚlengthZis_too_small_sequenceZis_too_large_sequenceZprioritized_encodingsZspecified_encodingZtestedZtested_but_hard_failureZtested_but_soft_failureZfallback_asciiZfallback_u8Zfallback_specifiedÚresultsZsig_encodingZsig_payloadZencoding_ianaZdecoded_payloadZbom_or_sig_availableZstrip_sig_or_bomZis_multi_byte_decoderÚeZsimilar_soft_failure_testZencoding_soft_failedZr_Zmulti_byte_bonusZmax_chunk_gave_upZearly_stop_countZlazy_str_hard_failureZ	md_chunksZ	md_ratiosÚiZcut_sequenceÚchunkZchunk_partial_size_chkÚjZmean_mess_ratioZfallback_entryZtarget_languagesZ	cd_ratiosZchunk_languagesZcd_ratios_mergedr.   r.   r/   Ú
from_bytes%   sì   ÿÿ



üüûÿþÿþÿ

ý

ü




ÿýýÿüÿü
ü

ü
ýÿ
ýü

þ
ü
ÿþ


ÿÿ
ÿþýü
ÿ
ú
ÿþ     ÿ

ü
 ÿþ
  ÿ ÿþúÿÿþ ÿ


ý

þþÿÿýü
ûù	



ý


rk   )	Úfpr   r    r!   r"   r#   r$   r%   r&   c              	   C   s   t |  ¡ |||||||ƒS )z†
    Same thing than the function from_bytes but using a file pointer that is already ready.
    Will not close the file pointer.
    )rk   Úread)rl   r   r    r!   r"   r#   r$   r%   r.   r.   r/   Úfrom_fp  s    ørn   )	Úpathr   r    r!   r"   r#   r$   r%   r&   c           	   
   C   s8   t | dƒ�$}t||||||||ƒW  5 Q R £ S Q R X dS )z•
    Same thing than the function from_bytes but with one extra step. Opening and reading given file path in binary mode.
    Can raise IOError.
    ÚrbN)Úopenrn   )	ro   r   r    r!   r"   r#   r$   r%   rl   r.   r.   r/   Ú	from_path  s    ørr   )ro   r   r    r!   r"   r#   r$   r&   c              	   C   s    t | ||||||ƒ}t| ƒ}tt|ƒƒ}	t|ƒdkrBtd |¡ƒ‚| ¡ }
|	d  d|
j 7  < t	d t
| ƒ |d |	¡¡¡dƒ�}| |
 ¡ ¡ W 5 Q R X |
S )zi
    Take a (text-based) file path and try to create another file next to it, this time using UTF-8.
    r   z;Unable to normalize "{}", no encoding charset seems to fit.ú-z{}r(   Úwb)rr   r   Úlistr   rM   ÚIOErrorrF   rd   r4   rq   rZ   ÚreplacerS   ÚwriteÚoutput)ro   r   r    r!   r"   r#   r$   rf   ÚfilenameZtarget_extensionsÚresultrl   r.   r.   r/   Ú	normalize7  s4    ù
ÿÿ ÿr|   )r   r   r   NNTF)r   r   r   NNTF)r   r   r   NNTF)r   r   r   NNT)1rP   Úos.pathr   r   Útypingr   r   r   r   Úosr   rY   rZ   Zcdr
   r   r   r   Zconstantr   r   r   r   Zmdr   Úmodelsr   r   Úutilsr   r   r   r   r   r   Ú	getLoggerrH   ÚStreamHandlerrK   ÚsetFormatterÚ	FormatterrD   rT   ÚfloatÚboolrk   rn   rr   r|   r.   r.   r.   r/   Ú<module>   s²   
 
ÿ       ø÷   b       ø÷       ø÷      ùø