a
    Bù’iŒX  ã                   @  sR  d dl mZ d dlZd dlmZ d dlmZ ddlmZm	Z	m
Z
mZ ddlmZmZmZmZ ddlmZ dd	lmZmZ dd
lmZmZmZmZmZmZmZ e d¡Ze  ¡ Z!e! "e #d¡¡ d(ddddddddddddœdd„Z$d)ddddddddddddœdd„Z%d*d ddddddddddd!œd"d#„Z&d+d$ddddddddddd%œd&d'„Z'dS ),é    )ÚannotationsN)ÚPathLike)ÚBinaryIOé   )Úcoherence_ratioÚencoding_languagesÚmb_encoding_languagesÚmerge_coherence_ratios)ÚIANA_SUPPORTEDÚTOO_BIG_SEQUENCEÚTOO_SMALL_SEQUENCEÚTRACE)Ú
mess_ratio)ÚCharsetMatchÚCharsetMatches)Úany_specified_encodingÚcut_sequence_chunksÚ	iana_nameÚidentify_sig_or_bomÚis_cp_similarÚis_multi_byte_encodingÚshould_strip_sig_or_bomZcharset_normalizerz)%(asctime)s | %(levelname)s | %(message)sé   é   çš™™™™™É?TFçš™™™™™¹?zbytes | bytearrayÚintÚfloatzlist[str] | NoneÚboolr   )Ú	sequencesÚstepsÚ
chunk_sizeÚ	thresholdÚcp_isolationÚcp_exclusionÚpreemptive_behaviourÚexplainÚlanguage_thresholdÚenable_fallbackÚreturnc
           2      C  sÊ	  t | ttfƒs td t| ƒ¡ƒ‚|r>tj}
t t	¡ t 
t¡ t| ƒ}|dkrŽt d¡ |rvt t	¡ t 
|
prtj¡ tt| dddg dƒgƒS |durºt td	d
 |¡¡ dd„ |D ƒ}ng }|durêt tdd
 |¡¡ dd„ |D ƒ}ng }||| k�rt td|||¡ d}|}|dk�r:|| |k �r:t|| ƒ}t| ƒtk }t| ƒtk}|�rlt td |¡¡ n|�r„t td |¡¡ g }|�r–t| ƒnd}|du�r¼| |¡ t td|¡ tƒ }g }g }d}d}d}tƒ }tƒ }t| ƒ\}}|du�r| |¡ t tdt|ƒ|¡ | d¡ d|v�r4| d¡ |t D �]Ž}|�rV||v�rV�q<|�rj||v �rj�q<||v �rx�q<| |¡ d}||k}|�ošt|ƒ}|dv �r¾|�s¾t td|¡ �q<|dv �rà|�sàt td|¡ �q<zt|ƒ}W n, t t!f�y   t td|¡ Y �q<Y n0 zr|�rd|du �rdt"|du �rH| dtdƒ… n| t|ƒtdƒ… |d� n&t"|du �rt| n| t|ƒd… |d�}W nb t#t$f�yî } zDt |t$ƒ�sÂt td|t"|ƒ¡ | |¡ W Y d}~�q<W Y d}~n
d}~0 0 d} |D ]}!t%||!ƒ�rød}  �q�qø| �r0t td||!¡ �q<t&|�s<dnt|ƒ|t|| ƒƒ}"|�ol|du�olt|ƒ|k }#|#�r‚t td |¡ tt|"ƒd! ƒ}$t'|$d"ƒ}$d}%d}&g }'g }(zšt(| ||"||||||ƒ	D ]|})|' |)¡ |( t)|)||du �odt|ƒ  k�o d"kn  ƒ¡ |(d# |k�r |%d7 }%|%|$k�s:|�rÆ|du �rÆ �qD�qÆW nB t#�yˆ } z(t td$|t"|ƒ¡ |$}%d}&W Y d}~n
d}~0 0 |&�s|�r|�sz| td%ƒd… j*|d&d'� W nR t#�y } z8t td(|t"|ƒ¡ | |¡ W Y d}~�q<W Y d}~n
d}~0 0 |(�r$t+|(ƒt|(ƒ nd}*|*|k�s<|%|$k�rÂ| |¡ t td)||%t,|*d* d+d,�¡ |	�r<|dd|d-d.fv �r<|&�s<t| |||g ||d/�}+||k�rª|+}n|dk�rº|+}n|+}�q<t td0|t,|*d* d+d,�¡ |�sît-|ƒ},nt.|ƒ},|,�rt td1 |t"|,ƒ¡¡ g }-|dk�rT|'D ],})t/|)||,�r@d2 |,¡ndƒ}.|- |.¡ �q&t0|-ƒ}/|/�rvt td3 |/|¡¡ t| ||*||/|du �sœ||ddfv �r |nd|d/�}0| |0¡ ||ddfv �r|*d4k �r|*dk�rt d5|0j1¡ |�r t t	¡ t 
|
¡ t|0gƒ  S | |0¡ t|ƒ�rˆ|du �s6||v �rˆd|v �rˆd|v �rˆ| 2¡ }1t d5|1j1¡ |�rzt t	¡ t 
|
¡ t|1gƒ  S ||k�r<t d6|¡ |�r¸t t	¡ t 
|
¡ t|| gƒ  S �q<t|ƒdk�	r~|�sî|�sî|�rút td7¡ |�	rt d8|j1¡ | |¡ nd|�	r*|du �	sN|�	rD|�	rD|j3|j3k�	sN|du�	rdt d9¡ | |¡ n|�	r~t d:¡ | |¡ |�	r¢t d;| 2¡ j1t|ƒd ¡ n
t d<¡ |�	rÆt t	¡ t 
|
¡ |S )=af  
    Given a raw bytes sequence, return the best possibles charset usable to render str objects.
    If there is no results, it is a strong indicator that the source is binary/not text.
    By default, the process will extract 5 blocks of 512o each to assess the mess and coherence of a given sequence.
    And will give up a particular code page after 20% of measured mess. Those criteria are customizable at will.

    The preemptive behavior DOES NOT replace the traditional detection workflow, it prioritize a particular code page
    but never take it for granted. Can improve the performance.

    You may want to focus your attention to some code page or/and not others, use cp_isolation and cp_exclusion for that
    purpose.

    This function will strip the SIG in the payload/sequence every time except on UTF-16, UTF-32.
    By default the library does not setup any handler other than the NullHandler, if you choose to set the 'explain'
    toggle to True it will alter the logger configuration to add a StreamHandler that is suitable for debugging.
    Custom logging format and handler can be set manually.
    z3Expected object of type bytes or bytearray, got: {}r   z<Encoding detection on empty bytes, assuming utf_8 intention.Úutf_8g        FÚ Nz`cp_isolation is set. use this flag for debugging purpose. limited list of encoding allowed : %s.z, c                 S  s   g | ]}t |d ƒ‘qS ©F©r   ©Ú.0Úcp© r1   úS/home/httpd/docs/test/DocsMgr/lib/python3.9/site-packages/charset_normalizer/api.pyÚ
<listcomp>[   ó    zfrom_bytes.<locals>.<listcomp>zacp_exclusion is set. use this flag for debugging purpose. limited list of encoding excluded : %s.c                 S  s   g | ]}t |d ƒ‘qS r,   r-   r.   r1   r1   r2   r3   f   r4   z^override steps (%i) and chunk_size (%i) as content does not fit (%i byte(s) given) parameters.r   z>Trying to detect encoding from a tiny portion of ({}) byte(s).zIUsing lazy str decoding because the payload is quite large, ({}) byte(s).z@Detected declarative mark in sequence. Priority +1 given for %s.zIDetected a SIG or BOM mark on first %i byte(s). Priority +1 given for %s.Úascii>   Úutf_16Úutf_32z\Encoding %s won't be tested as-is because it require a BOM. Will try some sub-encoder LE/BE.>   Úutf_7zREncoding %s won't be tested as-is because detection is unreliable without BOM/SIG.z2Encoding %s does not provide an IncrementalDecoderg    €„A)Úencodingz9Code page %s does not fit given bytes sequence at ALL. %sTzW%s is deemed too similar to code page %s and was consider unsuited already. Continuing!zpCode page %s is a multi byte encoding table and it appear that at least one character was encoded using n-bytes.é   é   éÿÿÿÿzaLazyStr Loading: After MD chunk decode, code page %s does not fit given bytes sequence at ALL. %sg     jè@Ústrict)Úerrorsz^LazyStr Loading: After final lookup, code page %s does not fit given bytes sequence at ALL. %szc%s was excluded because of initial chaos probing. Gave up %i time(s). Computed mean chaos is %f %%.éd   é   )Úndigitsr6   r7   )Zpreemptive_declarationz=%s passed initial chaos probing. Mean measured chaos is %f %%z&{} should target any language(s) of {}ú,z We detected language {} using {}r   z.Encoding detection: %s is most likely the one.zoEncoding detection: %s is most likely the one as we detected a BOM or SIG within the beginning of the sequence.zONothing got out of the detection process. Using ASCII/UTF-8/Specified fallback.z7Encoding detection: %s will be used as a fallback matchz:Encoding detection: utf_8 will be used as a fallback matchz:Encoding detection: ascii will be used as a fallback matchz]Encoding detection: Found %s as plausible (best-candidate) for content. With %i alternatives.z=Encoding detection: Unable to determine any suitable charset.)4Ú
isinstanceÚ	bytearrayÚbytesÚ	TypeErrorÚformatÚtypeÚloggerÚlevelÚ
addHandlerÚexplain_handlerÚsetLevelr   ÚlenÚdebugÚremoveHandlerÚloggingÚWARNINGr   r   ÚlogÚjoinr   r   r   r   ÚappendÚsetr   r
   Úaddr   r   ÚModuleNotFoundErrorÚImportErrorÚstrÚUnicodeDecodeErrorÚLookupErrorr   ÚrangeÚmaxr   r   ÚdecodeÚsumÚroundr   r   r   r	   r9   ÚbestÚfingerprint)2r   r    r!   r"   r#   r$   r%   r&   r'   r(   Zprevious_logger_levelÚlengthZis_too_small_sequenceZis_too_large_sequenceZprioritized_encodingsZspecified_encodingZtestedZtested_but_hard_failureZtested_but_soft_failureZfallback_asciiZfallback_u8Zfallback_specifiedÚresultsZearly_stop_resultsZsig_encodingZsig_payloadZencoding_ianaZdecoded_payloadZbom_or_sig_availableZstrip_sig_or_bomZis_multi_byte_decoderÚeZsimilar_soft_failure_testZencoding_soft_failedZr_Zmulti_byte_bonusZmax_chunk_gave_upZearly_stop_countZlazy_str_hard_failureZ	md_chunksZ	md_ratiosÚchunkZmean_mess_ratioZfallback_entryZtarget_languagesZ	cd_ratiosZchunk_languagesZcd_ratios_mergedZcurrent_matchZprobable_resultr1   r1   r2   Ú
from_bytes!   s(   ÿÿ



üüûÿþÿþÿ

ý

ü




ÿýýýÿú	ÿú
ü
$
ü
ýÿ
ýü
÷
&ýÿ
ÿÿÿüÿþýü
$
ú
ÿÿþüù	

ü
ÿþ
ýÿþþýò
ÿþ
þ


ÿþþýüþ


ý

þþÿÿýü
ûù	



ý


rh   r   )Úfpr    r!   r"   r#   r$   r%   r&   r'   r(   r)   c
           
      C  s   t |  ¡ |||||||||	ƒ
S )z†
    Same thing than the function from_bytes but using a file pointer that is already ready.
    Will not close the file pointer.
    )rh   Úread)
ri   r    r!   r"   r#   r$   r%   r&   r'   r(   r1   r1   r2   Úfrom_fp!  s    örk   zstr | bytes | PathLike)Úpathr    r!   r"   r#   r$   r%   r&   r'   r(   r)   c
                 C  sH   t | dƒ�*}
t|
|||||||||	ƒ
W  d  ƒ S 1 s:0    Y  dS )z•
    Same thing than the function from_bytes but with one extra step. Opening and reading given file path in binary mode.
    Can raise IOError.
    ÚrbN)Úopenrk   )rl   r    r!   r"   r#   r$   r%   r&   r'   r(   ri   r1   r1   r2   Ú	from_path?  s    öro   z!PathLike | str | BinaryIO | bytes)Úfp_or_path_or_payloadr    r!   r"   r#   r$   r%   r&   r'   r(   r)   c
                 C  sz   t | ttfƒr,t| |||||||||	d�
}
nHt | ttfƒrXt| |||||||||	d�
}
nt| |||||||||	d�
}
|
 S )a)  
    Detect if the given input (file, bytes, or path) points to a binary file. aka. not a string.
    Based on the same main heuristic algorithms and default kwargs at the sole exception that fallbacks match
    are disabled to be stricter around ASCII-compatible but unlikely to be a string.
    )	r    r!   r"   r#   r$   r%   r&   r'   r(   )rC   rZ   r   ro   rE   rD   rh   rk   )rp   r    r!   r"   r#   r$   r%   r&   r'   r(   Zguessesr1   r1   r2   Ú	is_binary^  sX    öþþöörq   )	r   r   r   NNTFr   T)	r   r   r   NNTFr   T)	r   r   r   NNTFr   T)	r   r   r   NNTFr   F)(Ú
__future__r   rQ   Úosr   Útypingr   Úcdr   r   r   r	   Zconstantr
   r   r   r   Úmdr   Úmodelsr   r   Úutilsr   r   r   r   r   r   r   Ú	getLoggerrI   ÚStreamHandlerrL   ÚsetFormatterÚ	Formatterrh   rk   ro   rq   r1   r1   r1   r2   Ú<module>   sr   $

ÿ         ö$             ö$          ö$!         ö