a
    Bù’ij0  ã                   @  sº   d dl mZ d dlmZ d dlmZ d dlmZ d dlm	Z	 d dl
mZmZmZmZ ddlmZmZ dd	lmZmZmZ G d
d„ dƒZG dd„ dƒZeeef Zee ZG dd„ dƒZdS )é    )Úannotations)Úaliases)Úsha256)Údumps)Úsub)ÚAnyÚIteratorÚListÚTupleé   )ÚRE_POSSIBLE_ENCODING_INDICATIONÚTOO_BIG_SEQUENCE)Ú	iana_nameÚis_multi_byte_encodingÚunicode_rangec                	   @  s¸  e Zd ZdAddddddddœd	d
„Zdddœdd„Zdddœdd„Zeddœdd„ƒZddœdd„Zddœdd„Z	d ddœdd„Z
eddœdd„ƒZeddœdd„ƒZeddœd d!„ƒZeddœd"d#„ƒZeddœd$d%„ƒZeddœd&d'„ƒZeddœd(d)„ƒZeddœd*d+„ƒZeddœd,d-„ƒZeddœd.d/„ƒZeddœd0d1„ƒZed2dœd3d4„ƒZeddœd5d6„ƒZeddœd7d8„ƒZeddœd9d:„ƒZdBddd<œd=d>„Zeddœd?d@„ƒZdS )CÚCharsetMatchNÚbytesÚstrÚfloatÚboolÚCoherenceMatchesú
str | None)ÚpayloadÚguessed_encodingÚmean_mess_ratioÚhas_sig_or_bomÚ	languagesÚdecoded_payloadÚpreemptive_declarationc                 C  sL   || _ || _|| _|| _|| _d | _g | _d| _d | _d | _	|| _
|| _d S )Nç        )Ú_payloadÚ	_encodingÚ_mean_mess_ratioÚ
_languagesÚ_has_sig_or_bomÚ_unicode_rangesÚ_leavesZ_mean_coherence_ratioÚ_output_payloadÚ_output_encodingÚ_stringÚ_preemptive_declaration)Úselfr   r   r   r   r   r   r   © r,   úV/home/httpd/docs/test/DocsMgr/lib/python3.9/site-packages/charset_normalizer/models.pyÚ__init__   s    
zCharsetMatch.__init__Úobject)ÚotherÚreturnc                 C  s>   t |tƒs&t |tƒr"t|ƒ| jkS dS | j|jko<| j|jkS )NF)Ú
isinstancer   r   r   ÚencodingÚfingerprint©r+   r0   r,   r,   r-   Ú__eq__*   s
    

zCharsetMatch.__eq__c                 C  sŒ   t |tƒst‚t| j|j ƒ}t| j|j ƒ}|dk rJ|dkrJ| j|jkS |dk r€|dkr€t| jƒtkrt| j|jk S | j	|j	kS | j|jk S )zQ
        Implemented to make sorted available upon CharsetMatches items.
        g{®Gáz„?g{®Gáz”?)
r2   r   Ú
ValueErrorÚabsÚchaosÚ	coherenceÚlenr    r   Úmulti_byte_usage)r+   r0   Zchaos_differenceZcoherence_differencer,   r,   r-   Ú__lt__1   s    
zCharsetMatch.__lt__©r1   c                 C  s   dt t| ƒƒt | jƒ  S )Ng      ð?)r;   r   Úraw©r+   r,   r,   r-   r<   G   s    zCharsetMatch.multi_byte_usagec                 C  s"   | j d u rt| j| jdƒ| _ | j S )NÚstrict)r)   r   r    r!   r@   r,   r,   r-   Ú__str__K   s    
zCharsetMatch.__str__c                 C  s   d| j › d| j› d�S )Nz<CharsetMatch 'z' bytes(z)>)r3   r4   r@   r,   r,   r-   Ú__repr__Q   s    zCharsetMatch.__repr__ÚNonec                 C  s8   t |tƒr|| kr"td |j¡ƒ‚d |_| j |¡ d S )Nz;Unable to add instance <{}> as a submatch of a CharsetMatch)r2   r   r7   ÚformatÚ	__class__r)   r&   Úappendr5   r,   r,   r-   Úadd_submatchT   s    ÿÿzCharsetMatch.add_submatchc                 C  s   | j S ©N)r!   r@   r,   r,   r-   r3   _   s    zCharsetMatch.encodingú	list[str]c                 C  sD   g }t  ¡ D ]2\}}| j|kr*| |¡ q| j|kr| |¡ q|S )z‚
        Encoding name are known by many name, using this could help when searching for IBM855 when it's listed as CP855.
        )r   Úitemsr3   rG   )r+   Zalso_known_asÚuÚpr,   r,   r-   Úencoding_aliasesc   s    

zCharsetMatch.encoding_aliasesc                 C  s   | j S rI   ©r$   r@   r,   r,   r-   Úbomp   s    zCharsetMatch.bomc                 C  s   | j S rI   rO   r@   r,   r,   r-   Úbyte_order_markt   s    zCharsetMatch.byte_order_markc                 C  s   dd„ | j D ƒS )zÔ
        Return the complete list of possible languages found in decoded sequence.
        Usually not really useful. Returned list may be empty even if 'language' property return something != 'Unknown'.
        c                 S  s   g | ]}|d  ‘qS )r   r,   )Ú.0Úer,   r,   r-   Ú
<listcomp>~   ó    z*CharsetMatch.languages.<locals>.<listcomp>©r#   r@   r,   r,   r-   r   x   s    zCharsetMatch.languagesc                 C  sp   | j sbd| jv rdS ddlm}m} t| jƒr8|| jƒn|| jƒ}t|ƒdksVd|v rZdS |d S | j d d S )z’
        Most probable language found in decoded sequence. If none were detected or inferred, the property will return
        "Unknown".
        ÚasciiZEnglishr   )Úencoding_languagesÚmb_encoding_languageszLatin BasedÚUnknown)r#   Úcould_be_from_charsetZcharset_normalizer.cdrX   rY   r   r3   r;   )r+   rX   rY   r   r,   r,   r-   Úlanguage€   s    
ÿýzCharsetMatch.languagec                 C  s   | j S rI   )r"   r@   r,   r,   r-   r9   œ   s    zCharsetMatch.chaosc                 C  s   | j s
dS | j d d S )Nr   r   r   rV   r@   r,   r,   r-   r:       s    zCharsetMatch.coherencec                 C  s   t | jd dd�S ©Néd   é   )Úndigits)Úroundr9   r@   r,   r,   r-   Úpercent_chaos¦   s    zCharsetMatch.percent_chaosc                 C  s   t | jd dd�S r]   )ra   r:   r@   r,   r,   r-   Úpercent_coherenceª   s    zCharsetMatch.percent_coherencec                 C  s   | j S )z+
        Original untouched bytes.
        )r    r@   r,   r,   r-   r?   ®   s    zCharsetMatch.rawzlist[CharsetMatch]c                 C  s   | j S rI   )r&   r@   r,   r,   r-   Úsubmatchµ   s    zCharsetMatch.submatchc                 C  s   t | jƒdkS ©Nr   )r;   r&   r@   r,   r,   r-   Úhas_submatch¹   s    zCharsetMatch.has_submatchc                 C  s@   | j d ur| j S dd„ t| ƒD ƒ}ttdd„ |D ƒƒƒ| _ | j S )Nc                 S  s   g | ]}t |ƒ‘qS r,   )r   )rR   Úcharr,   r,   r-   rT   Â   rU   z*CharsetMatch.alphabets.<locals>.<listcomp>c                 S  s   h | ]}|r|’qS r,   r,   )rR   Úrr,   r,   r-   Ú	<setcomp>Ä   rU   z)CharsetMatch.alphabets.<locals>.<setcomp>)r%   r   ÚsortedÚlist)r+   Zdetected_rangesr,   r,   r-   Ú	alphabets½   s
    
zCharsetMatch.alphabetsc                 C  s   | j gdd„ | jD ƒ S )zÜ
        The complete list of encoding that output the exact SAME str result and therefore could be the originating
        encoding.
        This list does include the encoding available in property 'encoding'.
        c                 S  s   g | ]
}|j ‘qS r,   )r3   )rR   Úmr,   r,   r-   rT   Î   rU   z6CharsetMatch.could_be_from_charset.<locals>.<listcomp>)r!   r&   r@   r,   r,   r-   r[   Ç   s    z"CharsetMatch.could_be_from_charsetÚutf_8)r3   r1   c                   s~   ˆ j du sˆ j |krx|ˆ _ tˆ ƒ}ˆ jdurjˆ j ¡ dvrjtt‡ fdd„|dd… dd�}||dd…  }| |d¡ˆ _ˆ jS )	z®
        Method to get re-encoded bytes payload using given target encoding. Default to UTF-8.
        Any errors will be simply ignored by the encoder NOT replaced.
        N)zutf-8Úutf8rn   c                   s<   | j |  ¡ d |  ¡ d …  |  ¡ d tˆ jƒ dd¡¡S )Nr   r   Ú_ú-)ÚstringÚspanÚreplaceÚgroupsr   r(   )rm   r@   r,   r-   Ú<lambda>ß   s   
þz%CharsetMatch.output.<locals>.<lambda>i    r   )Úcountrt   )r(   r   r*   Úlowerr   r   Úencoder'   )r+   r3   Údecoded_stringZpatched_headerr,   r@   r-   ÚoutputÐ   s$    ÿÿþ

ù
zCharsetMatch.outputc                 C  s   t |  ¡ ƒ ¡ S )zw
        Retrieve the unique SHA256 computed using the transformed (re-encoded) payload. Not the original one.
        )r   r{   Ú	hexdigestr@   r,   r,   r-   r4   í   s    zCharsetMatch.fingerprint)NN)rn   )Ú__name__Ú
__module__Ú__qualname__r.   r6   r=   Úpropertyr<   rB   rC   rH   r3   rN   rP   rQ   r   r\   r9   r:   rb   rc   r?   rd   rf   rl   r[   r{   r4   r,   r,   r,   r-   r      sV     ø	r   c                   @  s†   e Zd ZdZdddœdd„Zddœd	d
„Zdddœdd„Zddœdd„Zddœdd„Zdddœdd„Z	ddœdd„Z
ddœdd„ZdS )ÚCharsetMatchesz³
    Container with every CharsetMatch items ordered by default from most probable to the less one.
    Act like a list(iterable) but does not implements all related methods.
    Nzlist[CharsetMatch] | None)Úresultsc                 C  s   |rt |ƒng | _d S rI   )rj   Ú_results)r+   r‚   r,   r,   r-   r.   û   s    zCharsetMatches.__init__zIterator[CharsetMatch]r>   c                 c  s   | j E d H  d S rI   ©rƒ   r@   r,   r,   r-   Ú__iter__þ   s    zCharsetMatches.__iter__z	int | strr   )Úitemr1   c                 C  sN   t |tƒr| j| S t |tƒrFt|dƒ}| jD ]}||jv r.|  S q.t‚dS )z¸
        Retrieve a single item either by its position or encoding name (alias may be used here).
        Raise KeyError upon invalid index or encoding not present in results.
        FN)r2   Úintrƒ   r   r   r[   ÚKeyError)r+   r†   Úresultr,   r,   r-   Ú__getitem__  s    






zCharsetMatches.__getitem__r‡   c                 C  s
   t | jƒS rI   ©r;   rƒ   r@   r,   r,   r-   Ú__len__  s    zCharsetMatches.__len__r   c                 C  s   t | jƒdkS re   r‹   r@   r,   r,   r-   Ú__bool__  s    zCharsetMatches.__bool__rD   c                 C  s|   t |tƒstd t|jƒ¡ƒ‚t|jƒtk r`| j	D ],}|j
|j
kr2|j|jkr2| |¡  dS q2| j	 |¡ t| j	ƒ| _	dS )z~
        Insert a single match. Will be inserted accordingly to preserve sort.
        Can be inserted as a submatch.
        z-Cannot append instance '{}' to CharsetMatchesN)r2   r   r7   rE   r   rF   r;   r?   r   rƒ   r4   r9   rH   rG   rj   )r+   r†   Úmatchr,   r,   r-   rG     s    
ÿÿ

zCharsetMatches.appendzCharsetMatch | Nonec                 C  s   | j s
dS | j d S )zQ
        Simply return the first match. Strict equivalent to matches[0].
        Nr   r„   r@   r,   r,   r-   Úbest)  s    zCharsetMatches.bestc                 C  s   |   ¡ S )zP
        Redundant method, call the method best(). Kept for BC reasons.
        )r�   r@   r,   r,   r-   Úfirst1  s    zCharsetMatches.first)N)r}   r~   r   Ú__doc__r.   r…   rŠ   rŒ   r�   rG   r�   r�   r,   r,   r,   r-   r�   õ   s   r�   c                   @  sN   e Zd Zddddddddddddœdd„Zed	d
œdd„ƒZdd
œdd„ZdS )ÚCliDetectionResultr   r   rJ   r   r   ©Úpathr3   rN   Úalternative_encodingsr\   rl   r   r9   r:   Úunicode_pathÚis_preferredc                 C  sF   || _ |
| _|| _|| _|| _|| _|| _|| _|| _|	| _	|| _
d S rI   )r”   r–   r3   rN   r•   r\   rl   r   r9   r:   r—   )r+   r”   r3   rN   r•   r\   rl   r   r9   r:   r–   r—   r,   r,   r-   r.   =  s    zCliDetectionResult.__init__zdict[str, Any]r>   c                 C  s2   | j | j| j| j| j| j| j| j| j| j	| j
dœS )Nr“   r“   r@   r,   r,   r-   Ú__dict__W  s    õzCliDetectionResult.__dict__c                 C  s   t | jddd�S )NTé   )Úensure_asciiÚindent)r   r˜   r@   r,   r,   r-   Úto_jsong  s    zCliDetectionResult.to_jsonN)r}   r~   r   r.   r€   r˜   rœ   r,   r,   r,   r-   r’   <  s   "r’   N)Ú
__future__r   Zencodings.aliasesr   Úhashlibr   Újsonr   Úrer   Útypingr   r   r	   r
   Zconstantr   r   Úutilsr   r   r   r   r�   r   r   ZCoherenceMatchr   r’   r,   r,   r,   r-   Ú<module>   s    iC