U
    £�©jÚ4  ã                   @  s¢   d dl mZ d dlmZ d dlmZ d dlmZmZm	Z	m
Z
 ddlmZmZ ddlmZmZmZ G dd	„ d	ƒZG d
d„ dƒZe
eef Ze	e ZG dd„ dƒZdS )é    )Úannotations)Úaliases)Úsub)ÚAnyÚIteratorÚListÚTupleé   )ÚRE_POSSIBLE_ENCODING_INDICATIONÚTOO_BIG_SEQUENCE)Ú	iana_nameÚis_multi_byte_encodingÚunicode_rangec                	   @  s¸  e Zd ZdCddddddddœd	d
„Zdddœdd„Zdddœdd„Zeddœdd„ƒZddœdd„Zddœdd„Z	d ddœdd„Z
eddœdd„ƒZeddœdd„ƒZeddœd d!„ƒZeddœd"d#„ƒZeddœd$d%„ƒZeddœd&d'„ƒZeddœd(d)„ƒZeddœd*d+„ƒZeddœd,d-„ƒZeddœd.d/„ƒZeddœd0d1„ƒZed2dœd3d4„ƒZeddœd5d6„ƒZeddœd7d8„ƒZeddœd9d:„ƒZdDdd<d=œd>d?„Zed@dœdAdB„ƒZdS )EÚCharsetMatchNzbytes | bytearrayÚstrÚfloatÚboolÚCoherenceMatchesú
str | None)ÚpayloadÚguessed_encodingÚmean_mess_ratioÚhas_sig_or_bomÚ	languagesÚdecoded_payloadÚpreemptive_declarationc                 C  sL   || _ || _|| _|| _|| _d | _g | _d| _d | _d | _	|| _
|| _d S )Nç        )Ú_payloadÚ	_encodingÚ_mean_mess_ratioÚ
_languagesÚ_has_sig_or_bomÚ_unicode_rangesÚ_leavesZ_mean_coherence_ratioÚ_output_payloadÚ_output_encodingÚ_stringÚ_preemptive_declaration)Úselfr   r   r   r   r   r   r   © r)   ú=/tmp/pip-unpacked-wheel-bwd693va/charset_normalizer/models.pyÚ__init__   s    
zCharsetMatch.__init__Úobject)ÚotherÚreturnc                 C  s@   t |tƒs(t |tƒr$t|dƒ| jkS dS | j|jko>| j|jkS )NF)Ú
isinstancer   r   r   ÚencodingÚfingerprint©r(   r-   r)   r)   r*   Ú__eq__(   s
    

zCharsetMatch.__eq__c                 C  sŒ   t |tƒst‚t| j|j ƒ}t| j|j ƒ}|dk rJ|dkrJ| j|jkS |dk r€|dkr€t| jƒtkrt| j|jk S | j	|j	kS | j|jk S )zQ
        Implemented to make sorted available upon CharsetMatches items.
        g{®Gázt?g{®Gáz”?)
r/   r   Ú
ValueErrorÚabsÚchaosÚ	coherenceÚlenr   r   Úmulti_byte_usage)r(   r-   Zchaos_differenceZcoherence_differencer)   r)   r*   Ú__lt__3   s    
zCharsetMatch.__lt__©r.   c                 C  s*   t | jƒ}|dkrdS dt t| ƒƒ|  S )Nr   r   g      ð?)r8   Úrawr   )r(   Zraw_lenr)   r)   r*   r9   I   s    
zCharsetMatch.multi_byte_usagec                 C  sV   | j d krPt| j| jdƒ| _ | jrP| jdkrP| j rP| j d dkrP| j dd … | _ | j S )NÚstrictÚutf_7r   u   ï»¿r	   )r&   r   r   r   r!   ©r(   r)   r)   r*   Ú__str__Q   s    
ÿþýüzCharsetMatch.__str__c                 C  s   d| j › d| j› d�S )Nz<CharsetMatch 'z' fp(z)>)r0   r1   r?   r)   r)   r*   Ú__repr__a   s    zCharsetMatch.__repr__ÚNonec                 C  s8   t |tƒr|| kr"td |j¡ƒ‚d |_| j |¡ d S )Nz;Unable to add instance <{}> as a submatch of a CharsetMatch)r/   r   r4   ÚformatÚ	__class__r&   r#   Úappendr2   r)   r)   r*   Úadd_submatchd   s    ÿÿzCharsetMatch.add_submatchc                 C  s   | j S ©N)r   r?   r)   r)   r*   r0   o   s    zCharsetMatch.encodingú	list[str]c                 C  sD   g }t  ¡ D ]2\}}| j|kr*| |¡ q| j|kr| |¡ q|S )z‚
        Encoding name are known by many name, using this could help when searching for IBM855 when it's listed as CP855.
        )r   Úitemsr0   rE   )r(   Zalso_known_asÚuÚpr)   r)   r*   Úencoding_aliasess   s    

zCharsetMatch.encoding_aliasesc                 C  s   | j S rG   ©r!   r?   r)   r)   r*   Úbom€   s    zCharsetMatch.bomc                 C  s   | j S rG   rM   r?   r)   r)   r*   Úbyte_order_mark„   s    zCharsetMatch.byte_order_markc                 C  s   dd„ | j D ƒS )zÔ
        Return the complete list of possible languages found in decoded sequence.
        Usually not really useful. Returned list may be empty even if 'language' property return something != 'Unknown'.
        c                 S  s   g | ]}|d  ‘qS )r   r)   )Ú.0Úer)   r)   r*   Ú
<listcomp>Ž   s     z*CharsetMatch.languages.<locals>.<listcomp>©r    r?   r)   r)   r*   r   ˆ   s    zCharsetMatch.languagesc                 C  sp   | j sbd| jkrdS ddlm}m} t| jƒr8|| jƒn|| jƒ}t|ƒdksVd|krZdS |d S | j d d S )z’
        Most probable language found in decoded sequence. If none were detected or inferred, the property will return
        "Unknown".
        ÚasciiZEnglishr   )Úencoding_languagesÚmb_encoding_languageszLatin BasedÚUnknown)r    Úcould_be_from_charsetZcharset_normalizer.cdrU   rV   r   r0   r8   )r(   rU   rV   r   r)   r)   r*   Úlanguage�   s    
ÿýzCharsetMatch.languagec                 C  s   | j S rG   )r   r?   r)   r)   r*   r6   ¬   s    zCharsetMatch.chaosc                 C  s   | j s
dS | j d d S )Nr   r   r	   rS   r?   r)   r)   r*   r7   °   s    zCharsetMatch.coherencec                 C  s   t | jd dd�S ©Néd   é   )Úndigits)Úroundr6   r?   r)   r)   r*   Úpercent_chaos¶   s    zCharsetMatch.percent_chaosc                 C  s   t | jd dd�S rZ   )r^   r7   r?   r)   r)   r*   Úpercent_coherenceº   s    zCharsetMatch.percent_coherencec                 C  s   | j S )z+
        Original untouched bytes.
        )r   r?   r)   r)   r*   r<   ¾   s    zCharsetMatch.rawzlist[CharsetMatch]c                 C  s   | j S rG   )r#   r?   r)   r)   r*   ÚsubmatchÅ   s    zCharsetMatch.submatchc                 C  s   t | jƒdkS ©Nr   )r8   r#   r?   r)   r)   r*   Úhas_submatchÉ   s    zCharsetMatch.has_submatchc                 C  s@   | j d k	r| j S dd„ t| ƒD ƒ}ttdd„ |D ƒƒƒ| _ | j S )Nc                 S  s   g | ]}t |ƒ‘qS r)   )r   )rP   Úcharr)   r)   r*   rR   Ò   s     z*CharsetMatch.alphabets.<locals>.<listcomp>c                 S  s   h | ]}|r|’qS r)   r)   )rP   Úrr)   r)   r*   Ú	<setcomp>Ô   s      z)CharsetMatch.alphabets.<locals>.<setcomp>)r"   r   ÚsortedÚlist)r(   Zdetected_rangesr)   r)   r*   Ú	alphabetsÍ   s
    
zCharsetMatch.alphabetsc                 C  s   | j gdd„ | jD ƒ S )zÜ
        The complete list of encoding that output the exact SAME str result and therefore could be the originating
        encoding.
        This list does include the encoding available in property 'encoding'.
        c                 S  s   g | ]
}|j ‘qS r)   )r0   )rP   Úmr)   r)   r*   rR   Þ   s     z6CharsetMatch.could_be_from_charset.<locals>.<listcomp>)r   r#   r?   r)   r)   r*   rX   ×   s    z"CharsetMatch.could_be_from_charsetÚutf_8Úbytes)r0   r.   c                   s~   ˆ j dksˆ j |krx|ˆ _ tˆ ƒ}ˆ jdk	rjˆ j ¡ dkrjtt‡ fdd„|dd… dd�}||dd…  }| |d¡ˆ _ˆ jS )	z®
        Method to get re-encoded bytes payload using given target encoding. Default to UTF-8.
        Any errors will be simply ignored by the encoder NOT replaced.
        N)zutf-8Úutf8rk   c                   s<   | j |  ¡ d |  ¡ d …  |  ¡ d tˆ jƒ dd¡¡S )Nr   r	   Ú_ú-)ÚstringÚspanÚreplaceÚgroupsr   r%   )rj   r?   r)   r*   Ú<lambda>ï   s   
þz%CharsetMatch.output.<locals>.<lambda>i    r	   )Úcountrr   )r%   r   r'   Úlowerr   r
   Úencoder$   )r(   r0   Údecoded_stringZpatched_headerr)   r?   r*   Úoutputà   s$    ÿÿþ

ù
zCharsetMatch.outputÚintc                 C  s   t t| ƒƒS )z]
        Retrieve a hash fingerprint of the decoded payload, used for deduplication.
        )Úhashr   r?   r)   r)   r*   r1   ý   s    zCharsetMatch.fingerprint)NN)rk   )Ú__name__Ú
__module__Ú__qualname__r+   r3   r:   Úpropertyr9   r@   rA   rF   r0   rL   rN   rO   r   rY   r6   r7   r_   r`   r<   ra   rc   ri   rX   ry   r1   r)   r)   r)   r*   r      sV     ø	r   c                   @  s”   e Zd ZdZd ddœdd„Zddœd	d
„Zddœdd„Zdddœdd„Zddœdd„Zddœdd„Z	dddœdd„Z
ddœdd„Zddœdd„ZdS )!ÚCharsetMatchesz³
    Container with every CharsetMatch items ordered by default from most probable to the less one.
    Act like a list(iterable) but does not implements all related methods.
    Nzlist[CharsetMatch] | None)Úresultsc                 C  s   |rt |ƒng | _d| _d S ©NT)rg   Ú_resultsÚ
_is_sorted)r(   r�   r)   r)   r*   r+     s    zCharsetMatches.__init__rB   r;   c                 C  s   | j s| j ¡  d| _ d S r‚   )r„   rƒ   Úsortr?   r)   r)   r*   Ú_ensure_sorted  s    
zCharsetMatches._ensure_sortedzIterator[CharsetMatch]c                 c  s   |   ¡  | jE d H  d S rG   )r†   rƒ   r?   r)   r)   r*   Ú__iter__  s    zCharsetMatches.__iter__z	int | strr   )Úitemr.   c                 C  sV   t |tƒr|  ¡  | j| S t |tƒrNt|dƒ}| jD ]}||jkr6|  S q6t‚dS )z¸
        Retrieve a single item either by its position or encoding name (alias may be used here).
        Raise KeyError upon invalid index or encoding not present in results.
        FN)r/   rz   r†   rƒ   r   r   rX   ÚKeyError)r(   rˆ   Úresultr)   r)   r*   Ú__getitem__  s    






zCharsetMatches.__getitem__rz   c                 C  s
   t | jƒS rG   ©r8   rƒ   r?   r)   r)   r*   Ú__len__'  s    zCharsetMatches.__len__r   c                 C  s   t | jƒdkS rb   rŒ   r?   r)   r)   r*   Ú__bool__*  s    zCharsetMatches.__bool__c                 C  sv   t |tƒstd t|jƒ¡ƒ‚t|jƒtk r`| j	D ],}|j
|j
kr2|j|jkr2| |¡  dS q2| j	 |¡ d| _dS )z~
        Insert a single match. Will be inserted accordingly to preserve sort.
        Can be inserted as a submatch.
        z-Cannot append instance '{}' to CharsetMatchesNF)r/   r   r4   rC   r   rD   r8   r<   r   rƒ   r1   r6   rF   rE   r„   )r(   rˆ   Úmatchr)   r)   r*   rE   -  s    
ÿÿ

zCharsetMatches.appendzCharsetMatch | Nonec                 C  s   | j s
dS |  ¡  | j d S )zQ
        Simply return the first match. Strict equivalent to matches[0].
        Nr   )rƒ   r†   r?   r)   r)   r*   ÚbestA  s    zCharsetMatches.bestc                 C  s   |   ¡ S )zP
        Redundant method, call the method best(). Kept for BC reasons.
        )r�   r?   r)   r)   r*   ÚfirstJ  s    zCharsetMatches.first)N)r|   r}   r~   Ú__doc__r+   r†   r‡   r‹   r�   rŽ   rE   r�   r‘   r)   r)   r)   r*   r€     s   	r€   c                   @  sN   e Zd Zddddddddddddœdd„Zed	d
œdd„ƒZdd
œdd„ZdS )ÚCliDetectionResultr   r   rH   r   r   ©Úpathr0   rL   Úalternative_encodingsrY   ri   r   r6   r7   Úunicode_pathÚis_preferredc                 C  sF   || _ |
| _|| _|| _|| _|| _|| _|| _|| _|	| _	|| _
d S rG   )r•   r—   r0   rL   r–   rY   ri   r   r6   r7   r˜   )r(   r•   r0   rL   r–   rY   ri   r   r6   r7   r—   r˜   r)   r)   r*   r+   V  s    zCliDetectionResult.__init__zdict[str, Any]r;   c                 C  s2   | j | j| j| j| j| j| j| j| j| j	| j
dœS )Nr”   r”   r?   r)   r)   r*   Ú__dict__p  s    õzCliDetectionResult.__dict__c                 C  s   ddl m} || jddd�S )Nr   )ÚdumpsTé   )Úensure_asciiÚindent)Újsonrš   r™   )r(   rš   r)   r)   r*   Úto_json€  s    zCliDetectionResult.to_jsonN)r|   r}   r~   r+   r   r™   rŸ   r)   r)   r)   r*   r“   U  s   "r“   N)Ú
__future__r   Zencodings.aliasesr   Úrer   Útypingr   r   r   r   Zconstantr
   r   Úutilsr   r   r   r   r€   r   r   ZCoherenceMatchr   r“   r)   r)   r)   r*   Ú<module>   s    {L