Ë
    Dü´j*  ã                   ó|  — d Z ddlmZmZmZmZmZmZ ddlm	Z	 ddl
mZmZmZ ddlmZmZmZ ddlmZ edez  z   ZdZd	ed
eedf   dedeeeeedz  f      fd„Zd
eedf   dedeeeeedz  eeef      eeeeeedz  eef      f   fd„Zd
eedf   dedeeeeedz  f      fd„Zddœd	ed
eedf   dedee	   fd„Z y)zËStage 3: Statistical bigram scoring.

Note: ``from __future__ import annotations`` is intentionally omitted because
this module is compiled with mypyc, which does not support PEP 563 string
annotations.
é    )ÚBigramProfileÚ_get_model_normsÚget_enc_indexÚ
get_rowmaxÚscore_best_languageÚscore_with_profile)ÚDetectionResult)Ú_CONFUSION_BANDÚ_CONFUSION_FLOOR_RATIOÚ_STRICT_TIER_MAX_CONF)Ú_COMMON_LATIN_ENCODINGSÚ_DEMOTION_CANDIDATESÚ_RARE_ARBITRATION_MARGIN)ÚEncodingInfoé   é@   ÚdataÚ
candidates.ÚprofileÚreturnNc                 ó–   — g }|D ]A  }t        | |j                  |¬«      \  }}|dkD  sŒ$|j                  |j                  ||f«       ŒC |S )zFScore every candidate fully (no pruning).  Returns (enc, score, lang).)r   ç        )r   ÚnameÚappend)r   r   r   ÚscoresÚencÚsÚlangs          úZ/root/workspace/ytshorts/venv/lib/python3.12/site-packages/chardet/pipeline/statistical.pyÚ
_score_allr    +   sS   € ð 35€FØò /ˆÜ% d¨C¯H©H¸gÔF‰ˆˆ4Øˆs‹7Ø�M‰M˜3Ÿ8™8 Q¨Ð-Õ.ð/ð €Mó    c           
      ó–  — t        «       }t        «       }t        «       }|j                  }|j                  }|j
                  }g }g }	| D ]ç  }
|j                  |
j                  «      }|€Œ!|
j                  r8t        |«      D ])  \  }\  }}}|j                  |
j                  ||||f«       Œ+ Œet        |«      D ]u  \  }\  }}}||   }d}|D ]  }|||   ||   z  z  }Œ |j                  |«      }|€t        d«      }n|dkD  r	|||z  z  }nd}|	j                  |||
j                  |||f«       Œw Œé |	j                  d„ d¬«       ||	fS )u~  Flatten candidate model variants for pruned scoring.

    Returns ``(mb_entries, sb_entries)`` where multi-byte entries are
    ``(enc, lang, table, key, variant_index)`` and single-byte entries are
    ``(upper_bound, variant_index, enc, lang, table, key)`` sorted by
    descending bound.  The upper bound multiplies each lead byte's total
    profile weight by the model's maximum weight for that lead byte â€” at
    most 256 terms versus one term per distinct bigram for a full score.
    ``variant_index`` is the variant's position in the encoding index, so
    exact score ties resolve to the same variant the full path keeps.
    r   Úinfr   c                 ó   — | d   S )Nr   © )Úes    r   ú<lambda>z!_split_variants.<locals>.<lambda>l   s
   €  ! A¡$€ r!   T©ÚkeyÚreverse)r   r   r   Úrow_freqÚnonzero_rowsÚ
input_normÚgetr   Úis_multibyteÚ	enumerater   ÚfloatÚsort)r   r   ÚindexÚnormsÚrowmaxr+   r,   r-   Ú
mb_entriesÚ
sb_entriesr   ÚvariantsÚvir   Útabler)   ÚrmÚub_dotÚb1Ú
model_normÚubs                        r   Ú_split_variantsr@   9   sˆ  € ô$ ‹O€EÜÓ€EÜ‹\€FØ×Ñ€HØ×'Ñ'€LØ×#Ñ#€Jà@B€JØGI€JØò DˆØ—9‘9˜SŸX™XÓ&ˆØÐØØ×ÒÜ*3°HÓ*=ò DÑ&�Ñ&�T˜5 #Ø×!Ñ! 3§8¡8¨T°5¸#¸rÐ"BÕCðDàÜ&/°Ó&9ò 	DÑ"ˆBÑ"��u˜cØ˜‘ˆBØˆFØ"ò 0�Ø˜"˜R™& 8¨B¡<Ñ/Ñ/‘ð0àŸ™ 3›ˆJØÐ!ä˜5“\‘Ø˜cÒ!Ø˜z¨JÑ6Ñ7‘ð �Ø×Ñ˜r 2 s§x¡x°°u¸cÐBÕCñ	DðDð0 ‡O�O™°€OÔ5Ø�zÐ!Ð!r!   c           
      ó8  ‡‡‡‡‡‡— t        «       }t        | |«      \  }}i Ši Ši ŠdŠdŠdŠdt        dt        dt        dz  dt        ddf
ˆˆˆˆˆˆfd	„}|D ]  \  }}}}	}
 ||t        |||	«      ||
«       Œ  |D ]K  \  }}
}}}}	‰t        z
  }‰t        k  rt        |‰t        z  «      }||k  r n ||t        |||	«      ||
«       ŒM ‰t        z
  }‰t        k  rt        |‰t        z  «      }‰j                  «       D ��cg c]  \  }}||k\  sŒ|‘Œ }}}g }t        d
„ |D «       «      r|j                  t        «       d|v r|j                  d«       |rg| D ]b  }|j                  |vrŒt!        |j#                  |j                  g «      «      D ])  \  }
\  }}}	 ||j                  t        |||	«      ||
«       Œ+ Œd | D �cg c]J  }‰j#                  |j                  d«      dkD  r)|j                  ‰|j                     ‰|j                     f‘ŒL c}S c c}}w c c}w )uŽ  Score candidates, skipping single-byte variants that provably cannot matter.

    Multi-byte variants are always scored fully â€” the orchestrator may later
    boost their confidence based on structural coverage, so no raw-score
    bound can rule them out.  Single-byte variants are scored in descending
    upper-bound order (see :func:`_split_variants`) and skipped once their
    bound falls more than ``_PRUNE_MARGIN`` below the running second-best
    encoding score: such variants can affect neither the winner, nor
    position 1, nor any candidate within the confusion band of the top
    score.

    Encodings that ``postprocess_results`` inspects regardless of rank
    (the common Western Latin trio for niche-Latin demotion, KOI8-T for the
    KOI8-R promotion) are force-scored when their trigger could fire.
    Because confusion resolution can promote position 1 or any candidate
    within the band into position 0 before those triggers are evaluated,
    the trigger check covers every encoding near the top, not just the
    statistical winner.

    Returns (enc, score, lang) tuples for encodings scoring above zero, in
    candidate order.
    Ú r   Úenc_namer   r   Nr9   r   c                 ó¨   •— ‰j                  | «      }|�||k  s||k(  r	|‰|    k\  ry |‰| <   |‰| <   |‰| <   | ‰	k(  r|Šy |‰kD  r‰Š
|Š| Š	y |‰
kD  r|Š
y y ©N)r.   )rC   r   r   r9   ÚprevÚ	best_langÚ
best_scoreÚbest_viÚtop1Útop1_encÚtop2s        €€€€€€r   Úrecordz_score_pruned.<locals>.record–   s†   ø€ à�~‰~˜hÓ'ˆØÐ  T¢¨a°4ªi¸BÀ'È(ÑBSÒ<Sð Ø ˆ
�8ÑØ"ˆ	�(ÑØˆ�ÑØ�xÒØ‰DØ�ŠXØˆDØˆDØ‰HØ�ŠXØ‰Dð r!   c              3   ó,   K  — | ]  }|t         v –— Œ y ­wrE   )r   )Ú.0r&   s     r   ú	<genexpr>z _score_pruned.<locals>.<genexpr>Í   s   è ø€ Ò
7¨ˆ1Ô$Ô$Ñ
7ùs   ‚zkoi8-rzkoi8-t)r   r@   Ústrr1   Úintr   Ú_PRUNE_MARGINr   Úminr   ÚitemsÚanyÚextendr   r   r   r0   r.   )r   r   r3   r6   r7   rM   rC   r   r:   r)   r9   r?   Ú	thresholdÚtrigger_floorr&   r   Únear_topÚforcedr   rG   rH   rI   rJ   rK   rL   s                      @@@@@@r   Ú_score_prunedr\   p   s`  ý€ ô4 ‹O€EÜ,¨Z¸ÓAÑ€J�
à#%€JØ')€IØ €Gð €HØ€DØ€Dðœð ¤ð ¬c°D©jð ¼cð Àd÷ ò ð& +5ò LÑ&ˆ�$˜˜s BÙˆxÔ+¨G°U¸CÓ@À$ÈÕKðLð /9ò LÑ*ˆˆB�˜$  sð œ=Ñ(ˆ	ØÔ'Ò'Ü˜I tÔ.DÑ'DÓEˆIØ�	Š>ñ ÙˆxÔ+¨G°U¸CÓ@À$ÈÕKðLð8 œ=Ñ(€MØÔ#Ò#Ü˜M¨4Ô2HÑ+HÓIˆØ(×.Ñ.Ó0×G‘d�a˜°A¸Ó4F’ÐG€HÑGØ€FÜ
Ñ
7¨hÔ
7Ô7Ø�‰Ô-Ô.Ø�8ÑØ�‰�hÔÙØò 	TˆCØ�x‰x˜vÑ%Øô +4°E·I±I¸c¿h¹hÈÓ4KÓ*Lò TÑ&�Ñ&�T˜5 #Ù�s—x‘xÔ!3°G¸UÀCÓ!HÈ$ÐPRÕSñTð	Tð öàØ�>‰>˜#Ÿ(™( CÓ(¨3Ò.ð 
�‰�:˜cŸh™hÑ'¨°3·8±8Ñ)<Ò=òð ùó Hùòs   Ã?HÄHÆ?AHF)Úfull_rankingr]   c          
      ó4  — | r|sg S t        | «      }|j                  dk(  rg S |st        |j                  «      t        k  rt        | ||«      }nt        ||«      }|j                  d„ d¬«       |D ���cg c]  \  }}}t        |||¬«      ‘Œ c}}}S c c}}}w )a•  Score all candidates and return results sorted by confidence descending.

    :param data: The raw byte data to score.
    :param candidates: Encoding candidates to evaluate.
    :param full_ranking: When ``True``, score every candidate fully so the
        returned list is complete (needed by ``detect_all``).  When ``False``
        (the default), single-byte candidates that provably cannot affect the
        top of the ranking may be skipped; the winner, position 1, and all
        candidates within the confusion band of the top score are identical
        to the full ranking.
    :returns: A list of :class:`DetectionResult` sorted by confidence.
    r   c                 ó   — | d   S )Né   r%   )Úxs    r   r'   z"score_candidates.<locals>.<lambda>ÿ   s
   € ˜a ™d€ r!   Tr(   )ÚencodingÚ
confidenceÚlanguage)	r   r-   ÚlenÚnonzeroÚ_MIN_NONZERO_FOR_PRESCREENr    r\   r2   r	   )r   r   r]   r   r   r   r   r   s           r   Úscore_candidatesrh   á   sŸ   € ñ$ ‘zØˆ	ä˜DÓ!€GØ×Ñ˜SÒ Øˆ	á”s˜7Ÿ?™?Ó+Ô.HÒHÜ˜D *¨gÓ6‰ä˜z¨7Ó3ˆà
‡K�K‘N¨D€KÔ1ð $÷ð áˆD�!�Tô 	 °!¸dÖCôð ùô s   Á5B)!Ú__doc__Úchardet.modelsr   r   r   r   r   r   Úchardet.pipeliner	   Úchardet.pipeline.confusionr
   r   r   Úchardet.pipeline.postprocessr   r   r   Úchardet.registryr   rS   rg   ÚbytesÚtupleÚlistrQ   r1   r    rR   r@   r\   Úboolrh   r%   r!   r   ú<module>rs      s�  ðñ÷÷ õ -÷ñ ÷
ñ õ
 *ð )¨1¨Ñ+>Ñ>€ð  Ð ðØ
ðà�l CÐ'Ñ(ðð ðð 
ˆ%��U˜C $™JÐ&Ñ
'Ñ(ó	ð4"Ø�l CÐ'Ñ(ð4"àð4"ð Øˆˆs�C˜$‘J  s¨CÐ/Ñ	0Ñ1Øˆˆu�c˜3  d¡
¨E°3Ð6Ñ	7Ñ8ð:ñó4"ðnnØ�l CÐ'Ñ(ðnàðnð 
ˆ%��U˜C $™JÐ&Ñ
'Ñ(ónðj ò	"Ø
ð"à�l CÐ'Ñ(ð"ð ð	"ð
 
ˆ/Ñô"r!   