
    *pjed              
          % S r SSKJrJrJr  SSKJrJr  SSKJ	r	  SSK
Jr  SSKJrJr  SSKJr  \" 1 Sk5      r\\   \S	'   \" 1 S
k5      r\\   \S'   \" 1 Sk5      r\\   \S'   \" 1 Sk5      r\\   \S'   \" 1 Sk5      r\\   \S'   \\\\S.r\\\\   4   \S'   \" 1 Sk5      r\\   \S'   \R;                  5        V Vs0 s H  u  pU \" U5      _M     snn r\\\4   \S'   \" \5      r \\S'   S\S\S\!4S jr"S\S\#\   S\#\   4S jr$S\S\#\   S\#\   4S jr%Sr&S S S!S".r'\\\4   \S#'   S$r(S%r)S&r*S'r+S\S\4S( jr,S\S\S)S*S\!4S+ jr-S\S\#\   S\#\   4S, jr.S\#\   S-\S\#\   4S. jr/S\S\#\   S\#\   4S/ jr0S0r1S1r2S\#\   S\#\   4S2 jr3S\S\#\   S\#\   4S3 jr4S\S\S\!4S4 jr5S\S\#\   S5\!S\#\   4S6 jr6S7S8.S\S\#\   S5\!S\#\   4S9 jjr7g:s  snn f );u\  Stage 13: post-processing rank corrections.

After statistical scoring produces a ranked list of candidates, a chain
of rank corrections fixes up the ranking when bigrams alone are
insufficient — see :func:`postprocess_results` for the order.  The steps:
dead-heat priors (superset preference, era prevalence), rare-language
arbitration (ADR-0005), confusion-group resolution (delegated to
:mod:`chardet.pipeline.confusion`), niche Latin demotion, KOI8-T
promotion, classic-Mac line-ending promotion, and last of all the
decode-safety flip, which hands a winner whose only multi-byte evidence
is an undecodable trailing sequence to the best rival that can decode
the caller's complete input.

Note: ``from __future__ import annotations`` is intentionally omitted because
this module is compiled with mypyc, which does not support PEP 563 string
annotations.
    )dangling_tail_with_ascii_prefixdecodes_completelydecodes_without_error)RARE_LANGUAGESget_enc_index)_COMPAT_NAMES)DetectionResult)confusion_pair_winnerresolve_confusion_groups)REGISTRY>   cp1252	iso8859-1
iso8859-15_COMMON_LATIN_ENCODINGS>.                                                                                                                                             _ISO_8859_10_DISTINGUISHING>   r   r   r   r   r   r   r   r   r   r      r   r   r    r!   r"      r$   r%   r&   r'   r(   r)   r*   r+      r3         r<      _ISO_8859_14_DISTINGUISHING>   rB      rC   rD      rE   _WINDOWS_1254_DISTINGUISHING>   r,                     r-   r.      r/      r0            r1            r4   rG   rC   _HP_ROMAN8_DISTINGUISHING)z
iso8859-10z
iso8859-14cp1254z	hp-roman8_DEMOTION_CANDIDATES>                           r   r   r   r"   _KOI8_T_DISTINGUISHING_DEMOTION_DELETE_KOI8_T_DELETEencodingdatareturnc                     [         R                  U 5      nUc  g[        UR                  SU5      5      [        U5      :H  $ )aX  Return True if encoding is a demotion candidate with no distinguishing bytes.

Checks whether any byte in *data* falls in the set of byte values that
decode differently under the given encoding vs iso-8859-1.  If none do,
the data is equally valid under both encodings and there is no
byte-level evidence for preferring the candidate encoding.
NF)rd   getlen	translate)rf   rg   deletes      X/var/www/html/pdf-tiff/venv/lib/python3.13/site-packages/chardet/pipeline/postprocess.py_should_demotero      s;     !!(+F~t~~dF+,D	99    resultsc                    [        U5      S:  a  US   R                  b  [        US   R                  U 5      (       a  US   R                  nUS   R                  nUSS  H  nUR                  [        ;   d  M  [        UR                  X4R                  UR                  5      nU Vs/ s H  ofR                  U:w  d  M  XdLd  M  UPM     nnU Vs/ s H  ofR                  U:X  d  M  UPM     nnU/UQUQs  $    U$ s  snf s  snf )a  Demote niche Latin encodings when no distinguishing bytes are present.

Some bigram models (e.g. iso-8859-10, iso-8859-14, windows-1254) can win
on data that contains only bytes shared with common Western Latin
encodings.  When there is no byte-level evidence for the winning
encoding, promote the first common Western Latin candidate to the top and
push the demoted encoding to last.
   r   N)rk   rf   ro   
confidencer   r	   language	mime_type)	rg   rq   demoted_encodingtop_confrpromotedxothersdemoted_entriess	            rn   _demote_niche_latinr~      s    	GqAJ+71:..55"1:..1:((Azz44*JJ**akk  '&!**8H*HAQZAw   /6"XgGW9W1g"X <6<O<<  N #Ys   )C> C>C>D)Dc                    U(       a  US   R                   S:w  a  U$ [        S [        U5       5       S5      nUc  U$ [        U R	                  S[
        5      5      [        U 5      :w  aj  X   nUS   R                  n[        UR                   UUR                  UR                  5      n[        U5       VVs/ s H  u  pgXb:w  d  M  UPM     nnnU/UQ$ U$ s  snnf )ag  Promote KOI8-T over KOI8-R when Tajik-specific bytes are present.

KOI8-T and KOI8-R share the entire 0xC0-0xFF Cyrillic letter block,
making statistical discrimination difficult.  However, KOI8-T maps 12
bytes in 0x80-0xBF to Tajik-specific Cyrillic letters where KOI8-R has
box-drawing characters.  If any of these bytes appear, KOI8-T is the
better match.
r   zkoi8-rc              3   N   #    U  H  u  pUR                   S :X  d  M  Uv   M     g7f)zkoi8-tN)rf   ).0iry   s      rn   	<genexpr>!_promote_koi8t.<locals>.<genexpr>  s!     Q$6DA!**:Paa$6s   %	%N)
rf   next	enumeraterk   rl   re   rt   r	   ru   rv   )	rg   rq   	koi8t_idxkoi8t_resultrx   rz   r   ry   r|   s	            rn   _promote_koi8tr     s     gaj))X5QIg$6QSWXI
4>>$/0CI=)1:(("!!!!""	
 !*' 2E 2an! 2E"6""N Fs   7CCg-C6?cp932cp949)	shift_jisshift_jis_2004euc_kr_DEAD_HEAT_SUPERSETSg{Gzt?      i @  c                 j    [         R                  " U 5      nUc  g[        UR                  5      nX"* -  $ )zHReturn the lowest era bit for *encoding* (lower = more prevalent today).i   @)r   rj   intera)rf   infor   s      rn   	_era_rankr   I  s/    <<!D|
dhh-C:rp   ru   z
str | Nonec                 f   [        5       R                  U5      nU(       d  gU S[         n[        5       nUS   n[	        S[        U5      5       H,  nXG   nUS:  d  US:  a  UR                  US-  U-  5        UnM.     U(       d  gU H%  u  pnUb  X:w  a  M  U H  nX   (       d  M      g   M'     g)u  Return True if *encoding*'s winning model weights a high-byte bigram present in *data*.

A candidate whose model assigns zero weight to every non-ASCII bigram in
the data earned its statistical score purely from ASCII bigrams — noise
that cannot distinguish encodings.  Only the variant that actually won
(*language*) counts: another language's variant having weight for those
bytes says nothing about why *this* result is on top.  Only called on
dead heats, so the Python-level scan of the (capped) data is off the
hot path.
FNr   rs   r[      T)r   rj   _EVIDENCE_SCAN_MAX_BYTESsetrangerk   add)rg   rf   ru   variantswindowseenprevr   blangtable_keyidxs                rn   _has_high_byte_evidencer   R  s     ""8,H++,FUD!9D1c&k"I4<19HHdai1_%	 #
 %TD$4Czz  & rp   c                    U(       a  US   OSnUb  UR                   b  [        U5      S:  a  U$ [        XR                   UR                  5      (       a  U$ Sn[	        UR                   5      n[        S[        U5      5       HY  nX   nUR                   c  M  UR                  UR                  -
  [        :  a    O$[	        UR                   5      nXt:  d  MU  UnUnM[     US:X  a  U$ X   n[        UR                   UR                  UR                  UR                  5      n	[        U5       V
Vs/ s H  u  pX:w  d  M  UPM     nn
nU	/UQ$ s  snn
f )aK  Break statistical dead heats in favour of the more prevalent era.

When several encodings score within :data:`_DEAD_HEAT_EPSILON` of the
top result and the top result's models carry no weight for any high-byte
bigram in the data, the ranking is an artifact of ASCII-bigram noise.
Promote the candidate from the most prevalent era (modern web > legacy
ISO > Mac > regional > DOS > mainframe) so evidence-free dead heats
resolve to the likeliest real-world answer.  A top result whose models
do weight observed high-byte bigrams won on real evidence and is kept,
however small its margin.
r   N   rs   )rf   rk   r   ru   r   r   rt   _DEAD_HEAT_EPSILONr	   rv   r   )rg   rq   topbest_idx	best_rankr   ry   rankchosenrz   jrests               rn   _prefer_prevalent_on_dead_heatr   s  s-     '!*TC
{cll*c'lQ.>t\\3<<@@H#,,'I1c'l#J::>>ALL(+==$IH $ 1}F&:J:JH $G,>,$!A,D>t ?s   0E?Er   c                     X   n[        UR                  U S   R                  UR                  UR                  5      n[        U 5       VVs/ s H  u  pEXA:w  d  M  UPM     nnnU/UQ$ s  snnf )zDMove ``results[i]`` to the top, carrying the current top confidence.r   )r	   rf   rt   ru   rv   r   )rq   r   ry   rz   r   r{   r   s          rn   _promote_to_topr     sh    
A	

GAJ))1::q{{H $G,7,$!A,D7t 8s   A)A)c                    U(       a  US   OSnUb  UR                   b  [        U5      S:  a  U$ [        R                  UR                   5      nUc  U$ [	        S[        U5      5       HZ  nX   nUR
                  UR
                  -
  [        :  a    U$ UR                   U:X  d  M=  [        X5      (       d  MO  [        X5      s  $    U$ )zAPromote a Windows superset over its base encoding on a dead heat.r   Nr   rs   )	rf   rk   r   rj   r   rt   r   r   r   )rg   rq   r   supersetr   ry   s         rn   _promote_superset_on_dead_heatr     s    
  '!*TC
{cll*c'lQ.>#''5H1c'l#J>>ALL(+== N ::!&;D&K&K"7.. $ Nrp   g{Gz?g333333?c                    U (       a  U S   OSnUbD  UR                   b7  UR                  [        ;  d#  UR                  [        :  d  [        U 5      S:  a  U $ [        S[        U 5      5       Hh  nX   nUR                  UR                  -
  [        :  a    U $ UR                   b  UR                  c  MG  UR                  [        ;  d  M]  [        X5      s  $    U $ )a  Demote a rare-language winner that leads a prevalent rival by a coin flip.

Fires only when the winner's language is in
:data:`~chardet.models.RARE_LANGUAGES`, its absolute confidence is
inside the evidence-free zone, and a prevalent-language candidate sits
within :data:`_RARE_ARBITRATION_MARGIN`.  Genuine rare-language text fails
both gates: even short files score confidently, and their entire
neighborhood is same-language variants.
r   Nr   rs   )	rf   ru   r   rt    _RARE_ARBITRATION_MAX_CONFIDENCErk   r   _RARE_ARBITRATION_MARGINr   )rq   r   r   ry   s       rn   _arbitrate_rare_languager     s      '!*TC<<<<~->>==w<!1c'l#J>>ALL(+CC
 N	 ::!3::^+"7.. $ Nrp   c                    U(       a  US   OSnUb  UR                   b  [        U5      S:  a  U$ [        UR                   5      [        :X  a  U$ UR                  S:X  a  U$ U R                  S5      S:  d  U R                  S5      [        :  a  U$ [        S[        U5      5       H  nX   nUR                  UR                  -
  [        :  a    U$ UR                   c  M:  [        UR                   5      [        :X  d  MY  [        S UR                  UR                  4 5       5      n[        XR                   UR                   U5      UR                   :X  a    U$ [        X5      s  $    U$ )	u.  Promote a classic-Mac candidate when line endings are bare ``\r``.

Classic Mac OS is the only platform that terminated lines with a lone
carriage return, so data with several ``\r`` bytes and no ``\n`` is
near-certainly Mac-era text.  When a LEGACY_MAC candidate scores within
:data:`_CR_MAC_BAND` of a non-Mac top result, promote it — unless the
pair has a distinguishing-byte map and the byte-level evidence says the
current top wins: a platform prior must not overturn direct evidence
that confusion resolution may have just used to establish the top.
r   Nr   zxx   
   rs   c              3   .   #    U  H  oc  M  Uv   M     g 7f)N )r   r   s     rn   r   2_promote_mac_on_cr_line_endings.<locals>.<genexpr>  s      !;!;s   	)rf   rk   r   _LEGACY_MAC_ERAru   findcount_CR_MAC_MIN_LINESr   rt   _CR_MAC_BAND	frozensetr
   r   )rg   rq   r   r   ry   langss         rn   _promote_mac_on_cr_line_endingsr     s<     '!*TC
{cll*c'lQ.>/1
 ||uyy1

5 14E E1c'l#J>>ALL(<7" N! ::!i

&;&N  "%,,

!; E &dLL!**eL<<  N #7..' $( Nrp   c                 |    [        X5      (       d  g[        R                  " U5      nUSL =(       d    [        X5      $ )aL  Check that *data* decodes completely under *encoding* and its output name.

The flip's promise is that the caller's ``data.decode(result)`` works,
and the caller sees the *public* name: ``compat_names=True`` (the
default) can remap to a strictly narrower codec (``euc_jis_2004`` is
reported as ``EUC-JP``), so a rival must decode under both names to be
promoted.  ``prefer_superset=True`` can also narrow (cp125x leaves
codepoints undefined that iso-8859-x maps), but that is an opt-in
output transform applied to every detection, not only promoted ones,
and is out of this step's hands.
FN)r   r   rj   )rg   rf   displays      rn   _decodes_under_public_namesr   &  s7     d--)Gd??0??rp   input_truncatedc                   U(       a  US   OSnU(       dS  UbP  UR                   bC  [        U5      S:  d4  [        S U SS  5       5      (       a  [        XR                   5      (       d  U$ [	        S[        U5      5       H;  nX   nUR                   b  [        XR                   5      (       d  M0  [        X5      s  $    U$ )a  Promote a strictly decoding rival over a winner with no real evidence.

Byte-validity filtering runs incremental decoders with ``final=False``,
tolerating an incomplete multi-byte sequence at the end because detection
input is often a prefix of a larger whole.  When chardet examined the
caller's *entire* input, that tolerance can hand back an encoding the
caller's very next ``data.decode()`` will reject --- a four-byte
``iso-8859-1`` word ending in ``0xE1`` detected as utf-8 (issue #380).

Fires only when the input was not truncated by chardet itself (the
orchestrator's ``max_bytes`` slice or ``UniversalDetector``'s buffer
cap --- either way these bytes are not the whole story and the caller
was told so by *input_truncated*), the tail can actually hold a
dangling sequence (a high byte in the final four, multi-byte winner),
and the winner's tolerant decode is **non-empty pure ASCII** --- its
only multi-byte evidence is the dangling tail itself.  An empty
tolerant decode (the whole input is one clipped sequence) is zero
evidence, not ASCII evidence, and disqualifies the flip.  The
best-ranked rival that decodes the input completely under both its
internal and public names then takes the top slot, regardless of the
confidence gap: an all-ASCII-evidence winner detected nothing the
rival did not also detect, and any statistical lead it holds comes
from ASCII bigrams the rival matched equally well.  The scan sees the
ranking as given, which under ``full_ranking=False`` is pruned; if no
listed rival decodes, the winner stands (measured across a 648-case
accent-final sweep, the pruned ranking always carried a decodable
rival).

The pure-ASCII condition is what makes the unconditional flip safe.  A
short mid-character CJK cut has a correct answer that cannot decode the
input --- flipping it to whichever single-byte codec happens to decode
the bytes trades a right answer for a wrong one, and a 5-40 byte sweep
measured exactly that under a gap-based rule (34 correct CJK answers
lost, Big5 becoming cp1125).  Such a winner has decoded real multi-byte
characters and keeps its ranking; with the pure-ASCII condition in
place, the same sweep measures zero lost answers at any gap.
r   Nr   c              3   *   #    U  H	  oS :  v   M     g7f)r[   Nr   )r   r   s     rn   r   +_prefer_decodable_on_tie.<locals>.<genexpr>p  s     0i9is   rs   )rf   rk   anyr   r   r   r   )rg   rq   r   r   r   ry   s         rn   _prefer_decodable_on_tier   8  s    V  '!*TC;<<w<! 0d23i000.t\\BB1c'l#J::%@zz%R%Rw**	 $
 Nrp   Fr   c                    [        X5      n[        X5      n[        U5      n[        X5      n[	        X5      n[        X5      n[        X5      n[        XUS9$ )a  Apply rank corrections to the statistically scored results.

Steps run in sequence, weakest evidence first: dead-heat priors
(superset preference, era prevalence), then confusion-group resolution,
niche Latin demotion, and KOI8-T promotion (byte-level evidence), and
finally the classic-Mac line-ending promotion (platform evidence that
should override the priors).  The decode-safety tiebreak runs last of
all: whatever the ranking settled on, a winner that cannot decode the
caller's complete input, and whose own evidence is nothing but the
undecodable tail, yields to the best-ranked rival that can decode it.

:param data: The raw byte data the results were produced from.
:param results: A list of :class:`DetectionResult` ranked by confidence.
:param input_truncated: True when the caller's input was longer than
    ``max_bytes``, i.e. *data* is a chardet-made slice rather than the
    caller's whole input.
:returns: A new list (or the same list) with rank corrections applied.
r   )r   r   r   r   r~   r   r   r   )rg   rq   r   s      rn   postprocess_resultsr   |  sW    0 -T;G,T;G&w/G&t5G!$0GT+G-d<G#D?SSrp   N)8__doc__chardet._utilsr   r   r   chardet.modelsr   r   chardet.output_namesr   chardet.pipeliner	   chardet.pipeline.confusionr
   r   chardet.registryr   r   r   str__annotations__r?   r   rF   rI   rX   rZ   dictrc   itemsbytesrd   re   boolro   listr~   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   )encbyte_sets   00rn   <module>r      s  $ 
 9 . , & +4+ 3  /8/1/ Ys^ 1n /8 "/ Ys^ "T 09(0 in  -6- 9S> B .-**	3 d3	#./  *3L* 	#  /C.H.H.J&.J]SCx.J& $sEz"  45 5:S : :$ :
/" 
/@
/" 
/J   ( d38n      !   % 3 , SW B'
'/"' 
/'TT/2 s tO?T 
/" 
/J   
 $(  /"	/@/
//"/ 
//d@e @s @t @$A
A/"A 	A
 
/AP "	T
T/"T 	T
 
/TW&s   G