
    *pj,i                        % S r SSKrSSKrSSKrSSKrSSKrSSKrSSKrSSK	r	SSK
Jr  SSKJrJr  SSKJrJr  \R                  " S5      \R                  " S5      4r\R&                  R)                  S5      r\R,                  " S5      R.                  r\R,                  " S	5      R.                  r\" S
 \" S5       5       5      rSrSr0 r\ \!\!4   \"S'   \RF                  " 5        H7  r$\%" \$RL                  5      S:X  d  M  \$RL                  S   \\$RN                  '   M9      S2S\S\(S\)\!   S\(S\ \!\4   4
S jjr*S\S\+\ \!\4   \ \!\,4   4   4S jr-\R\                  S\+\ \!\4   \ \!\,4   4   4S j5       r/S\ \!\4   4S jr0S\ \!\4   S\ \!\)\+\!S-  \\!4      4   4S jr1\R\                  S\ \!\)\+\!S-  \\!4      4   4S j5       r2S\!S\!S-  4S jr3S\!S\44S jr5S\ \!\,4   4S jr6\R\                  S\ \!\4   4S  j5       r7\R\                  S\4S! j5       r8 " S" S#5      r9 S3S$\9S%S&S'\!S\,4S( jjr:\;" 1 S)k5      r<\;\!   \"S*'   S+r=S,r>S-r? S4S.S/.S\S\!S$\9S-  S0\4S\+\,\!S-  4   4
S1 jjjr@g)5zModel loading and bigram scoring utilities.

Note: ``from __future__ import annotations`` is intentionally omitted because
this module is compiled with mypyc, which does not support PEP 563 string
annotations.
    N)_kernel)
dot_packedpack_profile)REGISTRYlookup_encodingi)z.soz.pydz>Iz>dc              #   4   #    U  H  oS ;   a  SOSv   M     g7f))    	   
                  r   N ).0bs     S/var/www/html/pdf-tiff/venv/lib/python3.13/site-packages/chardet/models/__init__.py	<genexpr>r   9   s       ISA8	8Aa?s      s   CMD2s   CRM1_SINGLE_LANG_MAPr   dataoffsetnames
chunk_sizereturnc                    [        U5      nUS-  n[        R                  " 5       n0 n[        5       nSn	Un
[        U 5      nSn[        U5      U:  a  S[        U5      -
  nUR                  (       a  UR                  UR                  U5      nONX:  a&  X
X-    nU
[        U5      -  n
UR                  X5      nO#U(       d  SnUR                  5       nU(       d  OnOOlU	[        U5      -  n	X-  n[        U5      S:X  a*  [        U5      Xr[        U5         '   UR                  5         O[        U5      S:  a  O[        U5      U:  a  M  [        U5      U:X  a  SnUR                  (       a  UR                  UR                  S5      nU(       d[  UR                  (       dJ  X:  aE  X
X-    nU
[        U5      -  n
UR                  US5      nU(       d  UR                  (       d  X:  a  ME  U(       d(  UR                  (       d  U(       d  UR                  5       nU	[        U5      -  n	X:w  d  [        U5      U:w  a  SU	 SU 3n[        U5      eU$ )	u  Decompress the model tables from ``data[offset:]``, one per name.

Each model is stored as its own bytes object rather than a memoryview
slice of one big blob: mypyc compiles bytes indexing in the scoring hot
loop to a native C array access, while memoryview indexing goes through
a boxed generic call.  Decompression is incremental, one 64 KB table at
a time — materializing the whole multi-megabyte blob and slicing it
would transiently double the allocation and strand the freed pages in
process RSS.  Trailing compressed bytes are ignored, as with whole-blob
``zlib.decompress``; the decompressed size is validated instead.

:raises ValueError: If the decompressed size is not exactly
    ``len(names) * 65536``.
   r   FT    r   z&corrupt models.bin: decompressed size z != expected )lenzlibdecompressobj	bytearrayunconsumed_tail
decompressflushbytescleareof
ValueError)r   r   r   r   
num_modelsexpected_sizedecompmodelstableproducedposendflushedneedpiecechunkextramsgs                     r   _decompress_tablesr:   I   s   " UJ&M!F!FKEH
C
d)CG
f+

"s5z!!!%%f&<&<dCEYs/0E3u:C%%e2EGLLNE  CJu:).uFV%&KKMZ%+ f+

", 6{j 
 !!%%f&<&<a@E

sys/0E3u:C%%eQ/E 

sy VZZLLNECJ CK:$=4XJ ?(/+ 	 oMr    c                 `    U SS [         :w  a  Sn[        U5      eSn[        X5      u  nUS-  nUS:  a  SU S3n[        U5      e/ n0 n[        U5       Hl  n[        X5      u  nUS-  nUS:  a  SU S	3n[        U5      eXX'-    R	                  S
5      nX'-  n[        X5      u  n	US-  nUR                  U5        XU'   Mn     [        XU5      n
X4$ ! [        R                   a  nSU 3n[        U5      UeSnAf[        R                  [        4 a  nSU 3n[        U5      UeSnAff = f)zParse the v2 dense zlib-compressed models.bin format.

:param data: Raw bytes of models.bin (must be non-empty).
:returns: A ``(models, norms)`` tuple.
:raises ValueError: If the data is corrupt or truncated.
N   z&corrupt models.bin: missing CMD2 magici'  zcorrupt models.bin: num_models=z exceeds limitr   zcorrupt models.bin: name_len=z exceeds 256zutf-8   zcorrupt models.bin: )	_V2_MAGICr+   _unpack_uint32rangedecode_unpack_float64appendr:   r"   errorstructUnicodeDecodeError)r   r9   r   r,   r   norms_name_lennamenormr/   es               r   _parse_models_binrM      sd   #%8y :CS/!&t4!3J<~NCS/!"$z"A(6KXaKF#~5hZ|L o%!23::7CDF%d3GTaKFLL$K # $D%8 = :: %$QC(o1$LL,- %$QC(o1$%s$   CC D-)C::D-D((D-c                      [         R                  R                  S5      R                  S5      n U R	                  5       nU(       d  [
        R                  " S[        SS9  0 0 4$ [        U5      $ )zcLoad and parse models.bin, returning (models, norms).

Cached: only reads from disk on first call.
chardet.models
models.binuX   chardet models.bin is empty — statistical detection disabled; reinstall chardet to fix   
stacklevel)		importlib	resourcesfilesjoinpath
read_byteswarningswarnRuntimeWarningrM   refr   s     r   _load_models_datar^      sb     


#
#$4
5
>
>|
LC>>D'		
 2vT""r    c                      [        5       S   $ )zLoad all bigram models from the bundled models.bin file.

Each model is a bytes object of length 65536 (256*256).
Index: (b1 << 8) | b2 -> weight (0-255).

:returns: A dict mapping model key strings to 65536-byte lookup tables.
r   r^   r   r    r   load_modelsra      s     q!!r    r/   c                    0 nU R                  5        H<  u  p#UR                  SS5      u  pEUR                  U/ 5      R                  XCU45        M>     [	        U5       H   n[        U5      nUc  M  Xq;  d  M  X   X'   M"     U$ )zBuild a grouped index from a models dict.

:param models: Mapping of ``"lang/encoding"`` keys to 65536-byte tables.
:returns: Mapping of encoding name to ``[(lang, model, model_key), ...]``.
/r   )itemssplit
setdefaultrC   listr   )r/   indexkeymodellangencenc_name	canonicals           r   _build_enc_indexro      s     =?Elln
IIc1%	b!(($s);< % K#H-	 Y%;$E  
 Lr    c                  (    [        [        5       5      $ )zTReturn a pre-grouped index mapping encoding name -> [(lang, model, model_key), ...].)ro   ra   r   r    r   get_enc_indexrq      s     KM**r    encodingc                 ,    [         R                  U 5      $ )zReturn the language for a single-language encoding, or None.

:param encoding: The canonical encoding name.
:returns: An ISO 639-1 language code, or ``None`` if the encoding is
    multi-language.
)r   getrr   s    r   infer_languagerv      s     ))r    c                     U [        5       ;   $ )zReturn True if the encoding has language variants in the model index.

:param encoding: The canonical encoding name.
:returns: ``True`` if bigram models exist for this encoding.
)rq   ru   s    r   has_model_variantsrx   	  s     }&&r    c                      [        5       S   $ )zAReturn cached L2 norms for all models, keyed by model key string.r   r`   r   r    r   _get_model_normsrz     s    q!!r    c                    ^ [        5       n [        R                  R                  S5      n UR	                  S5      R                  5       n[        R                  " UR	                  S5      R                  5       5      R                  5       nSnUSS [        :X  aW  USU U:X  aN  [        U5      U[        U 5      S-  -   :X  a0  [        U 5       VVs0 s H  u  pVXbXES-  -   XES	-   S-  -    _M     snn$ U (       a  [        R                  " S
[         SS9  U R#                  5        VV^s0 s H'  u  nmU[%        U4S j['        SSS5       5       5      _M)     snn$ ! [        [        4 a    SnSn Nf = fs  snnf s  snnf )u  Return per-model row-maximum tables for upper-bound prescreening.

For each model, entry ``b1`` of its 256-byte table holds the maximum
weight in the model's row for lead byte ``b1``.  Because every bigram
weight is bounded by its row maximum, a dot product against the row
maxima (256 terms) upper-bounds the dot product against the full table
(65536 terms) — statistical scoring uses this to rule out candidate
models without scoring them fully.

Loads the precomputed ``rowmax.bin`` (written by ``scripts/train.py`` in
the same model order as ``models.bin``).  The file starts with a ``CRM1``
magic and the SHA-256 of the ``models.bin`` it was derived from: a stale
or mismatched file would silently under-estimate row maxima and break
the upper bound that pruning depends on, so anything that does not match
the *current* ``models.bin`` byte-for-byte is rejected and the tables
are derived from the models directly (slower, but always correct).
rO   z
rowmax.binrP   r    $   Nr<   r   r   zpchardet rowmax.bin is missing or does not match models.bin; deriving row maxima from the models (slower startup)rQ   rR   c              3   D   >#    U  H  n[        TXS -    5      v   M     g7f)r   N)max)r   startr0   s     r   r   get_rowmax.<locals>.<genexpr>G  s$     U@Tu3uUS[122@Ts    r   r   )ra   rT   rU   rV   rW   rX   hashlibsha256digestFileNotFoundErrorOSError_ROWMAX_MAGICr!   	enumeraterY   rZ   r[   rd   r(   r@   )r/   rV   r   models_digestheader_sizer   ri   r0   s          `r   
get_rowmaxr     s{   & ]F%%&67E~~l+668NN<(335

&( 	 KRaM!;=0Is6{S'888
 $F+
+ kG+kUcM.IJJ+
 	
 C		
 !,,.(JC 	UUaPS@TUUU( + w' 
s   A E E3'.E9E0/E0c                      [         R                  R                  S5      R                  S5      n U R	                  5       n[        U5      S:w  a,  [        R                  " S[        U5       S3[        SS9  SS-  $ U$ )	u  Return a 65536-byte IDF weight table for bigram profile construction.

Loads a precomputed table from ``idf.bin`` (generated at training time).
For each bigram index, the weight reflects how discriminative that bigram
is across all models:

- Bigrams in every model (common ASCII) → weight 1 (minimal signal)
- Bigrams in one model → weight 255 (maximum signal)
- Bigrams not in any model → weight 1 (unknown, treat as neutral)
rO   zidf.binr   z chardet idf.bin has wrong size (z"), falling back to uniform weightsrQ   rR      )	rT   rU   rV   rW   rX   r!   rY   rZ   r[   r\   s     r   get_idf_weightsr   L  sv     


#
#$4
5
>
>y
IC>>D
4yE.s4yk :. .		
 Kr    c            
           \ rS rSrSrSrS\SS4S jrS\\	   S	\\	   S
\\	   S\	SS4
S jr
\S\\	\	4   SS 4S j5       rSrg)BigramProfileie  u  Pre-computed bigram frequency distribution for a data sample.

Computing this once and reusing it across all models reduces per-model
scoring from O(n) to O(distinct_bigrams).

Each bigram is weighted by its IDF (inverse document frequency) across all
models — bigrams unique to few models get high weight, bigrams common to
all models get weight 1.  ``nonzero`` lists the indices carrying weight,
in first-encounter order.

The weights are reachable two ways, and which one a profile fills
depends on how it was built:

* ``idx_arr``/``val_arr`` — parallel ``array('i')`` buffers, what
  :func:`score_with_profile` reads for a streaming profile.  Filled by
  the streaming constructor only.
* ``values`` — a plain list parallel to ``nonzero``, filled by
  :meth:`from_weighted_freq` for the small focused profiles confusion
  resolution builds, which are scored inline instead.
``row_freq`` aggregates the weights by lead byte (256 entries) and
``nonzero_rows`` lists the lead bytes with non-zero total; together with
per-model row maxima (:func:`get_rowmax`) they let statistical scoring
compute a cheap upper bound on a model's score.

**Input limit.** Weights are packed as int32, which holds any value an
input of at most 16 MB can produce (``255`` per occurrence against a
2.1-billion ceiling).  Detection truncates to ``max_bytes`` long before
that; construct a profile directly from a larger buffer and packing
raises :exc:`OverflowError`.
)	freqidx_arr
input_normnonzerononzero_rowsrow_freqval_arrvalues
weight_sumr   r   Nc                    [        U5      S-
  nUS::  aE  / U l        / U l        / U l        [        u  U l        U l        / U l        / U l        SU l	        SU l
        g[        5       nS/S-  n/ nSn[        U5       HX  nX   nXS-      n	X:X  a  [        U   (       a  M#  US-  U	-  n
X:   nXJ   S:X  a  UR                  U
5        XJ==   U-  ss'   Xk-  nMZ     U R                  XE/ U5        g)a  Compute the bigram frequency distribution for *data*.

Each bigram is weighted by its IDF (inverse document frequency) across
all loaded models.  Bigrams unique to few models get high weight;
bigrams common to all models get weight 1.

:param data: The raw byte data to profile.
r   r           Nr   r=   )r!   r   r   r   _EMPTY_PACKEDr   r   r   r   r   r   r   r@   _ASCII_WHITESPACE_TABLErC   _finish)selfr   total_bigramsidfr   r   w_sumr   b1b2idxws               r   __init__BigramProfile.__init__  s     D	AA#%DI&(DL%'DK)6&DL$,')DM+-D#$DO%(DO#+}%AB!eB
 x3B77b.CAyA~s#INIJE &$ 	TB.r    r   r   r   r   c                 `   X l         X0l        X@l        SnS/S-  nU(       a9  [        [	        U5      5       H   nX7   nXXU-  -  nXbU   S-	  ==   U-  ss'   M"     O#U H  n	X   nXXU-  -  nXiS-	  ==   U-  ss'   M     [
        R                  " U5      U l        X`l        [        S5       V
s/ s H  oU
   (       d  M  U
PM     sn
U l	        U(       d  [        (       d  [        u  U l        U l        O[        X!5      u  U l        U l        U(       d  [        (       a  / U l        gUU l        gs  sn
f )a  Store the frequency data and derive the norm and row aggregates.

Single finalization path shared by both constructors so pruning
fields cannot silently diverge between them.

Exactly one of *freq* and *values* carries the weights, never both.
The streaming constructor passes the dense *freq* table it had to
build and leaves *values* empty; the sparse constructor fills
*values* parallel to *nonzero* and passes no table, which is what
lets it skip a 65536-entry allocation per call.

Neither is retained.  *freq* is read here to derive the norm, the row
aggregates and the packed buffers, and is then dropped -- scoring
reads ``idx_arr``/``val_arr`` or ``values``, never the dense table,
so keeping it alive would pin 512 KB per profile for nothing.
r   r   r=   N)r   r   r   r@   r!   mathsqrtr   r   r   _KERNEL_COMPILEDr   r   r   r   r   )r   r   r   r   r   norm_sqr   r   vr   r   s              r   r   BigramProfile._finish  s   . $ cCi3w<(Iq5 q)Q.) )
 Iq5 "a'"  ))G, */*E*BR*E )))6&DL$,)5g)D&DL$,!%5%5B	D	 Fs   ,D+=D+weighted_freqc                     U " S5      n/ n/ nSnUR                  5        H4  u  pgU(       d  M  UR                  U5        UR                  U5        XW-  nM6     UR                  / X4U5        U$ )u  Create a BigramProfile from pre-computed weighted frequencies.

Computes ``weight_sum`` and ``input_norm`` from *weighted_freq* to
ensure consistency between the stored fields.

Deliberately does not build the dense 65536-entry ``freq`` table
the streaming constructor uses.  Callers here pass a handful of
bigrams — confusion resolution's focused profiles hold a median of
eight — and allocating a 65536-element list per call cost about
24us, which measured as roughly 40% of the whole bigram-rescore
stage.  Scoring reads ``nonzero``/``values``, so the table is
never needed.

:param weighted_freq: Mapping of bigram index to weighted count.
:returns: A new :class:`BigramProfile` instance.
r    r   )rd   rC   r   )clsr   profiler   r   r   r   counts           r   from_weighted_freq BigramProfile.from_weighted_freq  sl    $ c('--/JCus#e$	 0
 	GU3r    )__name__
__module____qualname____firstlineno____doc__	__slots__r(   r   rg   intr   classmethoddictr   __static_attributes__r   r    r   r   r   e  s    >
I,/U ,/t ,/\1A3i1A c1A S		1A
 1A 
1Af tCH~ /  r    r   r   rj   zbytes | bytearray | memoryview	model_keyc                    [        U[        5      (       d  [        U5      nU R                  S:X  a  g[        5       nU(       a  UR	                  U5      OSnUc>  Sn[        S5       H  nX   nU(       d  M  XWU-  -  nM     [        R                  " U5      nUS:X  a  gU R                  nU R                  n	U	(       a,  Sn
[        [        U5      5       H  nXX      X   -  -  n
M     OP[        (       a"  [        U R                  U R                  U5      n
O#Sn
U R                  nU H  nXU   X   -  -  n
M     XU R                  -  -  $ )aZ  Score a pre-computed bigram profile against a single model using cosine similarity.

``bytearray``/``memoryview`` tables are accepted for compatibility but
copied to ``bytes`` first: the narrow type lets mypyc compile the
dot-product loop with native byte indexing (and keeps compiled and
pure-Python installs accepting the same argument types).
r   Nr   r   )
isinstancer(   r   rz   rt   r@   r   r   r   r   r!   r   r   r   r   r   )r   rj   r   rG   
model_normsq_sumr   r   r   r   dotr   r   s                r   score_with_profiler     s2    eU##eS E)29%JuAAqa%  YYv&
SooG^^F s7|$A$vy00C %		'//5A ||C:	))C w11122r    >   brcygagdRARE_LANGUAGESzxx   gQ?F)demote_thin_rarer   c                l   U (       d  Uc  g[        5       nUR                  U5      nUc  gUc  [        U 5      nSnSnSnSn	U HG  u  pn[        X+U5      nX:  a  UnU
nU(       d  M$  U
[        ;  d  M0  U
[
        :w  d  M<  X:  d  MC  UnU
n	MI     U(       a  Ub  U[        ;   a  U	b  Xh-
  [        :  a  U	nXg4$ )a  Score data against all language variants of an encoding.

Returns (best_score, best_language). Uses a pre-grouped index for O(L)
lookup where L is the number of language variants for the encoding.

If *profile* is provided, it is reused instead of recomputing the bigram
frequency distribution from *data*.

:param data: The raw byte data to score.
:param encoding: The canonical encoding name to match against.
:param profile: Optional pre-computed :class:`BigramProfile` to reuse.
:param demote_thin_rare: Pass true only when the caller has judged the
    *original* input thin (under :data:`_THIN_RARE_MAX_BYTES` before
    any transcoding).  A :data:`RARE_LANGUAGES` winner that leads the
    best prevalent-language variant by less than the measured noise
    band is then reported under the prevalent language instead.  The
    length judgment deliberately lives with the caller: this function
    may receive transcoded bytes or a profile without its source data,
    so ``len(data)`` here is not a reliable proxy for input size.
    Language-fill callers pass this; encoding-ranking callers must not,
    so that candidate ordering stays byte-identical.
:returns: A ``(score, language)`` tuple.  The score is always the best
    cosine similarity across variants; the language matches it except
    when ``demote_thin_rare`` fires, in which case the label is the
    best prevalent-language variant's while the score remains the
    rare winner's.
N)r   Nr   )rq   rt   r   r   r   ART_LANGUAGE_THIN_RARE_MARGIN)r   rr   r   r   rh   variants
best_score	best_langbest_prevalentbest_prevalent_langrk   rj   r   ss                 r   score_best_languager   ]  s    D GOOEyy"H%J IN&*"*Ywy9>JIN*$"N"& #+ 	!'+'*;;'	  r    )i   ) )N)Ar   array	functoolsr   importlib.resourcesrT   r   rE   rY   r"   chardetr   chardet._kernelr   r   chardet.registryr   r   r   __file__endswithr   Structunpack_fromr?   rB   r(   r@   r   r>   r   r   r   str__annotations__r   _encr!   	languagesrJ   r   rg   r:   tuplefloatrM   cacher^   ra   ro   rq   rv   boolrx   rz   r   r   r   r   	frozensetr   r   _THIN_RARE_MAX_BYTESr   r   r   r    r   <module>r      s            4 6 S!5;;s#34  ##,,_= t$00--%11    INs    	  $& $sCx. %OOD
4>>a&*nnQ&7#  CIE
EE%)#YE<?E	#u*EP.
.
4U
T#u*--..b #5c5j!14U
3C!CD # #("T#u*% "e	#tE#*eS012
23. +tCeC$Js,B&C!DDE + +
*S *S4Z *' ' '"$sEz* "
 1De$ 1 1h   0i i^ -3-3+-3 -3 	-3h "++C!D	# D
      %)H!
 #H!
H!H! T!H!
 H! 5#*H!r    