ó
    †ñ:iÒ8  ã                   ó  • S SK r S SKrS SKrS SKJr  SSKJr  SSK	J
r
Jr  SSKJr  SSKJrJr  SS	KJr  \ R$                  " S
5      r " S S\5      r\S 5       r\S 5       r " S S\5      r " S S\5      r " S S\5      rg)é    N)Únjité   )Úutils)ÚDeserializerÚ
Serializer)ÚMaskedModel)ÚDimensionErrorÚInvalidClusteringErroré   )ÚMaskerÚshapc                   ó\   ^ • \ rS rSrSrS	S jrS rS rU 4S jr\	S
U 4S jj5       r
SrU =r$ )ÚTabularé   z2A common base class for Independent and Partition.c                 óü  • SU l         [        U[        R                  5      (       a$  UR                  U l        UR                  nSU l         [        U[        5      (       aN  SU;   aH  UR                  SS5      U l	        UR                  SS5      U l
        [        R                  " US   S5      n[        US5      (       a)  UR                  S   U:”  a  [        R                   " X5      nXl        X0l        X@l        X l        Ub  Ub  [+        S5      eUbh  [        U[,        5      (       a  [        R.                  " XS	9U l        O1[        U[        R0                  5      (       a  X0l        O[3        S
5      eSU l        OUb  X@l        SU l        OSU l        SU l        UR5                  5       U l        [        R8                  " UR                  S   [:        S9U l        U R"                  R                  U l        SU l        g)aN  This masks out tabular features by integrating over the given background dataset.

Parameters
----------
data : np.array, pandas.DataFrame
    The background dataset that is used for masking.

max_samples : int
    The maximum number of samples to use from the passed background data. If data has more
    than max_samples then shap.utils.sample is used to subsample the dataset. The number of
    samples coming out of the masker (to be integrated over) matches the number of samples in
    the background dataset. This means larger background dataset cause longer runtimes. Normally
    about 1, 10, 100, or 1000 background samples are reasonable choices.

clustering : string or None (default) or numpy.ndarray
    The distance metric to use for creating the clustering of the features. The
    distance function can be any valid scipy.spatial.distance.pdist's metric argument.
    However we suggest using 'correlation' in most cases. The full list of options is
    `braycurtis`, `canberra`, `chebyshev`, `cityblock`, `correlation`, `cosine`, `dice`,
    `euclidean`, `hamming`, `jaccard`, `jensenshannon`, `kulsinski`, `mahalanobis`,
    `matching`, `minkowski`, `rogerstanimoto`, `russellrao`, `seuclidean`,
    `sokalmichener`, `sokalsneath`, `sqeuclidean`, `yule`. These are all
    the options from scipy.spatial.distance.pdist's metric argument.

FTÚmeanNÚcovr   ÚshapezKYou cannot pass both 'clustering' and 'partition'. Please provide only one.)ÚmetriczoUnknown clustering given! Make sure you pass a distance metric as a string, or a clustering as a numpy.ndarray.r   ©Údtype) Úoutput_dataframeÚ
isinstanceÚpdÚ	DataFrameÚcolumnsÚfeature_namesÚvaluesÚdictÚgetr   r   ÚnpÚexpand_dimsÚhasattrr   r   ÚsampleÚdataÚ
clusteringÚ	partitionÚmax_samplesÚ
ValueErrorÚstrÚhclustÚndarrayr
   ÚcopyÚ_masked_dataÚzerosÚboolÚ
_last_maskÚsupports_delta_masking)Úselfr%   r(   r&   r'   s        ÚX/srv/projetos/modelo_ml_acdoc/venv/lib/python3.13/site-packages/shap/maskers/_tabular.pyÚ__init__ÚTabular.__init__   s“  € ð4 !&ˆÔÜ�dœBŸL™L×)Ñ)Ø!%§¡ˆDÔØ—;‘;ˆDØ$(ˆDÔ!ä�dœD×!Ñ! f°£nØŸ™ ¨Ó.ˆDŒIØ—x‘x  tÓ,ˆDŒHÜ—>’> $ v¡,°Ó2ˆDä�4˜×!Ñ! d§j¡j°¡m°kÓ&AÜ—<’< Ó2ˆDàŒ	Ø$ŒØ"ŒØ&Ôð Ñ! iÑ&;ÜÐjÓkÐkØÑ#Ü˜*¤c×*Ñ*Ü"'§,¢,¨tÑ"G�•Ü˜J¬¯
©
×3Ñ3Ø",•ä,ð Fóð ð "ˆD�NØÑ"Ø&ŒNØ"ˆD�Oà"ˆDŒOØ!ˆDŒNð !ŸI™I›KˆÔÜŸ(š( 4§:¡:¨a¡=¼Ñ=ˆŒØ—Y‘Y—_‘_ˆŒ
Ø&*ˆÕ#ó    c                 óÀ  • U R                  X5      n[        UR                  5      S:w  d*  UR                  S   U R                  R                  S   :w  a  [	        S5      e[
        R                  " UR                  [
        R                  5      (       GaA  U R                  U5      ) n[
        R                  " [        U5      [        S9nUS:¬  R                  5       n[
        R                  " XPR                  S   4[        S9n[
        R                  " XPR                  S   -  U R                  S   45      nSU R                  S S & U R                  U R                  S S & [!        UUUUU R                  U R                  U R                  UU["        R$                  5
        U R&                  (       a!  [(        R*                  " XpR,                  S94U4$ U4U4$ X!-  U R                  [
        R.                  " U5      -  -   U R                  S S & XR                  S S & U R&                  (       a)  [(        R*                  " U R                  U R,                  S9$ U R                  4$ )Nr   r   zNThe input passed for tabular masking does not match the background data shape!r   F©r   )Ú_standardize_maskÚlenr   r%   r	   r!   Ú
issubdtyper   ÚintegerÚ
invariantsr/   ÚintÚsumr0   r1   r.   Ú_delta_maskingr   Údelta_mask_noop_valuer   r   r   r   Úinvert)r3   ÚmaskÚxÚvariantsÚcurr_delta_indsÚ	num_masksÚvarying_rows_outÚmasked_inputs_outs           r4   Ú__call__ÚTabular.__call__d   sà  € Ø×%Ñ% dÓ.ˆô ˆq�w‰w‹<˜1Ó §¡¨¡
¨d¯i©i¯o©o¸aÑ.@Ó @Ü Ð!qÓrÐrô �=Š=˜Ÿ™¤R§Z¡Z×0Ò0ØŸ™¨Ó*Ð*ˆHÜ Ÿhšh¤s¨4£y¼Ñ<ˆOØ ™Ÿ™Ó)ˆIÜ!Ÿxšx¨·J±J¸q±MÐ(BÌ$ÑOÐÜ "§¢¨)·j±jÀ±mÑ*CÀTÇZÁZÐPQÁ]Ð)SÓ TÐØ!&ˆD�O‰O™AÐØ#'§9¡9ˆD×Ñ™aÐ ÜØØØØ Ø×!Ñ!Ø—‘Ø—	‘	ØØ!Ü×1Ñ1ôð ×$×$ÜŸšÐ%6×@RÑ@RÑSÐUÐWgÐgÐgà%Ð'Ð)9Ð9Ð9ð  !™x¨$¯)©)´b·i²iÀ³oÑ*EÑEˆ×Ñ™!ÐØ!�‰™Ðà× × Ü—<’< × 1Ñ 1¸4×;MÑ;MÑNÐNà×!Ñ!Ð#Ð#r7   c                 ó$  • UR                   U R                  R                   SS :w  aJ  [        S[        UR                   5      -   S-   [        U R                  R                   SS 5      -   S-   5      e[        R
                  " XR                  5      $ )zÒThis returns a mask of which features change when we mask them.

This optional masking method allows explainers to avoid re-evaluating the model when
the features that would have been masked are all invariant.
r   Nz^The passed data does not match the background shape expected by the masker! The data of shape z4 was passed while the masker expected data of shape Ú.)r   r%   r	   r*   r!   Úisclose)r3   rE   s     r4   r>   ÚTabular.invariants–   s…   € ð �7‰7�d—i‘i—o‘o a bÐ)Ó)Ü ØpÜ�a—g‘g“,ñàHñIô �d—i‘i—o‘o a bÐ)Ó*ñ+ð ñ	óð ô �zŠz˜!ŸY™YÓ'Ð'r7   c           	      óR  >• [         TU ]  U5        [        USSS9 nU R                  (       a:  UR                  S[        R
                  " U R                  U R                  S95        OS[        U SS5      b)  UR                  SU R                  U R                  45        OUR                  SU R                  5        UR                  SU R                  5        UR                  S	U R                  5        UR                  S
U R                  5        SSS5        g! , (       d  f       g= f)z(Write a Tabular masker to a file stream.úshap.maskers.Tabularr   )Úversionr%   r9   r   Nr(   r&   r'   )ÚsuperÚsaver   r   r   r   r%   r   Úgetattrr   r   r(   r&   r'   )r3   Úout_fileÚsÚ	__class__s      €r4   rU   ÚTabular.save¨   sÍ   ø€ ä‰‰�XÔô ˜Ð"8À!ÒDÈà×$×$Ø—‘�vœrŸ|š|¨D¯I©I¸t×?QÑ?QÑRÕSÜ˜˜v tÓ,Ñ8Ø—‘�v §	¡	¨4¯8©8Ð4Õ5à—‘�v˜tŸy™yÔ)à�F‰F�= $×"2Ñ"2Ô3Ø�F‰F�< §¡Ô1Ø�F‰F�; §¡Ô/÷ E×DÖDús   œC3DÄ
D&c                 óB  >• U(       a  U R                  U5      $ [        TU ]	  USS9n[        USSSS9 nUR                  S5      US'   UR                  S5      US'   UR                  S5      US'   UR                  S	5      US	'   S
S
S
5        U$ ! , (       d  f       U$ = f)z)Load a Tabular masker from a file stream.F)ÚinstantiaterR   r   )Úmin_versionÚmax_versionr%   r(   r&   r'   N)Ú_instantiated_loadrT   Úloadr   )ÚclsÚin_filer\   ÚkwargsrX   rY   s        €r4   r`   ÚTabular.loadº   s¦   ø€ ö Ø×)Ñ)¨'Ó2Ð2ä‘‘˜g°5�Ð9ˆÜ˜'Ð#9ÀqÐVWÒXÐ\]ØŸV™V F›^ˆF�6‰NØ$%§F¡F¨=Ó$9ˆF�=Ñ!Ø#$§6¡6¨,Ó#7ˆF�<Ñ Ø"#§&¡&¨Ó"5ˆF�;Ñ÷	 Yð
 ˆ÷ YÔXð
 ˆús   ´ABÂ
B)r1   r.   r&   r   r%   r   r(   r   r   r'   r   r2   )éd   NN)T)Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r5   rK   r>   rU   Úclassmethodr`   Ú__static_attributes__Ú__classcell__©rY   s   @r4   r   r      s1   ø† Ù<ôJ+òb($òd(õ$0ð$ öó ör7   r   c                 ój   • X:X  a  g X    (       a  US S 2U 4   US S 2U 4'   SX '   g X@   US S 2U 4'   SX '   g )NFT© )ÚdindÚmasked_inputsÚ	last_maskr%   rE   Ú	noop_codes         r4   Ú_single_delta_maskru   É   sD   € àÓØØ	�Ø!%¢a¨ g¡ˆ’a˜�gÑØˆ	Šà!"¡ˆ’a˜�gÑØˆ	Šr7   c
                 óê  • Sn
SnSnSnUR                   S   nU[        U 5      :  aÍ  US-  nSn
X   US'   X*   S:  a2  X*   * S-
  X*'   [        X*   XEXaU	5        U
S-  n
XU
-      X*'   X*   S:  a  M2  [        X*   XEXaU	5        XHXÝU-   & XÊS-   -  nUS:X  a	  SX;SS24'   OCU
S:X  a  USS2X*   4   X;SS24'   O+[        R                  " USS2USU
S-    4   SS9S:„  X;SS24'   XÞ-  nU[        U 5      :  a  MÌ  gg)z¼Implements the special (high speed) delta masking API that only flips the positions we need to.

Note that we attempt to avoid doing any allocation inside this function for speed reasons.
r   éÿÿÿÿr   TN)Úaxis)r   r;   ru   r!   r@   )ÚmasksrE   rG   rI   Úmasked_inputs_tmprs   r%   rF   rJ   rt   ÚdposÚiÚ	masks_posÚ
output_posÚNs                  r4   rA   rA   Õ   s_  € ð" €DØ
€AØ€IØ€JØ×Ñ Ñ"€AØ
”c˜%“jÓ
 Ø	ˆQ‰ˆð ˆØ"Ñ-ˆ˜ÑØÑ# aÓ'Ø%4Ñ%:Ð$:¸QÑ$>ˆOÑ!Ü˜Ñ4Ð6GÐTXÐ]fÔgØ�A‰IˆDØ$)°dÑ*:Ñ$;ˆOÑ!ð	 Ñ# aÕ'ô
 	˜?Ñ0Ð2CÐPTÐYbÔcð :K˜*°A¡~Ð6Ø˜A‘XÑˆ	ð �‹6Ø%)Ð¢˜TÒ"ð �q‹yØ)1²!°_Ñ5JÐ2JÑ)KÐ ¢A Ò&ô *,¯ª°º¸OÈJÈdÐUVÉhÐ<WÐ9WÑ0XÐ_`Ñ)aÐdeÑ)eÐ ¢A Ñ&à‰ˆ
ð= ”c˜%“j×
 r7   c                   ó0   ^ • \ rS rSrSrSU 4S jjrSrU =r$ )ÚIndependenti  zQThis masks out tabular features by integrating over the given background dataset.c                 ó"   >• [         TU ]  XSS9  g)a�  Build a Independent masker with the given background data.

Parameters
----------
data : numpy.ndarray, pandas.DataFrame
    The background dataset that is used for masking.

max_samples : int
    The maximum number of samples to use from the passed background data. If data has more
    than max_samples then shap.utils.sample is used to subsample the dataset. The number of
    samples coming out of the masker (to be integrated over) matches the number of samples in
    the background dataset. This means larger background dataset cause longer runtimes. Normally
    about 1, 10, 100, or 1000 background samples are reasonable choices.

N)r(   r&   ©rT   r5   )r3   r%   r(   rY   s      €r4   r5   ÚIndependent.__init__  s   ø€ ô  	‰Ñ˜À4ÐÒHr7   rp   )re   ©rf   rg   rh   ri   rj   r5   rl   rm   rn   s   @r4   r�   r�     s   ø† Ù[÷Iõ Ir7   r�   c                   ó0   ^ • \ rS rSrSrSU 4S jjrSrU =r$ )Ú	Partitioni"  z This masks out tabular features by integrating over the given background dataset.

Unlike Independent, Partition respects a hierarchical structure of the data.
c                 ó$   >• [         TU ]  XUSS9  g)a�  Build a Partition masker with the given background data and clustering.

Parameters
----------
data : numpy.ndarray, pandas.DataFrame
    The background dataset that is used for masking.

max_samples : int
    The maximum number of samples to use from the passed background data. If data has more
    than max_samples then shap.utils.sample is used to subsample the dataset. The number of
    samples coming out of the masker (to be integrated over) matches the number of samples in
    the background dataset. This means larger background dataset cause longer runtimes. Normally
    about 1, 10, 100, or 1000 background samples are reasonable choices.

clustering : string or numpy.ndarray
    If a string, then this is the distance metric to use for creating the clustering of
    the features. The distance function can be any valid scipy.spatial.distance.pdist's metric
    argument. However we suggest using 'correlation' in most cases. The full list of options is
    `braycurtis`, `canberra`, `chebyshev`, `cityblock`, `correlation`, `cosine`, `dice`,
    `euclidean`, `hamming`, `jaccard`, `jensenshannon`, `kulsinski`, `mahalanobis`,
    `matching`, `minkowski`, `rogerstanimoto`, `russellrao`, `seuclidean`,
    `sokalmichener`, `sokalsneath`, `sqeuclidean`, `yule`. These are all
    the options from scipy.spatial.distance.pdist's metric argument.
    If an array, then this is assumed to be the clustering of the features.

N)r(   r&   r'   rƒ   )r3   r%   r(   r&   rY   s       €r4   r5   ÚPartition.__init__(  s   ø€ ô6 	‰Ñ˜À:ÐY]ÐÒ^r7   rp   )re   Úcorrelationr…   rn   s   @r4   r‡   r‡   "  s   ø† ñ÷
_õ _r7   r‡   c                   ó"   • \ rS rSrSrSS jrSrg)ÚImputeiF  z½This imputes the values of missing features using the values of the observed features.

Unlike Independent, Gaussian imputes missing values based on correlations with observed data points.
c                 óÊ   • U[         L aN  SU;   aH  UR                  SS5      U l        UR                  SS5      U l        [        R
                  " US   S5      nXl        X l        g)z÷Build a Partition masker with the given background data and clustering.

Parameters
----------
data : numpy.ndarray, pandas.DataFrame or {"mean: numpy.ndarray, "cov": numpy.ndarray} dictionary
    The background dataset that is used for masking.

r   Nr   r   )r   r    r   r   r!   r"   r%   Úmethod)r3   r%   rŽ   s      r4   r5   ÚImpute.__init__L  sS   € ð ”4Š<˜F d›NØŸ™ ¨Ó.ˆDŒIØ—x‘x  tÓ,ˆDŒHÜ—>’> $ v¡,°Ó2ˆDàŒ	Ø�r7   )r   r%   r   rŽ   N)Úlinear)rf   rg   rh   ri   rj   r5   rl   rp   r7   r4   rŒ   rŒ   F  s   † ñ÷
r7   rŒ   )ÚloggingÚnumpyr!   Úpandasr   Únumbar   Ú r   Ú_serializabler   r   r   Úutils._exceptionsr	   r
   Ú_maskerr   Ú	getLoggerÚlogr   ru   rA   r�   r‡   rŒ   rp   r7   r4   Ú<module>r›      s‘   ðÛ ã Û Ý å ß 4Ý ß FÝ à×Ò˜Ó€ôvˆfô vðr ñó ðð ñ3ó ð3ôlI�'ô Iô,!_�ô !_ôHˆVõ r7   