ó
    †ñ:i6Z  ã                  óv  • % S SK Jr  S SKrS SKJrJrJrJr  S SKJ	r	  S SK
rS SKrS SKrS SKr\(       a  S SKJr  SrS\S'   SSS jjrSSS	 jjrSSS
 jjrSSS jjrSSS jjrSSS jjr\SSS jj5       r\SS S jj5       rS!S"S jjrS!S#S jjrS!S#S jjrS$S%S jjrS$S%S jjr SS&S jjr!S'S jr"SS(S jjr#g))é    )ÚannotationsN)ÚTYPE_CHECKINGÚFinalÚLiteralÚoverload)Úurlretrievez-https://github.com/shap/shap/raw/master/data/z
Final[str]Úgithub_data_urlc           	     ód  • [         S-   n[        R                  " [        U U  SU  S35      5      R	                  [        R
                  5      n[        R                  " [        U S35      5      nUb<  [        R                  R                  X1SS9n[        R                  R                  XASS9nX44$ )aG  Return a set of 50 images representative of ImageNet images.

Parameters
----------
resolution : int
    The resolution of the images. At present, the only supported value is 224.
n_points : int, optional
    Number of data points to sample. If None, the entire dataset is used.

Returns
-------
X : np.ndarray
    Represents images from ImageNet of a certain resolution.
y : np.ndarray
    The target variables, that is, the ImageNet classes.

Notes
-----
This dataset was collected by randomly finding a working ImageNet link and then pasting the
original ImageNet image into Google image search restricted to images licensed for reuse. A
similar image (now with rights to reuse) was downloaded as a rough replacement for the original
ImageNet image. The point is to have a random sample of ImageNet for use as a background
distribution for explaining models trained on ImageNet data.

Note that because the images are only rough replacements, the labels might no longer be correct.

Examples
--------
To get the processed images and labels::

    images, labels = shap.datasets.imagenet50()

Úimagenet50_Úxz.npyz
labels.csvr   ©Úrandom_state)
r	   ÚnpÚloadÚcacheÚastypeÚfloat32ÚloadtxtÚshapÚutilsÚsample)Ú
resolutionÚn_pointsÚprefixÚXÚys        ÚP/srv/projetos/modelo_ml_acdoc/venv/lib/python3.13/site-packages/shap/datasets.pyÚ
imagenet50r      sž   € ôD ˜}Ñ,€FÜ—G’GœE V H¨Z¨L¸¸*¸ÀTÐ"JÓKÓL×SÑSÔTV×T^ÑT^Ó_€AÜ—J’Jœu¨ x¨zÐ%:Ó;Ó<€AàÑÜ�J‰J×Ñ˜a¸ÐÐ:ˆÜ�J‰J×Ñ˜a¸ÐÐ:ˆàˆ4€Kó    c                ó,  • [         R                  R                  5       n[        R                  " UR
                  UR                  S9nUR                  nU b<  [        R                  R                  X SS9n[        R                  R                  X0SS9nX#4$ )a[  Return the California housing data in a tabular format.

Used in predictive regression tasks.

Parameters
----------
n_points : int, optional
    Number of data points to sample. If provided, randomly samples the specified number of points.

Returns
-------
X : pd.DataFrame
    The feature data.
y : np.ndarray
    The target variable.

Notes
-----
The returned feature matrix ``X`` includes the following features:

- ``MedInc`` (float): Median income in block
- ``HouseAge`` (float): Median house age in block
- ``AveRooms`` (float): Average rooms in dwelling
- ``AveBedrms`` (float): Average bedrooms in dwelling
- ``Population`` (float): Block population
- ``AveOccup`` (float): Average house occupancy
- ``Latitude`` (float): House block latitude
- ``Longitude`` (float): House block longitude

The target column represents the median house value for California districts.

References
----------
California housing dataset: :external+scikit-learn:func:`sklearn.datasets.fetch_california_housing`

Examples
--------
To get the processed data and target labels::

    data, target = shap.datasets.california()

©ÚdataÚcolumnsr   r   )ÚsklearnÚdatasetsÚfetch_california_housingÚpdÚ	DataFramer"   Úfeature_namesÚtargetr   r   r   ©r   ÚdÚdfr*   s       r   Ú
californiar.   @   sz   € ôV 	×Ñ×1Ñ1Ó3€AÜ	�Š˜1Ÿ6™6¨1¯?©?Ñ	;€BØŸ™€FàÑÜ�Z‰Z×Ñ˜r¸!ÐÐ<ˆÜ—‘×"Ñ" 6À!Ð"ÐDˆàˆ:Ðr   c                óf  • [         R                  R                  5       n[        R                  " UR
                  UR                  S9n[        R                  " UR                  UR                  S9nU b<  [        R                  R                  X SS9n[        R                  R                  X0SS9nX#4$ )a®  Return the Linnerud dataset in a convenient package for multi-target regression.

Parameters
----------
n_points : int, optional
    Number of data points to sample. If provided, randomly samples the specified number
    of points.

Returns
-------
X : pd.DataFrame
    The feature data.
y : pd.DataFrame
    The multiclass target variables.

Notes
-----
- The Linnerud dataset contains physiological and exercise data for 20 individuals.
- The feature matrix ``X`` includes three exercise variables: ``Chins``, ``Situps``, ``Jumps``.
- The target variables ``y`` include three physiological measurements: ``Weight``, ``Waist``, ``Pulse``.

More details: :external+scikit-learn:func:`sklearn.datasets.load_linnerud`

Examples
--------
To get the feature matrix and target variables::

    features, targets = shap.datasets.linnerud()

To get a subset of the data::

    subset_features, subset_targets = shap.datasets.linnerud(n_points=100)

)r#   r   r   )r$   r%   Úload_linnerudr'   r(   r"   r)   r*   Útarget_namesr   r   r   )r   r,   r   r   s       r   Úlinnerudr2   v   sˆ   € ôF 	×Ñ×&Ñ&Ó(€AÜ
�Š�Q—V‘V Q§_¡_Ñ5€AÜ
�Š�Q—X‘X q§~¡~Ñ6€AàÑÜ�J‰J×Ñ˜a¸ÐÐ:ˆÜ�J‰J×Ñ˜a¸ÐÐ:ˆàˆ4€Kr   c                óN  • [        [        [        S-   5      SS9 nUR                  5       nSSS5        [        R
                  " S[        S9nSUSS& U b=  [        R                  R                  WU SS	9n[        R                  R                  X0SS	9nWU4$ ! , (       d  f       Np= f)
a  Return the classic IMDB sentiment analysis training data in a nice package.

Used in binary text classification tasks.

Parameters
----------
n_points : int, optional
    Number of data points to sample. If provided, randomly samples the specified number of points.

Returns
-------
X : list of strings
    Text data, where each string is a movie review.
y : np.ndarray
    The target variable. Contains booleans, where True indicates a positive sentiment and False
    indicates a negative sentiment.

Notes
-----
Full data is at: http://ai.stanford.edu/~amaas/data/sentiment/aclImdb_v1.tar.gz

Paper to cite when using the data is: http://www.aclweb.org/anthology/P11-1015

Examples
--------
To get the processed text data and labels::

    text_data, labels = shap.datasets.imdb()

zimdb_train.txtzutf-8)ÚencodingNi¨a  ©Údtyper   iÔ0  r   )
Úopenr   r	   Ú	readlinesr   ÚonesÚboolr   r   r   )r   Úfr"   r   s       r   Úimdbr<   ¤   s”   € ô> 
Œe”OÐ&6Ñ6Ó7À'Ò	JÈaØ�{‰{‹}ˆ÷ 
Kä
�Š�œTÑ"€AØ€A€f€u€IàÑÜ�z‰z× Ñ   x¸aÐ Ð@ˆÜ�J‰J×Ñ˜a¸ÐÐ:ˆà�ˆ7€N÷ 
KÕ	Jús   ›BÂ
B$c           	     óf  • [         R                  " [        [        S-   5      SS9n[        R
                  " [        R                  " [        R                  " UR                  SS2S4   5      5      5      S   nU b  [        R                  R                  X SS9n[        R                  " UR                  US4   [        S9nUR                  US	S
24   n[        R
                  " [        R                  " UR                  5      R                  S5      S:H  5      S   nUR                  SS2U4   nXC4$ )aa  Predict the total number of violent crimes per 100K population.

This dataset is from the classic UCI Machine Learning repository:
https://archive.ics.uci.edu/ml/datasets/Communities+and+Crime+Unnormalized

Used in predictive regression tasks.

Parameters
----------
n_points : int, optional
    Number of data points to sample. If provided, randomly samples the specified number of points.

Returns
-------
X : pd.DataFrame
    The feature data.
y : np.ndarray
    The target variable.

Examples
--------
To get the processed data and target labels::

    data, target = shap.datasets.communitiesandcrime()

z CommViolPredUnnormalizedData.txtÚ?)Ú	na_valuesNéþÿÿÿr   r   r5   é   iîÿÿÿ)r'   Úread_csvr   r	   r   ÚwhereÚinvertÚisnanÚilocr   r   r   ÚarrayÚfloatÚvaluesÚsum)r   Úraw_dataÚ
valid_indsr   r   Ú
valid_colss         r   ÚcommunitiesandcrimerN   Ï   sò   € ô6 �{Š{œ5¤Ð3UÑ!UÓVÐbeÑf€Hô —’œ"Ÿ)š)¤B§H¢H¨X¯]©]º1¸b¸5Ñ-AÓ$BÓCÓDÀQÑG€JàÑÜ—Z‘Z×&Ñ& zÈ!Ð&ÐLˆ
ä
�Š�—‘˜z¨2˜~Ñ.´eÑ<€Að 	�‰�j ! C %Ð'Ñ(€AÜ—’œ"Ÿ(š( 1§8¡8Ó,×0Ñ0°Ó3°qÑ8Ó9¸!Ñ<€JØ	�‰Šq�*ˆ}Ñ€Aàˆ4€Kr   c                ó,  • [         R                  R                  5       n[        R                  " UR
                  UR                  S9nUR                  nU b<  [        R                  R                  X SS9n[        R                  R                  X0SS9nX#4$ )a€  Return the diabetes data in a nice package.

Used in predictive regression tasks.

Parameters
----------
n_points : int, optional
    Number of data points to sample. If provided, randomly samples the specified number of points.

Returns
-------
X : pd.DataFrame
    The feature data.
y : np.ndarray
    The target variable.

Notes
-----
Feature Columns in ``X``:

- ``age`` (float): Age in years
- ``sex`` (float): Sex
- ``bmi`` (float): Body mass index
- ``bp`` (float): Average blood pressure
- ``s1`` (float): Total serum cholesterol
- ``s2`` (float): Low-density lipoproteins (LDL cholesterol)
- ``s3`` (float): High-density lipoproteins (HDL cholesterol)
- ``s4`` (float): Total cholesterol / HDL cholesterol ratio
- ``s5`` (float): Log of serum triglycerides level
- ``s6`` (float): Blood sugar level

Target ``y``:

- Progression of diabetes one year after baseline (float)

The diabetes dataset is a subset of the larger diabetes dataset from scikit-learn.
More details: :external+scikit-learn:func:`sklearn.datasets.load_diabetes`

Examples
--------
To get the processed data and target labels::

    data, target = shap.datasets.diabetes()

r!   r   r   )r$   r%   Úload_diabetesr'   r(   r"   r)   r*   r   r   r   r+   s       r   ÚdiabetesrQ   ü   sz   € ô\ 	×Ñ×&Ñ&Ó(€AÜ	�Š˜1Ÿ6™6¨1¯?©?Ñ	;€BØ�X‰X€FàÑÜ�Z‰Z×Ñ˜r¸!ÐÐ<ˆÜ—‘×"Ñ" 6À!Ð"ÐDˆàˆ:Ðr   c                ó   • g ©N© ©Údisplayr   s     r   ÚirisrW   5  s   € Øhkr   c                ó   • g rS   rT   rU   s     r   rW   rW   7  s   € Øfir   c                ó˜  • [         R                  R                  5       n[        R                  " UR
                  UR                  S9nUR                  nUb<  [        R                  R                  X1SS9n[        R                  R                  XASS9nU (       a*  X4 Vs/ s H  n[        UR                  U   5      PM     sn4$ X44$ s  snf )as  Return the classic Iris dataset in a convenient package.

Parameters
----------
display : bool
    If True, return the original feature matrix along with class labels (as strings). Default is False.
n_points : int, optional
    Number of data points to sample. If provided, randomly samples the specified number of points.

Returns
-------
X : pd.DataFrame
    The feature matrix.
y : np.ndarray or a list of strings
    If ``display`` is False, a numpy array representing the class labels encoded as integers is returned.
    If ``display`` is True, then a list of class labels is returned.

Notes
-----
- The dataset includes measurements of sepal length, sepal width, petal length, and petal width for three
  species of iris flowers.
- Class labels are encoded as integers (0, 1, 2) representing the species (setosa, versicolor, virginica).
- If ``display`` is True, class labels are returned as strings.

Examples
--------
To get the feature matrix and class labels::

    features, labels = shap.datasets.iris()

To get the feature matrix and class labels as strings::

    features, class_labels = shap.datasets.iris(display=True)

r!   r   r   )r$   r%   Ú	load_irisr'   r(   r"   r)   r*   r   r   r   Ústrr1   )rV   r   r,   r-   r*   Úvs         r   rW   rW   ;  sª   € ôH 	×Ñ×"Ñ"Ó$€AÜ	�Š˜1Ÿ6™6¨1¯?©?Ñ	;€BØŸ™€FàÑÜ�Z‰Z×Ñ˜r¸!ÐÐ<ˆÜ—‘×"Ñ" 6À!Ð"ÐDˆæØ°FÓ;²F¨q”C˜Ÿ™ qÑ)Ö*±FÑ;Ð;Ð;Øˆ:Ðùò <s   Â"Cc           	     óÎ  • / SQn[         R                  " [        [        S-   5      U Vs/ s H  o3S   PM	     snS[	        U5      S9nUb  [
        R                  R                  XASS9nUR                  S/SS	9n[        [        S
 U5      5      nUS   S:H  US'   SSSSSSS.nU Hj  u  p‰U	S:X  d  M  US:X  a=  [        R                  " XX    V
s/ s H  o§U
R                  5          PM     sn
5      XX'   MP  XX   R                  R                  XX'   Ml     U (       a!  UR                  / SQSS	9US   R                   4$ UR                  SS/SS	9US   R                   4$ s  snf s  sn
f )af  Return the Adult census data in a structured format.

Used in binary classification tasks.

Parameters
----------
display : bool, optional
    If True, return the raw data without target and redundant columns.
n_points : int, optional
    Number of data points to sample. If provided, randomly samples the specified number of points.

Returns
-------
X : pd.DataFrame
    If ``display`` is True, ``X`` contains the raw data without the 'Education', 'Target', and 'fnlwgt' columns.
    Otherwise, ``X`` contains the processed data without the 'Target' and 'fnlwgt' columns.
y : np.ndarray
    The 'Target' column returned as an array.

Notes
-----
- The original data includes the following columns:

    - ``Age`` (float) : Age in years.
    - ``Workclass`` (category) : Type of employment.
    - ``fnlwgt`` (float) : Final weight; the number of units in the target population that the record represents.
    - ``Education`` (category) : Highest level of education achieved.
    - ``Education-Num`` (float) : Numeric representation of education level.
    - ``Marital Status`` (category) : Marital status of the individual.
    - ``Occupation`` (category) : Type of occupation.
    - ``Relationship`` (category) : Relationship status.
    - ``Race`` (category) : Ethnicity of the individual.
    - ``Sex`` (category) : Gender of the individual.
    - ``Capital Gain`` (float) : Capital gains recorded.
    - ``Capital Loss`` (float) : Capital losses recorded.
    - ``Hours per week`` (float) : Number of hours worked per week.
    - ``Country`` (category) : Country of origin.
    - ``Target`` (category) : Binary target variable indicating whether the individual earns more than 50K.

- The Education' column is redundant with 'Education-Num' and is dropped for simplicity.
- The 'Target' column is converted to binary (True/False) where '>50K' is True and '<=50K' is False.
- Certain categorical columns are encoded for numerical representation.

Examples
--------
To get the processed data and target labels::

    data, target = shap.datasets.adult()

To get the raw data for display::

    raw_data, target = shap.datasets.adult(display=True)

))ÚAger   )Ú	WorkclassÚcategory)Úfnlwgtr   )Ú	Educationr`   )zEducation-Numr   )zMarital Statusr`   )Ú
Occupationr`   )ÚRelationshipr`   )ÚRacer`   )ÚSexr`   )zCapital Gainr   )zCapital Lossr   )zHours per weekr   )ÚCountryr`   )ÚTargetr`   z
adult.datar   r>   )Únamesr?   r6   r   rb   é   )Úaxisc                ó   • U S   S;  $ )Nr   )rh   rb   rT   )r   s    r   Ú<lambda>Úadult.<locals>.<lambda>¼  s   € ¨¨!©Ð4KÒ(Kr   rh   z >50Ké   é   é   rA   )zNot-in-familyÚ	UnmarriedzOther-relativez	Own-childÚHusbandÚWifer`   rd   )rb   rh   ra   ra   )r'   rB   r   r	   Údictr   r   r   ÚdropÚlistÚfilterr   rG   ÚstripÚcatÚcodesrI   )rV   r   Údtypesr,   rK   r"   Úfilt_dtypesÚrcodeÚkr6   r\   s              r   Úadultr€   l  sh  € òn€Fô" �{Š{ÜŒo Ñ,Ó-ÁFÓ5KÂF¸q¸´dÁFÑ5KÐWZÔbfÐgmÓbnñ€Hð ÑÜ—:‘:×$Ñ$ XÀaÐ$ÐHˆà�=‰=˜+˜¨Qˆ=Ð/€DÜ”vÑKÈVÓTÓU€KØ˜(‘^ wÑ.€Dˆ�NØ¨aÀ1ÐSTÐabÐlmÑn€EÛ‰ˆØ�JÕØ�NÓ"ÜŸ(š(¸dºgÓ#Fºg¸¨!¯'©'«)Ô$4¹gÑ#FÓG�“à™'Ÿ+™+×+Ñ+�“ñ  ö Ø�}‰}Ò>ÀQˆ}ÐGÈÈhÉ×I^ÑI^Ð^Ð^Ø�9‰9�h Ð)°ˆ9Ð2°D¸±N×4IÑ4IÐIÐIùò' 6Lùò $Gs   ªE
ÃE"
c                ó¨  • [         R                  " [        [        S-   5      SS9n[         R                  " [        [        S-   5      SS9S   nUb<  [        R
                  R                  X!SS9n[        R
                  R                  X1SS9nU (       a(  UR                  5       nU[        R                  " U5      4$ U[        R                  " U5      4$ )a¿  Return a nicely packaged version of NHANES I data with survival times as labels.

Used in survival analysis tasks.

Parameters
----------
display : bool, optional
    If True, returns the features with a modified display. Default is False.
n_points : int, optional
    Number of data points to sample. Default is None (returns the entire dataset).

Returns
-------
X : pd.DataFrame
    The feature data matrix. If ``display`` is True, a modified version of the features for display
    is returned as ``X`` instead.
y : np.ndarray
    The target variables representing survival times.

Examples
--------
Usage example::

    features, survival_times = shap.datasets.nhanesi(display=True, n_points=100)

zNHANESI_X.csvr   )Ú	index_colzNHANESI_y.csvr   r   )
r'   rB   r   r	   r   r   r   Úcopyr   rG   )rV   r   r   r   Ú	X_displays        r   Únhanesir…   Ë  s¨   € ô6 	�Š”Eœ/¨OÑ;Ó<ÈÑJ€AÜ
�Š”Eœ/¨OÑ;Ó<ÈÑJÈ3ÑO€AàÑÜ�J‰J×Ñ˜a¸ÐÐ:ˆÜ�J‰J×Ñ˜a¸ÐÐ:ˆæØ—F‘F“Hˆ	àœ"Ÿ(š( 1›+Ð%Ð%ØŒb�hŠh�q‹kˆ>Ðr   c                ó^  ^• [         R                  R                  5       n[         R                  R                  S5        U Sp2[         R                  " U5      mSTSSS2'   [         R                  " U5      n[        SSS5       H?  nS=XEUS-   4'   XES-   U4'   S=XEUS-   4'   XES-   U4'   S=XES-   US-   4'   XES-   US-   4'   MA     U4S jn[         R                  R                  X#5      nXwR                  S5      -
  n[         R                  " UR                  U5      UR                  S   -  n	[         R                  R                  [         R                  R                  U	5      5      R                  n
[         R                  " XŠR                  5      n[         R                  R                  [         R                  " [         R                  " XŠR                  5      R                  5      [         R                  " U5      -
  5      S	:  d   e[         R                  " U[         R                  R                  U5      R                  5      nUnU" U5      [         R                  R                  U5      S
-  -   n[         R                  R                  U5        [         R"                  " U5      U4$ )aÄ  Correlated Groups (60 features)

A synthetic dataset consisting of 60 features with tight correlations among distinct groups of features.

Parameters
----------
n_points : int, optional
    Number of data points to generate. Default is 1,000.

Returns
-------
X : pd.DataFrame
    The feature data matrix
y : np.ndarray
    The target variables

Notes
-----
- The dataset is generated with known correlations among distinct groups of features.
- Each feature is a unit variance Gaussian random variable centred around 0.
- The labels are generated based on a linear function of the features with added random noise.

Examples
--------
.. code-block:: python

    data, target = shap.datasets.corrgroups60()

r   é<   rj   é   rp   g®Gáz®ï?ro   c                ó2   >• [         R                  " U T5      $ rS   ©r   Úmatmul©r   Úbetas    €r   r;   Úcorrgroups60.<locals>.f$  ó   ø€ Ü�yŠy˜˜DÓ!Ð!r   g�íµ ÷Æ°>ç{®Gáz„?)r   ÚrandomÚseedÚzerosÚeyeÚrangeÚrandnÚmeanr‹   ÚTÚshapeÚlinalgÚcholeskyÚinvÚnormÚcorrcoefr'   r(   )r   Úold_seedÚNÚMÚCÚir;   ÚX_startÚ
X_centeredÚSigmaÚWÚX_whiteÚX_finalr   r   r�   s                  @r   Úcorrgroups60rª   ô  s  ø€ ô> �y‰y�~‰~Ó€HÜ‡I�I‡N�N�1Ôð �R€qô �8Š8�A‹;€DØ€Dˆˆ2ˆaˆ�Lô 	�Šˆq‹	€AÜ�1�b˜!Ž_ˆØ$(Ð(ˆˆQ�‰Uˆ(‰�a˜A™˜q˜‘kØ$(Ð(ˆˆQ�‰Uˆ(‰�a˜A™˜q˜‘kØ,0Ð0ˆˆa‰%��Q‘ˆ,‰˜! ™E 1 q¡5˜L›/ñ õ
"ô �i‰i�o‰o˜aÓ#€GØŸ<™<¨›?Ñ*€JÜ�IŠI�j—l‘l JÓ/°*×2BÑ2BÀ1Ñ2EÑE€EÜ
�	‰	×Ñœ2Ÿ9™9Ÿ=™=¨Ó/Ó0×2Ñ2€AÜ�iŠi˜
§C¡CÓ(€Gä
�	‰	�‰”r—{’{¤2§9¢9¨Z¿¹Ó#=×#?Ñ#?Ó@Ä2Ç6Â6È!Ã9ÑLÓMÐPTÓTðØTô �iŠi˜¤§¡×!3Ñ!3°AÓ!6×!8Ñ!8Ó9€GØ€AÙ	ˆ!‹Œr�y‰y�‰˜qÓ! DÑ(Ñ(€Aô ‡I�I‡N�N�8Ôä�<Š<˜‹?˜AÐÐr   c                óô  ^• [         R                  R                  5       n[         R                  R                  S5        U Sp2[         R                  " U5      mSTSSS2'   U4S jn[         R                  R	                  X#5      nXUR                  S5      -
  nU" U5      [         R                  R	                  U5      S-  -   n[         R                  R                  U5        [        R                  " U5      U4$ )a@  Independent Linear (60 features)

A synthetic dataset consisting of 60 features.

Parameters
----------
n_points : int, optional
    Number of data points to generate. Default is 1,000.

Returns
-------
X : pd.DataFrame
    The feature data matrix
y : np.ndarray
    The target variables

Notes
-----
- Each feature is a unit variance Gaussian random variable centred around 0.
- The labels are generated based on a linear function of the features with added random noise.

Examples
--------
.. code-block:: python

    features, labels = shap.datasets.independentlinear60()

r   r‡   rj   rˆ   rp   c                ó2   >• [         R                  " U T5      $ rS   rŠ   rŒ   s    €r   r;   Úindependentlinear60.<locals>.fd  r�   r   r�   )r   r‘   r’   r“   r–   r—   r'   r(   )	r   rŸ   r    r¡   r;   r¤   r   r   r�   s	           @r   Úindependentlinear60r®   <  sº   ø€ ô< �y‰y�~‰~Ó€HÜ‡I�I‡N�N�1Ôð �R€qô �8Š8�A‹;€DØ€Dˆˆ2ˆaˆ�Lõ"ô �i‰i�o‰o˜aÓ#€GØ—,‘,˜q“/Ñ!€AÙ	ˆ!‹Œr�y‰y�‰˜qÓ! DÑ(Ñ(€Aô ‡I�I‡N�N�8Ôä�<Š<˜‹?˜AÐÐr   c                óè   • [         R                  R                  [        [        S-   5      5      u  pU b<  [
        R                  R                  XSS9n[
        R                  R                  X SS9nX4$ )aÍ  
Return a sparse dataset in scipy csr matrix format.

Data Source: :external+scikit-learn:func:`sklearn.datasets.load_svmlight_file`

Parameters
----------
n_points : int, optional
    Number of data points to sample. If None, returns the entire dataset. Default is None.

Returns
-------
X : scipy.sparse.csr_matrix
    Sparse feature matrix.
y : np.ndarray
    Target labels.

Examples
--------
.. code-block:: python

    data, target = shap.datasets.a1a()

za1a.svmlightr   r   )r$   r%   Úload_svmlight_filer   r	   r   r   r   )r   r"   r*   s      r   Úa1ar±   r  sf   € ô6 ×#Ñ#×6Ñ6´u¼_È~Ñ=]Ó7^Ó_�L€DàÑÜ�z‰z× Ñ  ¸aÐ Ð@ˆÜ—‘×"Ñ" 6À!Ð"ÐDˆàˆ<Ðr   c                 óL  • Sn [         R                  R                  [        U S-   5      5      u  p[         R                  R                  [        U S-   5      5      u  p4[        R
                  " [        U S-   5      5      n[        R
                  " [        U S-   5      5      nXX4XV4$ )aŽ  Return ranking datasets from the LightGBM repository.

Used in ranking tasks.

Returns
-------
x_train : scipy.sparse.csr_matrix
    Training feature matrix.
y_train : numpy.ndarray
    Training labels.
x_test : scipy.sparse.csr_matrix
    Testing feature matrix.
y_test : numpy.ndarray
    Testing labels.
q_train : numpy.ndarray
    Training query information.
q_test : numpy.ndarray
    Testing query information.

Notes
-----
Data Source: LightGBM repository https://github.com/microsoft/LightGBM/tree/master/examples/lambdarank

Examples
--------
.. code-block:: python

    x_train, y_train, x_test, y_test, q_train, q_test = shap.datasets.rank()

zPhttps://raw.githubusercontent.com/Microsoft/LightGBM/master/examples/lambdarank/z
rank.trainz	rank.testzrank.train.queryzrank.test.query)r$   r%   r°   r   r   r   )Úrank_data_urlÚx_trainÚy_trainÚx_testÚy_testÚq_trainÚq_tests          r   Úrankrº   –  s�   € ð> g€MÜ×'Ñ'×:Ñ:¼5ÀÐQ]ÑA]Ó;^Ó_Ñ€GÜ×%Ñ%×8Ñ8¼¸}È{Ñ?ZÓ9[Ó\�N€FÜ�jŠjœ˜}Ð/AÑAÓBÓC€GÜ�ZŠZœ˜mÐ.?Ñ?Ó@ÓA€Fà˜V¨WÐ<Ð<r   c                ó’  • Uc  [         R                  R                  U 5      n[         R                  R                  [         R                  R	                  [
        5      S5      n[         R                  " USS9  [         R                  R                  X!5      n[         R                  R                  U5      (       d  [        X5        U$ )z0Loads a file from the URL and caches it locally.Úcached_dataT)Úexist_ok)	ÚosÚpathÚbasenameÚjoinÚdirnameÚ__file__ÚmakedirsÚisfiler   )ÚurlÚ	file_nameÚdata_dirÚ	file_paths       r   r   r   ¾  s~   € àÑÜ—G‘G×$Ñ$ SÓ)ˆ	Ü�w‰w�|‰|œBŸG™GŸO™O¬HÓ5°}ÓE€HÜ‡K‚K� 4Ò(ä—W‘W—\‘\ (Ó6€IÜ�7‰7�>‰>˜)×$Ñ$Ü�CÔ#àÐr   )éà   N)r   Úintr   ú
int | NoneÚreturnztuple[np.ndarray, np.ndarray]rS   )r   rÌ   rÍ   útuple[pd.DataFrame, np.ndarray])r   rÌ   rÍ   z!tuple[pd.DataFrame, pd.DataFrame])r   rÌ   rÍ   ztuple[list[str], np.ndarray])..)rV   zLiteral[False]r   rÌ   rÍ   rÎ   )rV   zLiteral[True]r   rÌ   rÍ   ztuple[pd.DataFrame, list[str]])FN)rV   r:   r   rÌ   rÍ   z+tuple[pd.DataFrame, np.ndarray | list[str]])rV   r:   r   rÌ   rÍ   rÎ   )iè  )r   rË   rÍ   rÎ   )r   rÌ   rÍ   z!tuple[ssp.csr_matrix, np.ndarray])rÍ   zUtuple[ssp.csr_matrix, np.ndarray, ssp.csr_matrix, np.ndarray, np.ndarray, np.ndarray])rÆ   r[   rÇ   z
str | NonerÍ   r[   )$Ú
__future__r   r¾   Útypingr   r   r   r   Úurllib.requestr   Únumpyr   Úpandasr'   Úsklearn.datasetsr$   r   Úscipy.sparseÚsparseÚsspr	   Ú__annotations__r   r.   r2   r<   rN   rQ   rW   r€   r…   rª   r®   r±   rº   r   rT   r   r   Ú<module>rÙ      s¤   ðÞ "ã 	ß :Ó :Ý &ã Û Û ã æÝàM€�Ó Mö*öZ3öl+ö\(öV*öZ6ðr 
Ý kó 
Ø kØ	Ý ió 
Ø iö.öb\Jö~&öREöP3öl!ôH%=÷Pr   