ó
    ¨ñ:iÔ!  ã                   ó„   • S r SSKrSSKrSSKJr  SSKJr  SSKJ	r	  SSK
Jr  SSKJr  SS	KJr  SS
KJr   " S S\\5      rg)z(Metrics to perform pairwise computation.é    N)Údistance_matrix)ÚBaseEstimator)Úcheck_consistent_length)Úunique_labels)Úcheck_is_fittedé   )Ú_ParamsValidationMixin)Ú
StrOptionsc                   ó–   • \ rS rSr% Sr\" S15      S/\R                  /\R                  /S.r\	\
S'   SSSS.S	 jrS
 rSS jrS rSrg)ÚValueDifferenceMetricé   aC  Class implementing the Value Difference Metric.

This metric computes the distance between samples containing only
categorical features. The distance between feature values of two samples is
defined as:

.. math::
   \delta(x, y) = \sum_{c=1}^{C} |p(c|x_{f}) - p(c|y_{f})|^{k} \ ,

where :math:`x` and :math:`y` are two samples and :math:`f` a given
feature, :math:`C` is the number of classes, :math:`p(c|x_{f})` is the
conditional probability that the output class is :math:`c` given that
the feature value :math:`f` has the value :math:`x` and :math:`k` an
exponent usually defined to 1 or 2.

The distance for the feature vectors :math:`X` and :math:`Y` is
subsequently defined as:

.. math::
   \Delta(X, Y) = \sum_{f=1}^{F} \delta(X_{f}, Y_{f})^{r} \ ,

where :math:`F` is the number of feature and :math:`r` an exponent usually
defined equal to 1 or 2.

The definition of this distance was propoed in [1]_.

Read more in the :ref:`User Guide <vdm>`.

.. versionadded:: 0.8

Parameters
----------
n_categories : "auto" or array-like of shape (n_features,), default="auto"
    The number of unique categories per features. If `"auto"`, the number
    of categories will be computed from `X` at `fit`. Otherwise, you can
    provide an array-like of such counts to avoid computation. You can use
    the fitted attribute `categories_` of the
    :class:`~sklearn.preprocesssing.OrdinalEncoder` to deduce these counts.

k : int, default=1
    Exponent used to compute the distance between feature value.

r : int, default=2
    Exponent used to compute the distance between the feature vector.

Attributes
----------
n_categories_ : ndarray of shape (n_features,)
    The number of categories per features.

proba_per_class_ : list of ndarray of shape (n_categories, n_classes)
    List of length `n_features` containing the conditional probabilities
    for each category given a class.

n_features_in_ : int
    Number of features in the input dataset.

    .. versionadded:: 0.10

feature_names_in_ : ndarray of shape (`n_features_in_`,)
    Names of features seen during `fit`. Defined only when `X` has feature
    names that are all strings.

    .. versionadded:: 0.10

See Also
--------
sklearn.neighbors.DistanceMetric : Interface for fast metric computation.

Notes
-----
The input data `X` are expected to be encoded by an
:class:`~sklearn.preprocessing.OrdinalEncoder` and the data type is used
should be `np.int32`. If other data types are given, `X` will be converted
to `np.int32`.

References
----------
.. [1] Stanfill, Craig, and David Waltz. "Toward memory-based reasoning."
   Communications of the ACM 29.12 (1986): 1213-1228.

Examples
--------
>>> import numpy as np
>>> X = np.array(["green"] * 10 + ["red"] * 10 + ["blue"] * 10).reshape(-1, 1)
>>> y = [1] * 8 + [0] * 5 + [1] * 7 + [0] * 9 + [1]
>>> from sklearn.preprocessing import OrdinalEncoder
>>> encoder = OrdinalEncoder(dtype=np.int32)
>>> X_encoded = encoder.fit_transform(X)
>>> from imblearn.metrics.pairwise import ValueDifferenceMetric
>>> vdm = ValueDifferenceMetric().fit(X_encoded, y)
>>> pairwise_distance = vdm.pairwise(X_encoded)
>>> pairwise_distance.shape
(30, 30)
>>> X_test = np.array(["green", "red", "blue"]).reshape(-1, 1)
>>> X_test_encoded = encoder.transform(X_test)
>>> vdm.pairwise(X_test_encoded)
array([[0.  ,  0.04,  1.96],
       [0.04,  0.  ,  1.44],
       [1.96,  1.44,  0.  ]])
Úautoz
array-like©Ún_categoriesÚkÚrÚ_parameter_constraintsé   r   c                ó(   • Xl         X l        X0l        g ©Nr   )Úselfr   r   r   s       Ú\/srv/projetos/modelo_ml_acdoc/venv/lib/python3.13/site-packages/imblearn/metrics/pairwise.pyÚ__init__ÚValueDifferenceMetric.__init__   s   € Ø(ÔØŒØ�ó    c           	      óì  • U R                  5         [        X5        U R                  XS[        R                  S9u  p[        U R                  [        5      (       a(  U R                  S:X  a  UR                  SS9S-   U l	        Ow[        U R                  5      U R                  :w  a/  [        S[        U R                  5       SU R                   S	35      e[        R                  " U R                  5      U l	        [        U5      nU R                   Vs/ s H1  n[        R                  " U[        U5      4[        R                   S
9PM3     snU l        [%        U R                  5       HT  n['        U5       HB  u  pg[        R(                  " XU:H  U4   U R                  U   S9U R"                  U   SS2U4'   MD     MV     [        R*                  " SS9   [%        U R                  5       Hf  nU R"                  U==   U R"                  U   R-                  SS9R/                  SS5      -  ss'   [        R0                  " U R"                  U   SS9  Mh     SSS5        U $ s  snf ! , (       d  f       U $ = f)ar  Compute the necessary statistics from the training set.

Parameters
----------
X : ndarray of shape (n_samples, n_features), dtype=np.int32
    The input data. The data are expected to be encoded with a
    :class:`~sklearn.preprocessing.OrdinalEncoder`.

y : ndarray of shape (n_features,)
    The target.

Returns
-------
self : object
    Return the instance itself.
T©ÚresetÚdtyper   r   )Úaxisr   zRThe length of n_categories is not consistent with the number of feature in X. Got z elements in n_categories and z in X.©Úshaper   )Ú	minlengthNÚignore)ÚinvalidéÿÿÿÿF)Úcopy)Ú_validate_paramsr   Ú_validate_dataÚnpÚint32Ú
isinstancer   ÚstrÚmaxÚn_categories_ÚlenÚn_features_in_Ú
ValueErrorÚasarrayr   ÚemptyÚfloat64Úproba_per_class_ÚrangeÚ	enumerateÚbincountÚerrstateÚsumÚreshapeÚ
nan_to_num)r   ÚXÚyÚclassesÚn_catÚfeature_idxÚ	klass_idxÚklasss           r   ÚfitÚValueDifferenceMetric.fit„   s/  € ð" 	×ÑÔÜ Ô%Ø×"Ñ" 1¨t¼2¿8¹8Ð"ÐD‰ˆä�d×'Ñ'¬×-Ñ-°$×2CÑ2CÀvÓ2Mà!"§¡¨A  °Ñ!2ˆDÕä�4×$Ñ$Ó%¨×)<Ñ)<Ó<Ü ð3Ü36°t×7HÑ7HÓ3IÐ2Jð K4Ø48×4GÑ4GÐ3Hð Iðóð ô "$§¢¨D×,=Ñ,=Ó!>ˆDÔÜ Ó"ˆð ×+Ò+ó!
â+�ô �HŠH˜E¤3 w£<Ð0¼¿
¹
ÔCÙ+ñ!
ˆÔô ! ×!4Ñ!4Ö5ˆKÜ$-¨gÖ$6Ñ �	ÜCEÇ;Â;Ø˜5‘j +Ð-Ñ.Ø"×0Ñ0°Ñ=ñD�×%Ñ% kÑ2²1°i°<Ó@ó %7ñ 6ô �[Š[ Ó*ä$ T×%8Ñ%8Ö9�Ø×%Ñ% kÓ2Ø×)Ñ)¨+Ñ6×:Ñ:ÀÐ:ÐB×JÑJÈ2ÈqÓQñÓ2ô —’˜d×3Ñ3°KÑ@ÀuÔMñ	  :÷ +ð ˆùò)!
÷ +Ô*ð ˆús   Ä8IÇB I$É$
I3Nc                 ó  • [        U 5        U R                  US[        R                  S9nUR                  S   nUb/  U R                  US[        R                  S9nUR                  S   nOUn[        R
                  " X44[        R                  S9n[        U R                  5       H_  nU R                  U   USS2U4      nUb  U R                  U   USS2U4      nOUnU[        XxU R                  S9U R                  -  -  nMa     U$ )a  Compute the VDM distance pairwise.

Parameters
----------
X : ndarray of shape (n_samples, n_features), dtype=np.int32
    The input data. The data are expected to be encoded with a
    :class:`~sklearn.preprocessing.OrdinalEncoder`.

Y : ndarray of shape (n_samples, n_features), dtype=np.int32
    The input data. The data are expected to be encoded with a
    :class:`~sklearn.preprocessing.OrdinalEncoder`.

Returns
-------
distance_matrix : ndarray of shape (n_samples, n_samples)
    The VDM pairwise distance.
Fr   r   Nr!   )Úp)r   r)   r*   r+   r"   Úzerosr5   r7   r1   r6   r   r   r   )	r   r>   ÚYÚn_samples_XÚn_samples_YÚdistancerB   Úproba_feature_XÚproba_feature_Ys	            r   ÚpairwiseÚValueDifferenceMetric.pairwise¿   s   € ô$ 	˜ÔØ×Ñ ¨´b·h±hÐÐ?ˆØ—g‘g˜a‘jˆà‰=Ø×#Ñ# A¨U¼"¿(¹(Ð#ÐCˆAØŸ'™' !™*‰Kà%ˆKä—8’8 ;Ð"<ÄBÇJÁJÑOˆÜ  ×!4Ñ!4Ö5ˆKØ"×3Ñ3°KÑ@ÀÂ1ÀkÀ>ÑARÑSˆOØ‰}Ø"&×"7Ñ"7¸Ñ"DÀQÂqÈ+À~ÑEVÑ"W‘à"1�ØÜ ÀDÇFÁFÑKÈtÏvÉvÑUñŠHñ 6ð ˆr   c                 ó
   • SS0$ )NÚrequires_positive_XT© )r   s    r   Ú
_more_tagsÚ ValueDifferenceMetric._more_tagsç   s   € à! 4ð
ð 	
r   )r   r   r/   r6   r   r   )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r
   ÚnumbersÚIntegralr   ÚdictÚ__annotations__r   rE   rP   rU   Ú__static_attributes__rT   r   r   r   r      s`   ‡ ñdñL $ V HÓ-¨|Ð<Ø×ÑÐØ×ÑÐñ$Ð˜Dó ð (.°°aõ ò
9ôv&õP
r   r   )r[   r\   Únumpyr*   Úscipy.spatialr   Úsklearn.baser   Úsklearn.utilsr   Úsklearn.utils.multiclassr   Úsklearn.utils.validationr   Úbaser	   Úutils._param_validationr
   r   rT   r   r   Ú<module>ri      s6   ðÙ .ó
 ã Ý )Ý &Ý 1Ý 2Ý 4å )Ý 0ôW
Ð2°Mõ W
r   