ó
    ¦ñ:iB  ã                   ó¨   • S SK r S SKJr  S SKrSSKJrJrJr  SSK	J
r
  SSKJrJrJr  SSKJr  SSKJr  SS	KJrJrJrJr  S
SKJr   " S S\\5      rg)é    N)ÚIntegralé   )ÚBaseEstimatorÚTransformerMixinÚ_fit_context)Úresample)ÚIntervalÚOptionsÚ
StrOptions)Ú"_deprecate_Xt_in_inverse_transform)Ú_weighted_percentile)Ú_check_feature_names_inÚ_check_sample_weightÚcheck_arrayÚcheck_is_fittedé   )ÚOneHotEncoderc                   ó  • \ rS rSr% Sr\" \SSSS9S/\" 1 Sk5      /\" 1 S	k5      /\" \	\
R                  \
R                  15      S/\" \S
SSS9S/S/S.r\\S'    SSSSSSS.S jjr\" SS9SS j5       rS rS rSSS.S jjrSS jrSrg)ÚKBinsDiscretizeré   aN  
Bin continuous data into intervals.

Read more in the :ref:`User Guide <preprocessing_discretization>`.

.. versionadded:: 0.20

Parameters
----------
n_bins : int or array-like of shape (n_features,), default=5
    The number of bins to produce. Raises ValueError if ``n_bins < 2``.

encode : {'onehot', 'onehot-dense', 'ordinal'}, default='onehot'
    Method used to encode the transformed result.

    - 'onehot': Encode the transformed result with one-hot encoding
      and return a sparse matrix. Ignored features are always
      stacked to the right.
    - 'onehot-dense': Encode the transformed result with one-hot encoding
      and return a dense array. Ignored features are always
      stacked to the right.
    - 'ordinal': Return the bin identifier encoded as an integer value.

strategy : {'uniform', 'quantile', 'kmeans'}, default='quantile'
    Strategy used to define the widths of the bins.

    - 'uniform': All bins in each feature have identical widths.
    - 'quantile': All bins in each feature have the same number of points.
    - 'kmeans': Values in each bin have the same nearest center of a 1D
      k-means cluster.

    For an example of the different strategies see:
    :ref:`sphx_glr_auto_examples_preprocessing_plot_discretization_strategies.py`.

dtype : {np.float32, np.float64}, default=None
    The desired data-type for the output. If None, output dtype is
    consistent with input dtype. Only np.float32 and np.float64 are
    supported.

    .. versionadded:: 0.24

subsample : int or None, default=200_000
    Maximum number of samples, used to fit the model, for computational
    efficiency.
    `subsample=None` means that all the training samples are used when
    computing the quantiles that determine the binning thresholds.
    Since quantile computation relies on sorting each column of `X` and
    that sorting has an `n log(n)` time complexity,
    it is recommended to use subsampling on datasets with a
    very large number of samples.

    .. versionchanged:: 1.3
        The default value of `subsample` changed from `None` to `200_000` when
        `strategy="quantile"`.

    .. versionchanged:: 1.5
        The default value of `subsample` changed from `None` to `200_000` when
        `strategy="uniform"` or `strategy="kmeans"`.

random_state : int, RandomState instance or None, default=None
    Determines random number generation for subsampling.
    Pass an int for reproducible results across multiple function calls.
    See the `subsample` parameter for more details.
    See :term:`Glossary <random_state>`.

    .. versionadded:: 1.1

Attributes
----------
bin_edges_ : ndarray of ndarray of shape (n_features,)
    The edges of each bin. Contain arrays of varying shapes ``(n_bins_, )``
    Ignored features will have empty arrays.

n_bins_ : ndarray of shape (n_features,), dtype=np.int64
    Number of bins per feature. Bins whose width are too small
    (i.e., <= 1e-8) are removed with a warning.

n_features_in_ : int
    Number of features seen during :term:`fit`.

    .. versionadded:: 0.24

feature_names_in_ : ndarray of shape (`n_features_in_`,)
    Names of features seen during :term:`fit`. Defined only when `X`
    has feature names that are all strings.

    .. versionadded:: 1.0

See Also
--------
Binarizer : Class used to bin values as ``0`` or
    ``1`` based on a parameter ``threshold``.

Notes
-----

For a visualization of discretization on different datasets refer to
:ref:`sphx_glr_auto_examples_preprocessing_plot_discretization_classification.py`.
On the effect of discretization on linear models see:
:ref:`sphx_glr_auto_examples_preprocessing_plot_discretization.py`.

In bin edges for feature ``i``, the first and last values are used only for
``inverse_transform``. During transform, bin edges are extended to::

  np.concatenate([-np.inf, bin_edges_[i][1:-1], np.inf])

You can combine ``KBinsDiscretizer`` with
:class:`~sklearn.compose.ColumnTransformer` if you only want to preprocess
part of the features.

``KBinsDiscretizer`` might produce constant features (e.g., when
``encode = 'onehot'`` and certain bins do not contain any data).
These features can be removed with feature selection algorithms
(e.g., :class:`~sklearn.feature_selection.VarianceThreshold`).

Examples
--------
>>> from sklearn.preprocessing import KBinsDiscretizer
>>> X = [[-2, 1, -4,   -1],
...      [-1, 2, -3, -0.5],
...      [ 0, 3, -2,  0.5],
...      [ 1, 4, -1,    2]]
>>> est = KBinsDiscretizer(
...     n_bins=3, encode='ordinal', strategy='uniform'
... )
>>> est.fit(X)
KBinsDiscretizer(...)
>>> Xt = est.transform(X)
>>> Xt  # doctest: +SKIP
array([[ 0., 0., 0., 0.],
       [ 1., 1., 1., 0.],
       [ 2., 2., 2., 1.],
       [ 2., 2., 2., 2.]])

Sometimes it may be useful to convert the data back into the original
feature space. The ``inverse_transform`` function converts the binned
data into the original feature space. Each value will be equal to the mean
of the two bin edges.

>>> est.bin_edges_[0]
array([-2., -1.,  0.,  1.])
>>> est.inverse_transform(Xt)
array([[-1.5,  1.5, -3.5, -0.5],
       [-0.5,  2.5, -2.5, -0.5],
       [ 0.5,  3.5, -1.5,  0.5],
       [ 0.5,  3.5, -1.5,  1.5]])
r   NÚleft)Úclosedz
array-like>   ÚonehotÚordinalúonehot-dense>   ÚkmeansÚuniformÚquantiler   Úrandom_state©Ún_binsÚencodeÚstrategyÚdtypeÚ	subsampler   Ú_parameter_constraintsr   r   i@ )r"   r#   r$   r%   r   c                óL   • Xl         X l        X0l        X@l        XPl        X`l        g ©Nr    )Úselfr!   r"   r#   r$   r%   r   s          Úh/srv/projetos/modelo_ml_acdoc/venv/lib/python3.13/site-packages/sklearn/preprocessing/_discretization.pyÚ__init__ÚKBinsDiscretizer.__init__¸   s#   € ð ŒØŒØ ŒØŒ
Ø"ŒØ(Õó    T)Úprefer_skip_nested_validationc                 óî  • U R                  USS9nU R                  [        R                  [        R                  4;   a  U R                  nOUR                  nUR
                  u  pVUb*  U R                  S:X  a  [        SU R                  < S35      eU R                  b/  XPR                  :”  a   [        USU R                  U R                  S9nUR
                  S	   nU R                  U5      nUb  [        X1UR                  S9n[        R                  " U[        S9n[        U5       GH�  n	USS2U	4   n
U
R!                  5       U
R#                  5       pËX¼:X  aV  [$        R&                  " S
U	-  5        S	Xy'   [        R(                  " [        R*                  * [        R*                  /5      X‰'   M‡  U R                  S:X  a   [        R,                  " X¼Xy   S	-   5      X‰'   GOQU R                  S:X  aŽ  [        R,                  " SSXy   S	-   5      nUc-  [        R.                  " [        R0                  " X­5      5      X‰'   Oô[        R.                  " U Vs/ s H  n[3        X£U5      PM     sn[        R                  S9X‰'   O³U R                  S:X  a£  SSKJn  [        R,                  " X¼Xy   S	-   5      nUS	S USS -   SS2S4   S-  nU" Xy   US	S9nUR9                  U
SS2S4   US9R:                  SS2S4   nUR=                  5         US	S USS -   S-  X‰'   [        R>                  X¸U	   U4   X‰'   U R                  S;   d  GM  [        R@                  " X‰   [        R*                  S9S:„  nX‰   U   X‰'   [C        X‰   5      S	-
  Xy   :w  d  GMe  [$        R&                  " SU	-  5        [C        X‰   5      S	-
  Xy'   GM“     X€l"        Xpl#        SU RH                  ;   a�  [K        U RF                   Vs/ s H  n[        RL                  " U5      PM     snU RH                  S:H  US9U l'        U RN                  R9                  [        R                  " S	[C        U RF                  5      45      5        U $ s  snf s  snf )aë  
Fit the estimator.

Parameters
----------
X : array-like of shape (n_samples, n_features)
    Data to be discretized.

y : None
    Ignored. This parameter exists only for compatibility with
    :class:`~sklearn.pipeline.Pipeline`.

sample_weight : ndarray of shape (n_samples,)
    Contains weight values to be associated with each sample.
    Cannot be used when `strategy` is set to `"uniform"`.

    .. versionadded:: 1.3

Returns
-------
self : object
    Returns the instance itself.
Únumeric©r$   Nr   zY`sample_weight` was provided but it cannot be used with strategy='uniform'. Got strategy=z	 instead.F)ÚreplaceÚ	n_samplesr   r   z3Feature %d is constant and will be replaced with 0.r   r   éd   r   r   )ÚKMeanséÿÿÿÿç      à?)Ú
n_clustersÚinitÚn_init)Úsample_weight)r   r   )Úto_beging:Œ0âŽyE>zqBins whose width are too small (i.e., <= 1e-8) in feature %d are removed. Consider decreasing the number of bins.r   )Ú
categoriesÚsparse_outputr$   )(Ú_validate_datar$   ÚnpÚfloat64Úfloat32Úshaper#   Ú
ValueErrorr%   r   r   Ú_validate_n_binsr   ÚzerosÚobjectÚrangeÚminÚmaxÚwarningsÚwarnÚarrayÚinfÚlinspaceÚasarrayÚ
percentiler   Úclusterr5   ÚfitÚcluster_centers_ÚsortÚr_Úediff1dÚlenÚ
bin_edges_Ún_bins_r"   r   ÚarangeÚ_encoder)r)   ÚXÚyr;   Úoutput_dtyper3   Ú
n_featuresr!   Ú	bin_edgesÚjjÚcolumnÚcol_minÚcol_maxÚ	quantilesÚqr5   Úuniform_edgesr9   ÚkmÚcentersÚmaskÚis                         r*   rS   ÚKBinsDiscretizer.fitÉ   sõ  € ð2 ×Ñ ¨ÐÐ3ˆà�:‰:œ"Ÿ*™*¤b§j¡jÐ1Ó1ØŸ:™:‰LàŸ7™7ˆLà !§¡Ñˆ	àÑ$¨¯©¸)Ó)CÜð>à—=‘=Ñ# 9ð.óð ð �>‰>Ñ%¨)·n±nÓ*DäØØØŸ.™.Ø!×.Ñ.ñ	ˆAð —W‘W˜Q‘Zˆ
Ø×&Ñ& zÓ2ˆàÑ$Ü0°ÈÏÉÑQˆMä—H’H˜Z¬vÑ6ˆ	Ü˜
×#ˆBØ’q˜"�u‘XˆFØ%Ÿz™z›|¨V¯Z©Z«\�WàÓ!Ü—’ØIÈBÑNôð �‘
Ü "§¢¬2¯6©6¨'´2·6±6Ð):Ó ;�	‘Ùà�}‰} 	Ó)Ü "§¢¨G¸f¹jÈ1¹nÓ M�	“à—‘ *Ó,ÜŸKšK¨¨3°±
¸Q±Ó?�	Ø Ñ(Ü$&§J¢J¬r¯}ª}¸VÓ/OÓ$P�I’Mä$&§J¢Jñ &/óâ%. ô 1°ÈÖJÙ%.ñô !Ÿj™jñ%�I’Mð —‘ (Ó*Ý,ô !#§¢¨G¸f¹jÈ1¹nÓ M�Ø% a bÐ)¨M¸#¸2Ð,>Ñ>ÂÀ4ÀÑHÈ3ÑN�ñ  v¡z¸ÀQÑG�ØŸ&™&Øš1˜d˜7‘O°=ð !ð ç"Ñ"¢1 a 4ñ)�ð —‘”Ø!(¨¨ ¨w°s¸¨|Ñ!;¸sÑ B�	‘Ü "§¡ g¸©}¸gÐ&EÑ F�	‘ð �}‰}Ð 6Ö6Ü—z’z )¡-¼"¿&¹&ÑAÀDÑH�Ø )¡¨dÑ 3�	‘Ü�y‘}Ó%¨Ñ)¨V©ZÖ7Ü—M’Mð9à;=ñ>ôô
 "% Y¡]Ó!3°aÑ!7�F”Jñm $ðp $ŒØŒà�t—{‘{Ó"Ü)Ø26·,²,Ó?²,¨QœBŸIšI ažL±,Ñ?Ø"Ÿk™k¨XÑ5Ø"ñˆDŒMð �M‰M×ÑœbŸhšh¨¬3¨t¯|©|Ó+<Ð'=Ó>Ô?àˆùòaùòP @s   ÉQ-
Ï. Q2c                 óä  • U R                   n[        U[        5      (       a  [        R                  " X[
        S9$ [        U[
        SSS9nUR                  S:”  d  UR                  S   U:w  a  [        S5      eUS:  X2:g  -  n[        R                  " U5      S   nUR                  S   S:”  aA  S	R                  S
 U 5       5      n[        SR                  [        R                  U5      5      eU$ )z0Returns n_bins_, the number of bins per feature.r1   TF)r$   ÚcopyÚ	ensure_2dr   r   z8n_bins must be a scalar or array of shape (n_features,).r   z, c              3   ó8   #   • U  H  n[        U5      v •  M     g 7fr(   )Ústr)Ú.0rl   s     r*   Ú	<genexpr>Ú4KBinsDiscretizer._validate_n_bins.<locals>.<genexpr>X  s   é € ÐBÒ0A¨1¤ A§ Ò0Aùs   ‚zk{} received an invalid number of bins at indices {}. Number of bins must be at least 2, and must be an int.)r!   Ú
isinstancer   r@   ÚfullÚintr   ÚndimrC   rD   ÚwhereÚjoinÚformatr   Ú__name__)r)   r`   Ú	orig_binsr!   Úbad_nbins_valueÚviolating_indicesÚindicess          r*   rE   Ú!KBinsDiscretizer._validate_n_binsI  sÚ   € à—K‘Kˆ	Ü�i¤×*Ñ*Ü—7’7˜:¼Ñ<Ð<ä˜Y¬c¸ÈÑNˆà�;‰;˜‹?˜fŸl™l¨1™o°Ó;ÜÐWÓXÐXà! A™:¨&Ñ*=Ñ>ˆäŸHšH _Ó5°aÑ8ÐØ×"Ñ" 1Ñ%¨Ó)Ø—i‘iÑBÑ0AÓBÓBˆGÜð:ç:@¹&Ü$×-Ñ-¨wó;óð ð ˆr-   c                 ó†  • [        U 5        U R                  c   [        R                  [        R                  4OU R                  nU R                  USUSS9nU R                  n[        UR                  S   5       H,  n[        R                  " XE   SS USS2U4   SS9USS2U4'   M.     U R                  S	:X  a  U$ SnS
U R                  ;   a1  U R                  R                  nUR                  U R                  l         U R                  R                  U5      nX`R                  l        U$ ! X`R                  l        f = f)a3  
Discretize the data.

Parameters
----------
X : array-like of shape (n_samples, n_features)
    Data to be discretized.

Returns
-------
Xt : {ndarray, sparse matrix}, dtype={np.float32, np.float64}
    Data in the binned space. Will be a sparse matrix if
    `self.encode='onehot'` and ndarray otherwise.
NTF)ro   r$   Úresetr   r6   Úright)Úsider   r   )r   r$   r@   rA   rB   r?   rY   rH   rC   Úsearchsortedr"   r\   Ú	transform)r)   r]   r$   ÚXtra   rb   Ú
dtype_initÚXt_encs           r*   rˆ   ÚKBinsDiscretizer.transformb  s  € ô 	˜Ôð -1¯J©JÑ,>”—‘œRŸZ™ZÑ(ÀDÇJÁJˆØ× Ñ  ¨°UÀ%Ð ÐHˆà—O‘Oˆ	Ü˜Ÿ™ ™Ö$ˆBÜŸš¨	©°a¸Ð(;¸RÂÀ2À¹YÈWÑUˆBŠq�"ˆu‹Iñ %ð �;‰;˜)Ó#ØˆIàˆ
Ø�t—{‘{Ó"ØŸ™×,Ñ,ˆJØ"$§(¡(ˆD�M‰MÔð	-Ø—]‘]×,Ñ,¨RÓ0ˆFð #-�M‰MÔØˆøð #-�M‰MÕús   ÄD. Ä.E )r‰   c                ó<  • [        X5      n[        U 5        SU R                  ;   a  U R                  R	                  U5      n[        US[        R                  [        R                  4S9nU R                  R                  S   nUR                  S   U:w  a'  [        SR                  XCR                  S   5      5      e[        U5       HO  nU R                  U   nUSS USS -   S	-  nXsSS2U4   R                  [        R                   5         USS2U4'   MQ     U$ )
a9  
Transform discretized data back to original feature space.

Note that this function does not regenerate the original data
due to discretization rounding.

Parameters
----------
X : array-like of shape (n_samples, n_features)
    Transformed data in the binned space.

Xt : array-like of shape (n_samples, n_features)
    Transformed data in the binned space.

    .. deprecated:: 1.5
        `Xt` was deprecated in 1.5 and will be removed in 1.7. Use `X` instead.

Returns
-------
Xinv : ndarray, dtype={np.float32, np.float64}
    Data in the original feature space.
r   T)ro   r$   r   r   z8Incorrect number of features. Expecting {}, received {}.Nr6   r7   )r   r   r"   r\   Úinverse_transformr   r@   rA   rB   rZ   rC   rD   r|   rH   rY   ÚastypeÚint64)r)   r]   r‰   ÚXinvr`   rb   ra   Úbin_centerss           r*   rŽ   Ú"KBinsDiscretizer.inverse_transform‰  s  € ô. /¨qÓ5ˆä˜Ôà�t—{‘{Ó"Ø—‘×/Ñ/°Ó2ˆAä˜1 4´·
±
¼B¿J¹JÐ/GÑHˆØ—\‘\×'Ñ'¨Ñ*ˆ
Ø�:‰:�a‰=˜JÓ&ÜØJ×QÑQØ§
¡
¨1¡óóð ô ˜
Ö#ˆBØŸ™¨Ñ+ˆIØ$ Q R˜=¨9°S°b¨>Ñ9¸SÑ@ˆKØ%ªA¨r¨E¡{×&:Ñ&:¼2¿8¹8Ó&DÑEˆD’�B�‹Kñ $ð
 ˆr-   c                 óŒ   • [        U S5        [        X5      n[        U S5      (       a  U R                  R	                  U5      $ U$ )a\  Get output feature names.

Parameters
----------
input_features : array-like of str or None, default=None
    Input features.

    - If `input_features` is `None`, then `feature_names_in_` is
      used as feature names in. If `feature_names_in_` is not defined,
      then the following input feature names are generated:
      `["x0", "x1", ..., "x(n_features_in_ - 1)"]`.
    - If `input_features` is an array-like, then `input_features` must
      match `feature_names_in_` if `feature_names_in_` is defined.

Returns
-------
feature_names_out : ndarray of str objects
    Transformed feature names.
Ún_features_in_r\   )r   r   Úhasattrr\   Úget_feature_names_out)r)   Úinput_featuress     r*   r—   Ú&KBinsDiscretizer.get_feature_names_out·  sC   € ô( 	˜Ð.Ô/Ü0°ÓFˆÜ�4˜×$Ñ$Ø—=‘=×6Ñ6°~ÓFÐFð Ðr-   )	r\   rY   r$   r"   r!   rZ   r   r#   r%   )é   )NNr(   )r}   Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r	   r   r   r
   Útyper@   rA   rB   r&   ÚdictÚ__annotations__r+   r   rS   rE   rˆ   rŽ   r—   Ú__static_attributes__© r-   r*   r   r      sÓ   ‡ ñRñj ˜H a¨°fÑ=¸|ÐLÙÒCÓDÐEÙÒ AÓBÐCÙ˜$ §¡¨R¯Z©ZÐ 8Ó9¸4Ð@Ù˜x¨¨D¸Ñ@À$ÐGØ'Ð(ñ$Ð˜Dó ð ð)ð ØØØØö)ñ" °Ñ5ó}ó 6ð}ò~ò2%ðN,¨dö ,÷\r-   r   )rK   Únumbersr   Únumpyr@   Úbaser   r   r   Úutilsr   Úutils._param_validationr	   r
   r   Úutils.deprecationr   Úutils.statsr   Úutils.validationr   r   r   r   Ú	_encodersr   r   r£   r-   r*   Ú<module>r­      sE   ðó Ý ã ç @Ñ @Ý ß CÑ CÝ BÝ .÷ó õ %ôwÐ'¨õ wr-   