ó
    ¨ñ:iMA  ã                   óÌ  • S r SSKrSSKrSSKrSSKrSSKrSSKJr  SSK	J
r
  SSKJr  SSKJr  SSKJr  SSKJr  SS	KJr  SS
KJr   SSKJrJr  SSKJr  SSKJ r   SSK!J"r"  SSK#J$r$  SSK%J&r&J'r'J(r(  SSK)J*r*  SSK+J,r,J-r-  SSK.J/r/J0r0J1r1  SSK2J3r3  SSK4J5r5J6r6  \" \Rn                  5      r8\&" \$Rr                  \,\-S9 " S S\\
5      5       r:g! \\4 a    SSKJr  SSKJr   N’f = f)z9Bagging classifier trained on balanced bootstrap samples.é    N)Úclone)ÚBaggingClassifier)Ú_parallel_decision_function)Ú_partition_estimators)ÚNotFittedError)ÚDecisionTreeClassifier)Úparse_version)Úcheck_is_fitted)ÚParallelÚdelayed)r   )r   é   )Ú_ParamsValidationMixin)ÚPipeline)ÚRandomUnderSampler)ÚBaseUnderSampler)ÚSubstitutionÚcheck_sampling_strategyÚcheck_target_type)Úavailable_if)Ú_n_jobs_docstringÚ_random_state_docstring)Ú
HasMethodsÚIntervalÚ
StrOptions)Ú_fit_contexté   )Ú_bagging_parameter_constraintsÚ_estimator_has)Úsampling_strategyÚn_jobsÚrandom_statec                   óö  ^ • \ rS rSrSr\\" S5      :¼  a  \R                  " \	R                  5      r
O\R                  " \5      r
\
R                  \" \R                  SSSS9\" 1 Sk5      \\/S	/\" S
/5      S/S.5        S\
;   a  \
S	   SSSSSSSSSSSSSS.U 4S jjjrU 4S jr\" 5       4S jr\S 5       r\" SS9U 4S j5       rS U 4S jjr\" \" S5      5      S 5       r\S 5       r U 4S jr!Sr"U =r#$ )!ÚBalancedBaggingClassifieré+   uÝ  A Bagging classifier with additional balancing.

This implementation of Bagging is similar to the scikit-learn
implementation. It includes an additional step to balance the training set
at fit time using a given sampler.

This classifier can serves as a basis to implement various methods such as
Exactly Balanced Bagging [6]_, Roughly Balanced Bagging [7]_,
Over-Bagging [6]_, or SMOTE-Bagging [8]_.

Read more in the :ref:`User Guide <bagging>`.

Parameters
----------
estimator : estimator object, default=None
    The base estimator to fit on random subsets of the dataset.
    If None, then the base estimator is a decision tree.

    .. versionadded:: 0.10

n_estimators : int, default=10
    The number of base estimators in the ensemble.

max_samples : int or float, default=1.0
    The number of samples to draw from X to train each base estimator.

    - If int, then draw ``max_samples`` samples.
    - If float, then draw ``max_samples * X.shape[0]`` samples.

max_features : int or float, default=1.0
    The number of features to draw from X to train each base estimator.

    - If int, then draw ``max_features`` features.
    - If float, then draw ``max_features * X.shape[1]`` features.

bootstrap : bool, default=True
    Whether samples are drawn with replacement.

    .. note::
       Note that this bootstrap will be generated from the resampled
       dataset.

bootstrap_features : bool, default=False
    Whether features are drawn with replacement.

oob_score : bool, default=False
    Whether to use out-of-bag samples to estimate
    the generalization error.

warm_start : bool, default=False
    When set to True, reuse the solution of the previous call to fit
    and add more estimators to the ensemble, otherwise, just fit
    a whole new ensemble.

{sampling_strategy}

replacement : bool, default=False
    Whether or not to randomly sample with replacement or not when
    `sampler is None`, corresponding to a
    :class:`~imblearn.under_sampling.RandomUnderSampler`.

{n_jobs}

{random_state}

verbose : int, default=0
    Controls the verbosity of the building process.

sampler : sampler object, default=None
    The sampler used to balanced the dataset before to bootstrap
    (if `bootstrap=True`) and `fit` a base estimator. By default, a
    :class:`~imblearn.under_sampling.RandomUnderSampler` is used.

    .. versionadded:: 0.8

Attributes
----------
estimator_ : estimator
    The base estimator from which the ensemble is grown.

    .. versionadded:: 0.10

n_features_ : int
    The number of features when `fit` is performed.

    .. deprecated:: 1.0
       `n_features_` is deprecated in `scikit-learn` 1.0 and will be removed
       in version 1.2. When the minimum version of `scikit-learn` supported
       by `imbalanced-learn` will reach 1.2, this attribute will be removed.

estimators_ : list of estimators
    The collection of fitted base estimators.

sampler_ : sampler object
    The validate sampler created from the `sampler` parameter.

estimators_samples_ : list of ndarray
    The subset of drawn samples (i.e., the in-bag samples) for each base
    estimator. Each subset is defined by a boolean mask.

estimators_features_ : list of ndarray
    The subset of drawn features for each base estimator.

classes_ : ndarray of shape (n_classes,)
    The classes labels.

n_classes_ : int or list
    The number of classes.

oob_score_ : float
    Score of the training dataset obtained using an out-of-bag estimate.

oob_decision_function_ : ndarray of shape (n_samples, n_classes)
    Decision function computed with out-of-bag estimate on the training
    set. If n_estimators is small it might be possible that a data point
    was never left out during the bootstrap. In this case,
    ``oob_decision_function_`` might contain NaN.

n_features_in_ : int
    Number of features in the input dataset.

    .. versionadded:: 0.9

feature_names_in_ : ndarray of shape (`n_features_in_`,)
    Names of features seen during `fit`. Defined only when `X` has feature
    names that are all strings.

    .. versionadded:: 0.9

See Also
--------
BalancedRandomForestClassifier : Random forest applying random-under
    sampling to balance the different bootstraps.

EasyEnsembleClassifier : Ensemble of AdaBoost classifier trained on
    balanced bootstraps.

RUSBoostClassifier : AdaBoost classifier were each bootstrap is balanced
    using random-under sampling at each round of boosting.

Notes
-----
This is possible to turn this classifier into a balanced random forest [5]_
by passing a :class:`~sklearn.tree.DecisionTreeClassifier` with
`max_features='auto'` as a base estimator.

See
:ref:`sphx_glr_auto_examples_ensemble_plot_comparison_ensemble_classifier.py`.

References
----------
.. [1] L. Breiman, "Pasting small votes for classification in large
       databases and on-line", Machine Learning, 36(1), 85-103, 1999.

.. [2] L. Breiman, "Bagging predictors", Machine Learning, 24(2), 123-140,
       1996.

.. [3] T. Ho, "The random subspace method for constructing decision
       forests", Pattern Analysis and Machine Intelligence, 20(8), 832-844,
       1998.

.. [4] G. Louppe and P. Geurts, "Ensembles on Random Patches", Machine
       Learning and Knowledge Discovery in Databases, 346-361, 2012.

.. [5] C. Chen Chao, A. Liaw, and L. Breiman. "Using random forest to
       learn imbalanced data." University of California, Berkeley 110,
       2004.

.. [6] R. Maclin, and D. Opitz. "An empirical evaluation of bagging and
       boosting." AAAI/IAAI 1997 (1997): 546-551.

.. [7] S. Hido, H. Kashima, and Y. Takahashi. "Roughly balanced bagging
       for imbalanced data." Statistical Analysis and Data Mining: The ASA
       Data Science Journal 2.5â€�6 (2009): 412-426.

.. [8] S. Wang, and X. Yao. "Diversity analysis on imbalanced data sets by
       using ensemble models." 2009 IEEE symposium on computational
       intelligence and data mining. IEEE, 2009.

Examples
--------
>>> from collections import Counter
>>> from sklearn.datasets import make_classification
>>> from sklearn.model_selection import train_test_split
>>> from sklearn.metrics import confusion_matrix
>>> from imblearn.ensemble import BalancedBaggingClassifier
>>> X, y = make_classification(n_classes=2, class_sep=2,
... weights=[0.1, 0.9], n_informative=3, n_redundant=1, flip_y=0,
... n_features=20, n_clusters_per_class=1, n_samples=1000, random_state=10)
>>> print('Original dataset shape %s' % Counter(y))
Original dataset shape Counter({{1: 900, 0: 100}})
>>> X_train, X_test, y_train, y_test = train_test_split(X, y,
...                                                     random_state=0)
>>> bbc = BalancedBaggingClassifier(random_state=42)
>>> bbc.fit(X_train, y_train)
BalancedBaggingClassifier(...)
>>> y_pred = bbc.predict(X_test)
>>> print(confusion_matrix(y_test, y_pred))
[[ 23   0]
 [  2 225]]
z1.4r   r   Úright)Úclosed>   ÚallÚautoÚmajorityúnot majorityúnot minorityÚbooleanÚfit_resampleN)r   ÚreplacementÚsamplerÚbase_estimatorg      ð?TFr(   )Úmax_samplesÚmax_featuresÚ	bootstrapÚbootstrap_featuresÚ	oob_scoreÚ
warm_startr   r.   r    r!   Úverboser/   c                ób   >• [         TU ]  UUUUUUUUUUS9
  Xl        X�l        X l        Xàl        g )N)
Ún_estimatorsr1   r2   r3   r4   r5   r6   r    r!   r7   )ÚsuperÚ__init__Ú	estimatorr   r.   r/   )Úselfr<   r9   r1   r2   r3   r4   r5   r6   r   r.   r    r!   r7   r/   Ú	__class__s                  €Ú]/srv/projetos/modelo_ml_acdoc/venv/lib/python3.13/site-packages/imblearn/ensemble/_bagging.pyr;   Ú"BalancedBaggingClassifier.__init__  sM   ø€ ô$ 	‰ÑØ%Ø#Ø%ØØ1ØØ!ØØ%Øð 	ñ 	
ð #ŒØ!2ÔØ&ÔØ�ó    c                 óÀ  >• [         TU ]  U5      n[        U R                  [        5      (       a—  U R
                  R                  S:w  a}  [        U R                  UU R
                  R                  5      R                  5        VVs0 s H/  u  p4[        R                  " U R                  U:H  5      S   S   U_M1     snnU l        U$ U R                  U l        U$ s  snnf )NÚbypassr   )r:   Ú_validate_yÚ
isinstancer   ÚdictÚsampler_Ú_sampling_typer   ÚitemsÚnpÚwhereÚclasses_Ú_sampling_strategy)r=   ÚyÚ	y_encodedÚkeyÚvaluer>   s        €r?   rD   Ú%BalancedBaggingClassifier._validate_y4  sÏ   ø€ Ü‘GÑ'¨Ó*ˆ	ä�t×-Ñ-¬t×4Ñ4Ø—‘×,Ñ,°Ó8ô #:Ø×*Ñ*ØØ—M‘M×0Ñ0ó#÷ ‘%“'ð	#ô'ò#‘J�Cô —’˜Ÿ™¨#Ñ-Ó.¨qÑ1°!Ñ4°eÒ;ñ#ò'ˆDÔ#ð Ðð '+×&<Ñ&<ˆDÔ#ØÐùó's   Â6Cc                 ó  • U R                   b  [        U R                   5      nO[        U5      nU R                  R                  S:w  a#  U R                  R	                  U R
                  S9  [        SU R                  4SU4/5      U l        g)zRCheck the estimator and the n_estimator attribute, set the
`estimator_` attribute.NrC   )r   r/   Ú
classifier)r<   r   rG   rH   Ú
set_paramsrM   r   Ú
estimator_)r=   Údefaultr<   s      r?   Ú_validate_estimatorÚ-BalancedBaggingClassifier._validate_estimatorF  st   € ð �>‰>Ñ%Ü˜dŸn™nÓ-‰Iä˜g›ˆIà�=‰=×'Ñ'¨8Ó3Ø�M‰M×$Ñ$°t×7NÑ7NÐ$ÑOä"Ø˜Ÿ™Ð'¨,¸	Ð)BÐCó
ˆ�rA   c                 óP   • [         R                  " S[        5        U R                  $ )z-Number of features when ``fit`` is performed.z’`n_features_` was deprecated in scikit-learn 1.0. This attribute will not be accessible when the minimum supported version of scikit-learn is 1.2.)ÚwarningsÚwarnÚFutureWarningÚn_features_in_)r=   s    r?   Ún_features_Ú%BalancedBaggingClassifier.n_features_V  s(   € ô 	�Šðô ô		
ð ×"Ñ"Ð"rA   )Úprefer_skip_nested_validationc                 óB   >• U R                  5         [        TU ]	  X5      $ )aÃ  Build a Bagging ensemble of estimators from the training set (X, y).

Parameters
----------
X : {array-like, sparse matrix} of shape (n_samples, n_features)
    The training input samples. Sparse matrices are accepted only if
    they are supported by the base estimator.

y : array-like of shape (n_samples,)
    The target values (class labels in classification, real numbers in
    regression).

Returns
-------
self : object
    Fitted estimator.
)Ú_validate_paramsr:   Úfit)r=   ÚXrN   r>   s      €r?   rd   ÚBalancedBaggingClassifier.fita  s    ø€ ð( 	×ÑÔÜ‰w‰{˜1Ó Ð rA   c                 óÎ   >• [        U5        U R                  c  [        U R                  S9U l        O[        U R                  5      U l        [        TU ]  XU R                  5      $ )N)r.   )	r   r/   r   r.   rG   r   r:   Ú_fitr1   )r=   re   rN   r1   Ú	max_depthÚsample_weightr>   s         €r?   rh   ÚBalancedBaggingClassifier._fitx  sW   ø€ Ü˜!Ôð �<‰<ÑÜ.Ø ×,Ñ,ñˆD�Mô " $§,¡,Ó/ˆDŒMô ‰w‰|˜A $×"2Ñ"2Ó3Ð3rA   Údecision_functionc                 ó   ^ ^^• [        T 5        T R                  TSS/SSSS9m[        T R                  T R                  5      u  p#m[        UT R                  S9" UU U4S j[        U5       5       5      n[        U5      T R                  -  nU$ )aD  Average of the decision functions of the base classifiers.

Parameters
----------
X : {array-like, sparse matrix} of shape (n_samples, n_features)
    The training input samples. Sparse matrices are accepted only if
    they are supported by the base estimator.

Returns
-------
score : ndarray of shape (n_samples, k)
    The decision function of the input samples. The columns correspond
    to the classes in sorted order, as they appear in the attribute
    ``classes_``. Regression and binary classification are special
    cases with ``k == 1``, otherwise ``k==n_classes``.
ÚcsrÚcscNF)Úaccept_sparseÚdtypeÚforce_all_finiteÚreset)r    r7   c           	   3   óª   >#   • U  HH  n[        [        5      " TR                  TU   TUS -       TR                  TU   TUS -       T5      v •  MJ     g7f)r   N)r   r   Úestimators_Úestimators_features_)Ú.0Úire   r=   Ústartss     €€€r?   Ú	<genexpr>Ú>BalancedBaggingClassifier.decision_function.<locals>.<genexpr>¨  sh   øé € ð F
ò #�ô Ô/Ô0Ø× Ñ  ¨¡¨V°A¸±E©]Ð;Ø×)Ñ)¨&°©)°f¸QÀ¹U±mÐDØ÷ð ò
 #ùs   ƒAA)	r
   Ú_validate_datar   r9   r    r   r7   ÚrangeÚsum)r=   re   r    Ú_Úall_decisionsÚ	decisionsry   s   ``    @r?   rl   Ú+BalancedBaggingClassifier.decision_functionˆ  s›   ú€ ô$ 	˜Ôð ×ÑØØ  %˜.ØØ"Øð  ð 
ˆô 2°$×2CÑ2CÀTÇ[Á[ÓQÑˆ�6ä ¨¸¿¹ÒEö F
ô ˜6”]óF
ó 
ˆô ˜Ó&¨×):Ñ):Ñ:ˆ	àÐrA   c                 óÀ   • [        U R                  R                   S35      n[        [	        S5      :  a   [        U 5        U R                  $ Ue! [         a    Uef = f)z2Attribute for older sklearn version compatibility.z+ object has no attribute 'base_estimator_'.z1.2)ÚAttributeErrorr>   Ú__name__Úsklearn_versionr	   r
   rV   r   )r=   Úerrors     r?   Úbase_estimator_Ú)BalancedBaggingClassifier.base_estimator_¶  sg   € ô Ø�~‰~×&Ñ&Ð'Ð'RÐSó
ˆô œ]¨5Ó1Ó1ðÜ Ô%Ø—‘Ð&ð ˆøô "ó Ø�ðús   ·A ÁAc                 óV   >• [         TU ]  5       nSnSnSnX!;   a	  XAU   U'   U$ X40X'   U$ )NÚ_xfail_checksÚcheck_estimators_nan_infz9Fails because the sampler removed infinity and NaN values)r:   Ú
_more_tags)r=   ÚtagsÚtags_keyÚfailing_testÚreasonr>   s        €r?   r�   Ú$BalancedBaggingClassifier._more_tagsÆ  sI   ø€ Ü‰wÑ!Ó#ˆØ"ˆØ1ˆØLˆØÓØ+1�‰N˜<Ñ(ð ˆð +Ð3ˆD‰NØˆrA   )rM   r<   rV   r.   r/   rG   r   )Né
   )NNN)$r…   Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r†   r	   ÚcopyÚdeepcopyr   Ú_parameter_constraintsr   Úupdater   ÚnumbersÚRealr   rF   Úcallabler   r;   rD   r   rX   Úpropertyr_   r   rd   rh   r   r   rl   rˆ   r�   Ú__static_attributes__Ú__classcell__)r>   s   @r?   r#   r#   +   sf  ø† ñHðV ™-¨Ó.Ó.Ø!%§¢Ð/@×/WÑ/WÓ!XÑà!%§¢Ð/MÓ!NÐà×!Ñ!ñ ˜Ÿ™ q¨!°GÑ<ÙÒVÓWØØð	"ð &˜;Ù" NÐ#3Ó4°dÐ;ñ		
ôð Ð1Ó1Ø"Ð#3Ð4ð Øð!ð
 ØØØ ØØØ ØØØØØ÷!!ñ !õFñ$ +AÓ*Bô 
ð  ñ#ó ð#ñ °Ñ6ô!ó 7ð!÷,4ñ  ‘.Ð!4Ó5Ó6ñ+ó 7ð+ðZ ñó ð÷	ó 	rA   r#   );r—   r˜   rœ   r[   ÚnumpyrJ   ÚsklearnÚsklearn.baser   Úsklearn.ensembler   Úsklearn.ensemble._baggingr   Úsklearn.ensemble._baser   Úsklearn.exceptionsr   Úsklearn.treer   Úsklearn.utils.fixesr	   Úsklearn.utils.validationr
   Úsklearn.utils.parallelr   r   ÚImportErrorÚModuleNotFoundErrorÚjoblibÚbaser   Úpipeliner   Úunder_samplingr   Úunder_sampling.baser   Úutilsr   r   r   Úutils._available_ifr   Úutils._docstringr   r   Úutils._param_validationr   r   r   Úutils.fixesr   Ú_commonr   r   Ú__version__r†   Ú_sampling_strategy_docstringr#   © rA   r?   Ú<module>r½      sÂ   ðÙ ?ó Û Û ã Û Ý Ý .Ý AÝ 8Ý -Ý /Ý -Ý 4ð,ç8õ
 *Ý Ý /Ý 2ß LÑ LÝ .ß Iß FÑ FÝ &ß Cá × 3Ñ 3Ó4€ñ Ø&×CÑCØØ(ñô
_Ð 6Ð8Ió _óñ
_øð/ 	Ð(Ð)ó ,Ýß+ð,ús   ÁC ÃC#Ã"C#