Ë
    ÜÍ:j¹O  ã                   ó    — d dl Z d dlmZ d dlZd dlmZmZmZ d dl	m
Z
 d dlmZ d dlmZmZmZ d dlmZ d dlmZmZmZmZmZ  G d	„ d
ee«      Zy)é    N)ÚIntegral)ÚBaseEstimatorÚTransformerMixinÚ_fit_context)ÚOneHotEncoder)Úresample)ÚIntervalÚOptionsÚ
StrOptions)Ú_weighted_percentile)Ú_check_feature_names_inÚ_check_sample_weightÚcheck_arrayÚcheck_is_fittedÚvalidate_datac                   ó*  — e Zd ZU dZ eeddd¬«      dg eh d£«      g eh d£«      g eh d	£«      g eee	j                  e	j                  h«      dg eed
dd¬«      dgdgdœZeed<   	 ddddddddœd„Z ed¬«      dd„«       Zd„ Zd„ Zd„ Zdd„Zy)ÚKBinsDiscretizeraž  
    Bin continuous data into intervals.

    Read more in the :ref:`User Guide <preprocessing_discretization>`.

    .. versionadded:: 0.20

    Parameters
    ----------
    n_bins : int or array-like of shape (n_features,), default=5
        The number of bins to produce. Raises ValueError if ``n_bins < 2``.

    encode : {'onehot', 'onehot-dense', 'ordinal'}, default='onehot'
        Method used to encode the transformed result.

        - 'onehot': Encode the transformed result with one-hot encoding
          and return a sparse matrix. Ignored features are always
          stacked to the right.
        - 'onehot-dense': Encode the transformed result with one-hot encoding
          and return a dense array. Ignored features are always
          stacked to the right.
        - 'ordinal': Return the bin identifier encoded as an integer value.

    strategy : {'uniform', 'quantile', 'kmeans'}, default='quantile'
        Strategy used to define the widths of the bins.

        - 'uniform': All bins in each feature have identical widths.
        - 'quantile': All bins in each feature have the same number of points.
        - 'kmeans': Values in each bin have the same nearest center of a 1D
          k-means cluster.

        For an example of the different strategies see:
        :ref:`sphx_glr_auto_examples_preprocessing_plot_discretization_strategies.py`.

    quantile_method : {"inverted_cdf", "averaged_inverted_cdf",
            "closest_observation", "interpolated_inverted_cdf", "hazen",
            "weibull", "linear", "median_unbiased", "normal_unbiased"},
            default="averaged_inverted_cdf"
            Method to pass on to np.percentile calculation when using
            strategy="quantile". Only `averaged_inverted_cdf` and `inverted_cdf`
            support the use of `sample_weight != None` when subsampling is not
            active.

            .. versionadded:: 1.7

            .. versionchanged:: 1.9
                The default value changed from `"linear"` to `"averaged_inverted_cdf"`.

    dtype : {np.float32, np.float64}, default=None
        The desired data-type for the output. If None, output dtype is
        consistent with input dtype. Only np.float32 and np.float64 are
        supported.

        .. versionadded:: 0.24

    subsample : int or None, default=200_000
        Maximum number of samples, used to fit the model, for computational
        efficiency.
        `subsample=None` means that all the training samples are used when
        computing the quantiles that determine the binning thresholds.
        Since quantile computation relies on sorting each column of `X` and
        that sorting has an `n log(n)` time complexity,
        it is recommended to use subsampling on datasets with a
        very large number of samples.

        .. versionchanged:: 1.3
            The default value of `subsample` changed from `None` to `200_000` when
            `strategy="quantile"`.

        .. versionchanged:: 1.5
            The default value of `subsample` changed from `None` to `200_000` when
            `strategy="uniform"` or `strategy="kmeans"`.

    random_state : int, RandomState instance or None, default=None
        Determines random number generation for subsampling.
        Pass an int for reproducible results across multiple function calls.
        See the `subsample` parameter for more details.
        See :term:`Glossary <random_state>`.

        .. versionadded:: 1.1

    Attributes
    ----------
    bin_edges_ : ndarray of ndarray of shape (n_features,)
        The edges of each bin. Contain arrays of varying shapes ``(n_bins_, )``
        Ignored features will have empty arrays.

    n_bins_ : ndarray of shape (n_features,), dtype=np.int64
        Number of bins per feature. Bins whose width are too small
        (i.e., <= 1e-8) are removed with a warning.

    n_features_in_ : int
        Number of features seen during :term:`fit`.

        .. versionadded:: 0.24

    feature_names_in_ : ndarray of shape (`n_features_in_`,)
        Names of features seen during :term:`fit`. Defined only when `X`
        has feature names that are all strings.

        .. versionadded:: 1.0

    See Also
    --------
    Binarizer : Class used to bin values as ``0`` or
        ``1`` based on a parameter ``threshold``.

    Notes
    -----

    For a visualization of discretization on different datasets refer to
    :ref:`sphx_glr_auto_examples_preprocessing_plot_discretization_classification.py`.
    On the effect of discretization on linear models see:
    :ref:`sphx_glr_auto_examples_preprocessing_plot_discretization.py`.

    In bin edges for feature ``i``, the first and last values are used only for
    ``inverse_transform``. During transform, bin edges are extended to::

      np.concatenate([-np.inf, bin_edges_[i][1:-1], np.inf])

    You can combine ``KBinsDiscretizer`` with
    :class:`~sklearn.compose.ColumnTransformer` if you only want to preprocess
    part of the features.

    ``KBinsDiscretizer`` might produce constant features (e.g., when
    ``encode = 'onehot'`` and certain bins do not contain any data).
    These features can be removed with feature selection algorithms
    (e.g., :class:`~sklearn.feature_selection.VarianceThreshold`).

    Examples
    --------
    >>> from sklearn.preprocessing import KBinsDiscretizer
    >>> X = [[-2, 1, -4,   -1],
    ...      [-1, 2, -3, -0.5],
    ...      [ 0, 3, -2,  0.5],
    ...      [ 1, 4, -1,    2]]
    >>> est = KBinsDiscretizer(
    ...     n_bins=3, encode='ordinal', strategy='uniform'
    ... )
    >>> est.fit(X)
    KBinsDiscretizer(...)
    >>> Xt = est.transform(X)
    >>> Xt  # doctest: +SKIP
    array([[ 0., 0., 0., 0.],
           [ 1., 1., 1., 0.],
           [ 2., 2., 2., 1.],
           [ 2., 2., 2., 2.]])

    Sometimes it may be useful to convert the data back into the original
    feature space. The ``inverse_transform`` function converts the binned
    data into the original feature space. Each value will be equal to the mean
    of the two bin edges.

    >>> est.bin_edges_[0]
    array([-2., -1.,  0.,  1.])
    >>> est.inverse_transform(Xt)
    array([[-1.5,  1.5, -3.5, -0.5],
           [-0.5,  2.5, -2.5, -0.5],
           [ 0.5,  3.5, -1.5,  0.5],
           [ 0.5,  3.5, -1.5,  1.5]])

    While this preprocessing step can be an optimization, it is important
    to note the array returned by ``inverse_transform`` will have an internal type
    of ``np.float64`` or ``np.float32``, denoted by the ``dtype`` input argument.
    This can drastically increase the memory usage of the array. See the
    :ref:`sphx_glr_auto_examples_cluster_plot_face_compress.py`
    where `KBinsDescretizer` is used to cluster the image into bins and increases
    the size of the image by 8x.
    é   NÚleft)Úclosedz
array-like>   úonehot-denseÚonehotÚordinal>   ÚkmeansÚuniformÚquantile>	   ÚhazenÚlinearÚweibullÚinverted_cdfÚmedian_unbiasedÚnormal_unbiasedÚclosest_observationÚaveraged_inverted_cdfÚinterpolated_inverted_cdfé   Úrandom_state©Ún_binsÚencodeÚstrategyÚquantile_methodÚdtypeÚ	subsampler'   Ú_parameter_constraintsr   r   r$   i@ )r*   r+   r,   r-   r.   r'   c                óf   — || _         || _        || _        || _        || _        || _        || _        y ©Nr(   )Úselfr)   r*   r+   r,   r-   r.   r'   s           úz/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/sklearn/preprocessing/_discretization.pyÚ__init__zKBinsDiscretizer.__init__Û   s7   € ð ˆŒØˆŒØ ˆŒØ.ˆÔØˆŒ
Ø"ˆŒØ(ˆÕó    T)Úprefer_skip_nested_validationc                 óL	  — t        | |d¬«      }| j                  t        j                  t        j                  fv r| j                  }n|j                  }|j
                  \  }}|�t        |||j                  ¬«      }| j                  �5|| j                  kD  r&t        |d| j                  | j                  |¬«      }d}|j
                  d   }| j                  |«      }t        j                  |t        ¬«      }| j                  }	| j                  dk(  r|	dvr|�t        d	|	› d
�«      ‚| j                  dk7  r|�|dk7  }
nt!        d«      }
t#        |«      D �]›  }|dd…|f   }||
   j%                  «       }||
   j'                  «       }||k(  rUt)        j*                  d|z  «       d||<   t        j,                  t        j.                   t        j.                  g«      ||<   Œ�| j                  dk(  r"t        j0                  ||||   dz   «      ||<   �nS| j                  dk(  r‡t        j0                  dd||   dz   «      }i }|	dk7  r|€|	|d<   |€>t        j2                  t        j4                  ||fi |¤Žt        j                  ¬«      ||<   nÙ|	dk(  rdnd}t7        ||||¬«      ||<   n½| j                  dk(  r®ddlm} t        j0                  ||||   dz   «      }|dd |dd z   dd…df   dz  } |||   |d¬«      }|j=                  |dd…df   |¬«      j>                  dd…df   }|jA                  «        |dd |dd z   dz  ||<   t        jB                  |||   |f   ||<   | j                  dv s�Œ!t        jD                  ||   t        j.                  ¬«      dkD  }||   |   ||<   tG        ||   «      dz
  ||   k7  s�Œpt)        j*                  d|z  «       tG        ||   «      dz
  ||<   �Œž || _$        || _%        d| jL                  v rŽtO        | jJ                  D �cg c]  }t        jP                  |«      ‘Œ c}| jL                  dk(  |¬«      | _)        | jR                  j=                  t        j                  dtG        | jJ                  «      f«      «       | S c c}w ) a‹  
        Fit the estimator.

        Parameters
        ----------
        X : array-like of shape (n_samples, n_features)
            Data to be discretized.

        y : None
            Ignored. This parameter exists only for compatibility with
            :class:`~sklearn.pipeline.Pipeline`.

        sample_weight : ndarray of shape (n_samples,)
            Contains weight values to be associated with each sample.

            .. versionadded:: 1.3

            .. versionchanged:: 1.7
               Added support for strategy="uniform".

        Returns
        -------
        self : object
            Returns the instance itself.
        Únumeric©r-   NT)ÚreplaceÚ	n_samplesr'   Úsample_weightr&   r   )r    r$   z¢When fitting with strategy='quantile' and sample weights, quantile_method should either be set to 'averaged_inverted_cdf' or 'inverted_cdf', got quantile_method='z
' instead.r   z3Feature %d is constant and will be replaced with 0.r   éd   r   Úmethodr$   F)Úaverager   )ÚKMeanséÿÿÿÿç      à?)Ú
n_clustersÚinitÚn_init)r<   )r   r   )Úto_beging:Œ0âŽyE>zqBins whose width are too small (i.e., <= 1e-8) in feature %d are removed. Consider decreasing the number of bins.r   )Ú
categoriesÚsparse_outputr-   )*r   r-   ÚnpÚfloat64Úfloat32Úshaper   r.   r   r'   Ú_validate_n_binsÚzerosÚobjectr,   r+   Ú
ValueErrorÚsliceÚrangeÚminÚmaxÚwarningsÚwarnÚarrayÚinfÚlinspaceÚasarrayÚ
percentiler   Úsklearn.clusterr@   ÚfitÚcluster_centers_ÚsortÚr_Úediff1dÚlenÚ
bin_edges_Ún_bins_r*   r   ÚarangeÚ_encoder)r2   ÚXÚyr<   Úoutput_dtyper;   Ú
n_featuresr)   Ú	bin_edgesr,   Únnz_weight_maskÚjjÚcolumnÚcol_minÚcol_maxÚpercentile_levelsÚpercentile_kwargsr?   r@   Úuniform_edgesrD   ÚkmÚcentersÚmaskÚis                            r3   r]   zKBinsDiscretizer.fitî   s�  € ô6 ˜$ ¨Ô3ˆà�:‰:œ"Ÿ*™*¤b§j¡jÐ1Ñ1ØŸ:™:‰LàŸ7™7ˆLà !§¡Ñˆ	�:àÐ$Ü0°ÀÈÏÉÔQˆMà�>‰>Ð%¨)°d·n±nÒ*Dô ØØØŸ.™.Ø!×.Ñ.Ø+ôˆAð !ˆMà—W‘W˜Q‘Zˆ
Ø×&Ñ& zÓ2ˆä—H‘H˜Z¬vÔ6ˆ	à×.Ñ.ˆð �M‰M˜ZÒ'ØÐ'PÑPØÐ)äð8à8GÐ7HÈ
ðTóð ð �=‰=˜JÒ&¨=Ð+Dð ,¨qÑ0‰Oô $ D›kˆOä˜
Ó#ó A	8ˆBØ’q˜"�u‘XˆFØ˜_Ñ-×1Ñ1Ó3ˆGØ˜_Ñ-×1Ñ1Ó3ˆGà˜'Ò!Ü—‘ØIÈBÑNôð ��r‘
Ü "§¡¬2¯6©6¨'´2·6±6Ð):Ó ;�	˜"‘Øà�}‰} 	Ò)Ü "§¡¨G°W¸fÀR¹jÈ1¹nÓ M�	˜"“à—‘ *Ò,Ü$&§K¡K°°3¸¸r¹
ÀQ¹Ó$GÐ!ð
 %'Ð!Ø" hÒ.°=Ð3HØ2AÐ% hÑ/à Ð(Ü$&§J¡JÜŸ™ fÐ.?ÑUÐCTÑUÜ Ÿj™jô%�I˜b’Mð !0Ð3JÒ J™ÐPUð ô %9Ø Ð/@È'ô%�I˜b’Mð —‘ (Ò*Ý2ô !#§¡¨G°W¸fÀR¹jÈ1¹nÓ M�Ø% a bÐ)¨M¸#¸2Ð,>Ñ>ÂÀ4ÀÑHÈ3ÑN�ñ  v¨b¡z¸ÀQÔG�ØŸ&™&Øš1˜d˜7‘O°=ð !ó ç"Ñ"¢1 a 4ñ)�ð —‘”Ø!(¨¨ ¨w°s¸¨|Ñ!;¸sÑ B�	˜"‘Ü "§¡ g¨y¸©}¸gÐ&EÑ F�	˜"‘ð �}‰}Ð 6Ó6Ü—z‘z )¨B¡-¼"¿&¹&ÔAÀDÑH�Ø )¨"¡¨dÑ 3�	˜"‘Ü�y ‘}Ó%¨Ñ)¨V°B©ZÔ7Ü—M‘Mð9à;=ñ>ôô
 "% Y¨r¡]Ó!3°aÑ!7�F˜2“JðCA	8ðF $ˆŒØˆŒà�t—{‘{Ñ"Ü)Ø26·,±,Ö?¨QœBŸI™I a�LÒ?Ø"Ÿk™k¨XÑ5Ø"ôˆDŒMð �M‰M×ÑœbŸh™h¨¬3¨t¯|©|Ó+<Ð'=Ó>Ô?àˆùò @s   Ð$R!c                 óà  — | j                   }t        |t        «      rt        j                  ||t
        ¬«      S t        |t
        dd¬«      }|j                  dkD  s|j                  d   |k7  rt        d«      ‚|dk  ||k7  z  }t        j                  |«      d   }|j                  d   dkD  rAd	j                  d
„ |D «       «      }t        dj                  t        j                  |«      «      ‚|S )z0Returns n_bins_, the number of bins per feature.r9   TF)r-   ÚcopyÚ	ensure_2dr&   r   z8n_bins must be a scalar or array of shape (n_features,).r   z, c              3   ó2   K  — | ]  }t        |«      –— Œ y ­wr1   )Ústr)Ú.0rw   s     r3   ú	<genexpr>z4KBinsDiscretizer._validate_n_bins.<locals>.<genexpr>¦  s   è ø€ ÒB¨1¤ A§ÑBùs   ‚zk{} received an invalid number of bins at indices {}. Number of bins must be at least 2, and must be an int.)r)   Ú
isinstancer   rI   ÚfullÚintr   ÚndimrL   rP   ÚwhereÚjoinÚformatr   Ú__name__)r2   rj   Ú	orig_binsr)   Úbad_nbins_valueÚviolating_indicesÚindicess          r3   rM   z!KBinsDiscretizer._validate_n_bins—  sÛ   € à—K‘Kˆ	Ü�i¤Ô*Ü—7‘7˜: y¼Ô<Ð<ä˜Y¬c¸ÈÔNˆà�;‰;˜Š?˜fŸl™l¨1™o°Ò;ÜÐWÓXÐXà! A™:¨&°IÑ*=Ñ>ˆäŸH™H _Ó5°aÑ8ÐØ×"Ñ" 1Ñ%¨Ò)Ø—i‘iÑBÐ0AÔBÓBˆGÜð:ç:@¹&Ü$×-Ñ-¨wó;óð ð ˆr5   c                 ó€  — t        | «       | j                  € t        j                  t        j                  fn| j                  }t        | |d|d¬«      }| j                  }t        |j                  d   «      D ].  }t        j                  ||   dd |dd…|f   d¬«      |dd…|f<   Œ0 | j                  d	k(  r|S d}d
| j                  v r1| j                  j                  }|j                  | j                  _        	 | j                  j                  |«      }|| j                  _        |S # || j                  _        w xY w)a‹  
        Discretize the data.

        Parameters
        ----------
        X : array-like of shape (n_samples, n_features)
            Data to be discretized.

        Returns
        -------
        Xt : {ndarray, sparse matrix}, dtype={np.float32, np.float64}
            Data in the binned space. Will be a sparse matrix if
            `self.encode='onehot'` and ndarray otherwise.
        NTF)ry   r-   Úresetr&   rA   Úright)Úsider   r   )r   r-   rI   rJ   rK   r   rc   rR   rL   Úsearchsortedr*   rf   Ú	transform)r2   rg   r-   ÚXtrk   rm   Ú
dtype_initÚXt_encs           r3   r�   zKBinsDiscretizer.transform°  s  € ô 	˜Ôð -1¯J©JÐ,>”—‘œRŸZ™ZÑ(ÀDÇJÁJˆÜ˜4 ¨°UÀ%ÔHˆà—O‘Oˆ	Ü˜Ÿ™ ™Ó$ò 	VˆBÜŸ™¨	°"©°a¸Ð(;¸RÂÀ2À¹YÈWÔUˆBŠq�"ˆuŠIð	Vð �;‰;˜)Ò#ØˆIàˆ
Ø�t—{‘{Ñ"ØŸ™×,Ñ,ˆJØ"$§(¡(ˆD�M‰MÔð	-Ø—]‘]×,Ñ,¨RÓ0ˆFð #-ˆD�M‰MÔØˆøð #-ˆD�M‰MÕús   Ã<D* Ä*D=c                 ó&  — t        | «       d| j                  v r| j                  j                  |«      }t	        |dt
        j                  t
        j                  f¬«      }| j                  j                  d   }|j                  d   |k7  r(t        dj                  ||j                  d   «      «      ‚t        |«      D ]O  }| j                  |   }|dd |dd z   d	z  }||dd…|f   j                  t
        j                  «         |dd…|f<   ŒQ |S )
aÚ  
        Transform discretized data back to original feature space.

        Note that this function does not regenerate the original data
        due to discretization rounding.

        Parameters
        ----------
        X : array-like of shape (n_samples, n_features)
            Transformed data in the binned space.

        Returns
        -------
        X_original : ndarray, dtype={np.float32, np.float64}
            Data in the original feature space.
        r   T)ry   r-   r   r&   z8Incorrect number of features. Expecting {}, received {}.NrA   rB   )r   r*   rf   Úinverse_transformr   rI   rJ   rK   rd   rL   rP   r…   rR   rc   ÚastypeÚint64)r2   rg   ÚXinvrj   rm   rk   Úbin_centerss          r3   r•   z"KBinsDiscretizer.inverse_transform×  s  € ô$ 	˜Ôà�t—{‘{Ñ"Ø—‘×/Ñ/°Ó2ˆAä˜1 4´·
±
¼B¿J¹JÐ/GÔHˆØ—\‘\×'Ñ'¨Ñ*ˆ
Ø�:‰:�a‰=˜JÒ&ÜØJ×QÑQØ §
¡
¨1¡óóð ô ˜
Ó#ò 	FˆBØŸ™¨Ñ+ˆIØ$ Q R˜=¨9°S°b¨>Ñ9¸SÑ@ˆKØ% tªA¨r¨E¡{×&:Ñ&:¼2¿8¹8Ó&DÑEˆD’�B�ŠKð	Fð
 ˆr5   c                 ó„   — t        | d«       t        | |«      }t        | d«      r| j                  j	                  |«      S |S )aÔ  Get output feature names.

        Parameters
        ----------
        input_features : array-like of str or None, default=None
            Input features.

            - If `input_features` is `None`, then `feature_names_in_` is
              used as feature names in. If `feature_names_in_` is not defined,
              then the following input feature names are generated:
              `["x0", "x1", ..., "x(n_features_in_ - 1)"]`.
            - If `input_features` is an array-like, then `input_features` must
              match `feature_names_in_` if `feature_names_in_` is defined.

        Returns
        -------
        feature_names_out : ndarray of str objects
            Transformed feature names.
        Ún_features_in_rf   )r   r   Úhasattrrf   Úget_feature_names_out)r2   Úinput_featuress     r3   r�   z&KBinsDiscretizer.get_feature_names_outþ  sB   € ô( 	˜Ð.Ô/Ü0°°~ÓFˆÜ�4˜Ô$Ø—=‘=×6Ñ6°~ÓFÐFð Ðr5   )é   )NNr1   )r†   Ú
__module__Ú__qualname__Ú__doc__r	   r   r   r
   ÚtyperI   rJ   rK   r/   ÚdictÚ__annotations__r4   r   r]   rM   r�   r•   r�   © r5   r3   r   r      sã   … ñhñV ˜H a¨°fÔ=¸|ÐLÙÒCÓDÐEÙÒ AÓBÐCáò
óð
ñ ˜$ §¡¨R¯Z©ZÐ 8Ó9¸4Ð@Ù˜x¨¨D¸Ô@À$ÐGØ'Ð(ñ+$Ð˜Dó ð4 ð)ð ØØ/ØØØô)ñ& °Ô5òfó 6ðfòPò2%òN%ôNr5   r   )rU   Únumbersr   ÚnumpyrI   Úsklearn.baser   r   r   Úsklearn.preprocessing._encodersr   Úsklearn.utilsr   Úsklearn.utils._param_validationr	   r
   r   Úsklearn.utils.statsr   Úsklearn.utils.validationr   r   r   r   r   r   r¦   r5   r3   ú<module>r¯      s@   ðó
 Ý ã ç FÑ FÝ 9Ý "ß IÑ IÝ 4÷õ ô@Ð'¨õ @r5   