From b473b05d1ad89daac3aa4e622f2d36cf537d9ea3 Mon Sep 17 00:00:00 2001 From: apbassett <43486400+apbassett@users.noreply.github.com> Date: Thu, 17 Sep 2026 17:47:31 -0400 Subject: [PATCH 1/5] IFA Suggestion bugfixes, warning cleanup --- howso/utilities/feature_attributes/base.py | 43 +++++++---- howso/utilities/feature_attributes/pandas.py | 27 +++---- .../feature_attributes/suggestions.py | 75 +++++++++---------- .../tests/test_infer_feature_attributes.py | 30 ++++++++ .../feature_attributes/tests/test_warnings.py | 28 ++++++- .../utilities/feature_attributes/warnings.py | 69 ++++++++++++----- 6 files changed, 178 insertions(+), 94 deletions(-) diff --git a/howso/utilities/feature_attributes/base.py b/howso/utilities/feature_attributes/base.py index 83f471ea..ba7ac5f0 100644 --- a/howso/utilities/feature_attributes/base.py +++ b/howso/utilities/feature_attributes/base.py @@ -376,12 +376,22 @@ def _validate_bounds(self, data: pd.DataFrame, feature: str, f'"{feature}" had {additional_errors} additional values outside of bounds that were not displayed.') return errors + @staticmethod + def _is_numeric_dtype(dtype: str | np.dtype | pd.api.extensions.ExtensionDtype | pd.CategoricalDtype) -> bool: + """Return whether `dtype` holds numbers, i.e. an integer, nullable integer, or float dtype.""" + try: + dtype = pd.api.types.pandas_dtype(dtype) + except TypeError: + return False + return pd.api.types.is_numeric_dtype(dtype) and not pd.api.types.is_bool_dtype(dtype) + def _validate_dtype(self, data: pd.DataFrame, feature: str, expected_dtype: str | pd.CategoricalDtype, coerced_df: pd.DataFrame, coerce: bool = False, localize_datetimes: bool = True) -> list[str]: """Validate the data type of a feature and optionally attempt to coerce.""" errors = [] series = coerced_df[feature] + actual_dtype = data[feature].dtype is_valid = False coerce_err = "" @@ -392,8 +402,8 @@ def _validate_dtype(self, data: pd.DataFrame, feature: str, if coerce: coerced_df[feature] = series is_valid = True - except Exception: # noqa: Intentionally broad - pass + except Exception as err: # noqa: Intentionally broad + coerce_err = str(err) elif expected_dtype == "datetime64": try: format = self[feature]["date_time_format"] # pyright: ignore[reportTypedDictNotRequiredAccess] @@ -401,17 +411,17 @@ def _validate_dtype(self, data: pd.DataFrame, feature: str, format = "ISO8601" series = pd.to_datetime(coerced_df[feature], format=format) if coerce: - if localize_datetimes and not isinstance(series, pd.DatetimeTZDtype): + if localize_datetimes and not isinstance(series.dtype, pd.DatetimeTZDtype): coerced_df[feature] = series.dt.tz_localize( "UTC", ambiguous="infer", nonexistent="NaT" ) else: coerced_df[feature] = series is_valid = True - except Exception: # noqa: Intentionally broad - pass + except Exception as err: # noqa: Intentionally broad + coerce_err = str(err) # Else, compare the dtype directly - elif data[feature].dtype.name == expected_dtype: + elif actual_dtype.name == expected_dtype: is_valid = True # If the feature can be converted, consider it valid (slightly differing numeric types, etc.) else: @@ -420,22 +430,22 @@ def _validate_dtype(self, data: pd.DataFrame, feature: str, if coerce: coerced_df[feature] = series is_valid = True - except pd.errors.IntCastingNaNError: - # If this happens, there is a null value, thus a float dtype is OK - if pd.api.types.is_float_dtype(series): - is_valid = True except Exception as err: # noqa: Intentionally broad + # Integer, nullable integer, and float columns all carry the same values to the + # engine, so a numeric feature is described faithfully by any numeric dtype, even + # one that cannot be cast losslessly. Such a column keeps its original dtype. + is_valid = self._is_numeric_dtype(expected_dtype) and self._is_numeric_dtype(actual_dtype) coerce_err = str(err) # Raise warnings if the types do not match if not is_valid: if coerce: errors.append(f"Expected dtype '{expected_dtype}' for feature '{feature}' " - f"but could not coerce:\nActual dtype: {data[feature].dtype}" + f"but could not coerce:\nActual dtype: {actual_dtype}" f"\nError raised from Pandas.astype():\n\n{coerce_err}") else: errors.append(f"Feature '{feature}' should be '{expected_dtype}' dtype, but found " - f"'{data[feature].dtype}'") + f"'{actual_dtype}'") return errors @@ -482,9 +492,10 @@ def _validate_df(self, data: pd.DataFrame, coerce: bool = False, errors.extend(self._validate_dtype(data, feature, "int64", coerced_df, coerce=coerce)) elif attributes.get("data_type") == "boolean": - # Check type (boolean) - errors.extend(self._validate_dtype(data, feature, "bool", - coerced_df, coerce=coerce)) + # Check type (boolean). A boolean column that also holds nulls stays an object + # column, since casting it to `bool` would turn every null into `False`. + errors.extend(self._validate_dtype(data, feature, "bool", coerced_df, + coerce=coerce and not data[feature].isna().any())) elif attributes.get("bounds") and attributes["bounds"].get("allowed"): # pyright: ignore[reportTypedDictNotRequiredAccess] # Check type (categorical) schema_dtype = pd.CategoricalDtype(attributes["bounds"]["allowed"], # pyright: ignore[reportTypedDictNotRequiredAccess] @@ -2027,7 +2038,7 @@ def _process_rare_values( # noqa: PLR0912, PLR0915 max_chunk_size=max_distilled_cases) else: # Set a small default - max_distilled_cases = 25_000 + max_distilled_cases = 50_000 # Workflow 1: User provided a config with protected multipliers; may need to compute unprotected multipliers if preserve_rare_values_config is not None: diff --git a/howso/utilities/feature_attributes/pandas.py b/howso/utilities/feature_attributes/pandas.py index 0e5d6962..4a608002 100644 --- a/howso/utilities/feature_attributes/pandas.py +++ b/howso/utilities/feature_attributes/pandas.py @@ -783,23 +783,18 @@ def _infer_floating_point_attributes(self, feature_name: str) -> dict: for r in col_array ]) - # specify decimal place. Proceed with training but issue a warning. + # Specify decimal places for features the engine can represent exactly. Features beyond + # that precision are trained without the attribute and reported in a single warning. if pd.api.types.is_float_dtype(col.dtype): - try: - if getattr(col.dtype, 'itemsize') <= 8: - attributes['decimal_places'] = decimals - else: - warnings.warn( - f'Feature "{feature_name}" contains floating point ' - 'values that exceed the maximum supported precision ' - 'of 64 bits.' - ) - except AttributeError: - warnings.warn( - f'Feature "{feature_name}" may contain floating point ' - 'values that exceed the maximum supported precision ' - 'of 64 bits.' - ) + itemsize = getattr(col.dtype, 'itemsize', None) + if itemsize is None: + self.warnings_collector.triage( + IFAWarningEmitterType.POSSIBLE_EXCESSIVE_FLOAT_PRECISION, feature_name) + elif itemsize <= 8: + attributes['decimal_places'] = decimals + else: + self.warnings_collector.triage( + IFAWarningEmitterType.EXCESSIVE_FLOAT_PRECISION, feature_name) return attributes diff --git a/howso/utilities/feature_attributes/suggestions.py b/howso/utilities/feature_attributes/suggestions.py index 8c466508..fc03e726 100644 --- a/howso/utilities/feature_attributes/suggestions.py +++ b/howso/utilities/feature_attributes/suggestions.py @@ -342,55 +342,50 @@ def summary(self) -> str: return (f"Found {_count(num_values, 'rare value')} across {_count(len(self._prvc), 'column')} " "whose signal may be lost during data distillation workflows") - def apply(self, attributes: dict) -> None: + def _warn_default_max_distilled_cases(self, addendum: str = "") -> None: + """ + Warn that the case weight multipliers were computed from a default ``max_distilled_cases``. + + Parameters + ---------- + addendum : str, default "" + An additional sentence appended to the warning, describing the consequence for the + calling method. + """ + warnings.warn( + "The computed case weights for rare value multipliers are likely inaccurate as " + "`max_distilled_cases` was not provided to `infer_feature_attributes`. Please provide " + "this parameter or be aware that the case weight multipliers were computed based on a " + "default `max_distilled_cases` value of 50,000. " + "An accurate `max_distilled_cases` enables Howso to correctly weight the influence of rare " + "values in the data, since the weighting is calibrated proportionally to the number of cases " + "remaining after distillation." + addendum, + UserWarning, + # Point past this helper at the caller of the public method that invoked it. + stacklevel=4, + ) + + def apply(self, attributes: Mapping[str, Any]) -> None: """Apply the computed rare values preservation config to the FeatureAttributesBase object.""" if not self._user_set_mdc: - warnings.warn( - "The computed case weights for Rare values multipliers are likely inaccurate as " - "`max_distilled_cases` was not provided to `infer_feature_attributes`. Please provide " - "this parameter or be aware that the case weight multipliers were computed based on a " - "default `max_distilled_cases` value of 25,000. " - "An accurate max_distilled_cases enables Howso to correctly weight the influence of rare " - "values in the data, since the weighting is calibrated proportionally to the number of cases " - "remaining after distillation. Since an inaccurate value may result in rare values being " - "under-weighted or over-weighted, this suggestion was not applied.", - UserWarning, - stacklevel=3, + self._warn_default_max_distilled_cases( + " Since an inaccurate value may result in rare values being under-weighted or " + "over-weighted, this suggestion was not applied." ) - if self._user_set_mdc: - for feature, config in self._prvc.items(): - attributes[feature]["preserve_rare_values"] = config + return + for feature, config in self._prvc.items(): + attributes[feature]["preserve_rare_values"] = config - def get_config(self) -> FullPreserveRareValuesConfig: + def get_config(self, enable_warnings: bool = True) -> FullPreserveRareValuesConfig: """Get the `preserve_rare_values_config` for use in future calls to `infer_feature_attributes`.""" - if not self._user_set_mdc: - warnings.warn( - "The computed case weights for Rare values multipliers are likely inaccurate as " - "`max_distilled_cases` was not provided to `infer_feature_attributes`. Please provide " - "this parameter or be aware that the case weight multipliers were computed based on a " - "default `max_distilled_cases` value of 25,000. " - "An accurate max_distilled_cases enables Howso to correctly weight the influence of rare " - "values in the data, since the weighting is calibrated proportionally to the number of cases " - "remaining after distillation.", - UserWarning, - stacklevel=3, - ) + if not self._user_set_mdc and enable_warnings: + self._warn_default_max_distilled_cases() return self._prvc def get_values_map(self) -> PreserveRareValuesMap: """Get the `preserve_rare_values_map` for use in future calls to `infer_feature_attributes.""" if not self._user_set_mdc: - warnings.warn( - "The computed case weights for Rare values multipliers are likely inaccurate as " - "`max_distilled_cases` was not provided to `infer_feature_attributes`. Please provide " - "this parameter or be aware that the case weight multipliers were computed based on a " - "default `max_distilled_cases` value of 25,000. " - "An accurate max_distilled_cases enables Howso to correctly weight the influence of rare " - "values in the data, since the weighting is calibrated proportionally to the number of cases " - "remaining after distillation.", - UserWarning, - stacklevel=3, - ) + self._warn_default_max_distilled_cases() values_map = {} for feature, config in self._prvc.items(): multipliers = config["protected_values_multipliers"] @@ -401,7 +396,7 @@ def merge(self, other: IFASuggestion) -> None: """Merge another PRVSuggestion into this one if there are no conflicts.""" if not isinstance(other, PRVSuggestion): raise TypeError(f"Cannot merge {type(other).__name__} into PRVSuggestion.") - for feature, config in other.get_config().items(): + for feature, config in other.get_config(enable_warnings=False).items(): if feature not in self._prvc: self._prvc[feature] = config elif self._prvc[feature] != config: diff --git a/howso/utilities/feature_attributes/tests/test_infer_feature_attributes.py b/howso/utilities/feature_attributes/tests/test_infer_feature_attributes.py index bdf10507..e7447ca9 100644 --- a/howso/utilities/feature_attributes/tests/test_infer_feature_attributes.py +++ b/howso/utilities/feature_attributes/tests/test_infer_feature_attributes.py @@ -308,6 +308,36 @@ def test_get_feature_type_raises(data, data_type): infer_feature_attributes(df) +def test_excessive_float_precision_warning(): + """Test that features exceeding 64-bit float precision are reported in a single warning.""" + # Place this here to avoid circular import + from howso.utilities.feature_attributes.pandas import InferFeatureAttributesDataFrame + if not hasattr(np, "float128"): + pytest.skip("Unsupported platform") + + df = pd.DataFrame({ + "a": np.arange(20, dtype=np.float128) + 0.5, + "b": np.arange(20, dtype=np.float128) * 1.5, + "c": np.arange(20, dtype="float64") + 0.25, + }) + ifa = InferFeatureAttributesDataFrame(df) + ifa.attributes = {} + attributes = {feature: ifa._infer_floating_point_attributes(feature) for feature in df.columns} + + # Features beyond the supported precision get no `decimal_places` + assert "decimal_places" not in attributes["a"] + assert "decimal_places" not in attributes["b"] + assert attributes["c"]["decimal_places"] == 2 + + with pytest.warns(UserWarning, match="exceed the maximum supported precision") as record: + ifa.warnings_collector.emit_all() + + assert len(record) == 1 + message = str(record[0].message) + assert "- a" in message and "- b" in message + assert "- c" not in message + + @pytest.mark.parametrize("should_fail, data", [ (True, [[1]]), (True, {3: [1]}), diff --git a/howso/utilities/feature_attributes/tests/test_warnings.py b/howso/utilities/feature_attributes/tests/test_warnings.py index 3280c2e7..c1b76cb6 100644 --- a/howso/utilities/feature_attributes/tests/test_warnings.py +++ b/howso/utilities/feature_attributes/tests/test_warnings.py @@ -11,7 +11,31 @@ def test_warnings_emitters(): collector.triage(IFAWarningEmitterType.MISSING_TZ_FEATURES, "b") collector.triage(IFAWarningEmitterType.UNKNOWN_DATETIME_FORMAT, "c") collector.triage(IFAWarningEmitterType.UTC_OFFSET, "d") + collector.triage(IFAWarningEmitterType.EXCESSIVE_FLOAT_PRECISION, "e") + collector.triage(IFAWarningEmitterType.POSSIBLE_EXCESSIVE_FLOAT_PRECISION, "f") - with pytest.warns(UserWarning, match=r"- [a-d]") as record: + with pytest.warns(UserWarning, match=r"- [a-f]") as record: collector.emit_all() - assert len(record) == 4 + assert len(record) == 6 + + +# `SIMPLE` collects whole messages rather than feature names, so it emits one warning per message. +@pytest.mark.parametrize("emitter_type", [t for t in IFAWarningEmitterType if t != IFAWarningEmitterType.SIMPLE]) +def test_warnings_emitters_list_all_features(emitter_type): + """Test that features sharing an emitter are listed in a single warning.""" + collector = IFAWarningCollector() + for feature in ("a", "b", "c"): + collector.triage(emitter_type, feature) + + with pytest.warns(UserWarning) as record: + collector.emit_all() + + assert len(record) == 1 + message = str(record[0].message) + assert all(feature in message for feature in ("a", "b", "c")) + + +def test_warnings_emitters_unknown_type(): + """Test that an unknown emitter type is rejected.""" + with pytest.raises(ValueError, match="Unknown `emitter_type` provided."): + IFAWarningCollector().triage("not_an_emitter_type", "a") diff --git a/howso/utilities/feature_attributes/warnings.py b/howso/utilities/feature_attributes/warnings.py index ff985634..8d217f2c 100644 --- a/howso/utilities/feature_attributes/warnings.py +++ b/howso/utilities/feature_attributes/warnings.py @@ -11,6 +11,8 @@ class IFAWarningEmitterType(Enum): UNKNOWN_DATETIME_FORMAT = "unknown_datetime_format" UTC_OFFSET = "utc_offset" VALUE_COUNTS_PROCESSING = "value_counts_processing" + EXCESSIVE_FLOAT_PRECISION = "excessive_float_precision" + POSSIBLE_EXCESSIVE_FLOAT_PRECISION = "possible_excessive_float_precision" SIMPLE = "simple" @@ -95,6 +97,31 @@ def emit(self): "suggested or computed `preserve_rare_values` configurations`.", UserWarning) +class FloatPrecisionWarningEmitter(IFAWarningEmitter): + """Base emitter for warnings about float features that exceed the precision the engine supports.""" + + #: How the warning relates the features to the precision limit. + _certainty: str + + def emit(self): + """Emit the warning.""" + warnings.warn(f"The following features {self._certainty} floating point values that exceed the " + f"maximum supported precision of 64 bits: {self.features_list}" + "\nThese features are trained without a `decimal_places` attribute.", UserWarning) + + +class ExcessiveFloatPrecisionWarningEmitter(FloatPrecisionWarningEmitter): + """Emitter for a warning about float features whose dtype is wider than 64 bits.""" + + _certainty = "contain" + + +class PossibleExcessiveFloatPrecisionWarningEmitter(FloatPrecisionWarningEmitter): + """Emitter for a warning about float features whose dtype does not report its size.""" + + _certainty = "may contain" + + class SimpleWarningEmitter(IFAWarningEmitter): """Emitter for simple warnings that are saved via the `features_list`.""" @@ -107,6 +134,18 @@ def emit(self): class IFAWarningCollector: """A collector for IFAWarningEmitters that can triage new feature entries.""" + #: The emitter that serves each type of warning. + _EMITTERS: dict[IFAWarningEmitterType, type[IFAWarningEmitter]] = { + IFAWarningEmitterType.NEAR_UNIQUE_DEPENDENT_FEATURES: NearUniqueDependentFeaturesWarningEmitter, + IFAWarningEmitterType.MISSING_TZ_FEATURES: MissingTZFeaturesWarningEmitter, + IFAWarningEmitterType.UNKNOWN_DATETIME_FORMAT: UnknownDatetimeFormatWarningEmitter, + IFAWarningEmitterType.UTC_OFFSET: UTCOffsetFeaturesWarningEmitter, + IFAWarningEmitterType.VALUE_COUNTS_PROCESSING: ValueCountsProcessing, + IFAWarningEmitterType.EXCESSIVE_FLOAT_PRECISION: ExcessiveFloatPrecisionWarningEmitter, + IFAWarningEmitterType.POSSIBLE_EXCESSIVE_FLOAT_PRECISION: PossibleExcessiveFloatPrecisionWarningEmitter, + IFAWarningEmitterType.SIMPLE: SimpleWarningEmitter, + } + def __init__(self, emitters: dict[str, IFAWarningEmitter] | None = None) -> None: self._emitters = emitters or {} @@ -120,28 +159,18 @@ def triage(self, emitter_type: IFAWarningEmitterType, feature_name: str) -> None The type of Warning Emitter this feature should be sorted to. feature_name : str The name of the feature applicable to the warning. + + Raises + ------ + ValueError + If `emitter_type` is not a known type of warning. """ - if emitter_type == IFAWarningEmitterType.NEAR_UNIQUE_DEPENDENT_FEATURES: - key = IFAWarningEmitterType.NEAR_UNIQUE_DEPENDENT_FEATURES.value - emitter = NearUniqueDependentFeaturesWarningEmitter - elif emitter_type == IFAWarningEmitterType.MISSING_TZ_FEATURES: - key = IFAWarningEmitterType.MISSING_TZ_FEATURES.value - emitter = MissingTZFeaturesWarningEmitter - elif emitter_type == IFAWarningEmitterType.UNKNOWN_DATETIME_FORMAT: - key = IFAWarningEmitterType.UNKNOWN_DATETIME_FORMAT.value - emitter = UnknownDatetimeFormatWarningEmitter - elif emitter_type == IFAWarningEmitterType.UTC_OFFSET: - key = IFAWarningEmitterType.UTC_OFFSET.value - emitter = UTCOffsetFeaturesWarningEmitter - elif emitter_type == IFAWarningEmitterType.SIMPLE: - key = IFAWarningEmitterType.SIMPLE.value - emitter = SimpleWarningEmitter - elif emitter_type == IFAWarningEmitterType.VALUE_COUNTS_PROCESSING: - key = IFAWarningEmitterType.VALUE_COUNTS_PROCESSING.value - emitter = ValueCountsProcessing - else: - raise ValueError("Unknown `emitter_type` provided.") + try: + emitter = self._EMITTERS[emitter_type] + except KeyError: + raise ValueError("Unknown `emitter_type` provided.") from None + key = emitter_type.value if key not in self._emitters: self._emitters[key] = emitter(features={feature_name}) else: From f0219d5f951841bdfb0d70ef855610ef0c37922b Mon Sep 17 00:00:00 2001 From: apbassett <43486400+apbassett@users.noreply.github.com> Date: Thu, 17 Sep 2026 17:54:42 -0400 Subject: [PATCH 2/5] Update docstrings/comments --- howso/utilities/feature_attributes/base.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/howso/utilities/feature_attributes/base.py b/howso/utilities/feature_attributes/base.py index ba7ac5f0..90ea5ae1 100644 --- a/howso/utilities/feature_attributes/base.py +++ b/howso/utilities/feature_attributes/base.py @@ -431,9 +431,9 @@ def _validate_dtype(self, data: pd.DataFrame, feature: str, coerced_df[feature] = series is_valid = True except Exception as err: # noqa: Intentionally broad - # Integer, nullable integer, and float columns all carry the same values to the - # engine, so a numeric feature is described faithfully by any numeric dtype, even - # one that cannot be cast losslessly. Such a column keeps its original dtype. + # Numeric dtypes differ only in representation here: validation does not alter the + # data unless `coerce` is set, so a numeric column is trained as it stands whichever + # dtype the attributes imply. A column that cannot be cast keeps its own dtype. is_valid = self._is_numeric_dtype(expected_dtype) and self._is_numeric_dtype(actual_dtype) coerce_err = str(err) @@ -2037,7 +2037,7 @@ def _process_rare_values( # noqa: PLR0912, PLR0915 max_distilled_cases, _ = get_optimized_max_chunk_size(row_count=self._get_row_count(), max_chunk_size=max_distilled_cases) else: - # Set a small default + # Set a small default; keep consistent with Enterprise max_distilled_cases = 50_000 # Workflow 1: User provided a config with protected multipliers; may need to compute unprotected multipliers From 7d6f41d5c9ae0a955a1813046b9fd1ccbf2577f8 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" Date: Mon, 21 Sep 2026 14:29:24 +0000 Subject: [PATCH 3/5] Trigger build after approval From 29f16821b7dcb178f164ae577f9ca5264b864f08 Mon Sep 17 00:00:00 2001 From: apbassett <43486400+apbassett@users.noreply.github.com> Date: Mon, 21 Sep 2026 11:24:20 -0400 Subject: [PATCH 4/5] Cleanup, add missing stacklevel args --- howso/utilities/feature_attributes/pandas.py | 6 +-- .../feature_attributes/suggestions.py | 6 ++- .../utilities/feature_attributes/warnings.py | 43 ++++++++++++++++--- 3 files changed, 43 insertions(+), 12 deletions(-) diff --git a/howso/utilities/feature_attributes/pandas.py b/howso/utilities/feature_attributes/pandas.py index 4a608002..ddf76159 100644 --- a/howso/utilities/feature_attributes/pandas.py +++ b/howso/utilities/feature_attributes/pandas.py @@ -786,11 +786,11 @@ def _infer_floating_point_attributes(self, feature_name: str) -> dict: # Specify decimal places for features the engine can represent exactly. Features beyond # that precision are trained without the attribute and reported in a single warning. if pd.api.types.is_float_dtype(col.dtype): - itemsize = getattr(col.dtype, 'itemsize', None) - if itemsize is None: + item_size = getattr(col.dtype, 'itemsize', None) + if item_size is None: self.warnings_collector.triage( IFAWarningEmitterType.POSSIBLE_EXCESSIVE_FLOAT_PRECISION, feature_name) - elif itemsize <= 8: + elif item_size <= 8: attributes['decimal_places'] = decimals else: self.warnings_collector.triage( diff --git a/howso/utilities/feature_attributes/suggestions.py b/howso/utilities/feature_attributes/suggestions.py index fc03e726..2003315d 100644 --- a/howso/utilities/feature_attributes/suggestions.py +++ b/howso/utilities/feature_attributes/suggestions.py @@ -342,7 +342,7 @@ def summary(self) -> str: return (f"Found {_count(num_values, 'rare value')} across {_count(len(self._prvc), 'column')} " "whose signal may be lost during data distillation workflows") - def _warn_default_max_distilled_cases(self, addendum: str = "") -> None: + def _warn_default_max_distilled_cases(self, addendum: str = "", stack_level: int = 4) -> None: """ Warn that the case weight multipliers were computed from a default ``max_distilled_cases``. @@ -351,6 +351,8 @@ def _warn_default_max_distilled_cases(self, addendum: str = "") -> None: addendum : str, default "" An additional sentence appended to the warning, describing the consequence for the calling method. + stack_level : int, default 4 + The stack level value to pass into `warn` via `stacklevel`. """ warnings.warn( "The computed case weights for rare value multipliers are likely inaccurate as " @@ -362,7 +364,7 @@ def _warn_default_max_distilled_cases(self, addendum: str = "") -> None: "remaining after distillation." + addendum, UserWarning, # Point past this helper at the caller of the public method that invoked it. - stacklevel=4, + stacklevel=stack_level, ) def apply(self, attributes: Mapping[str, Any]) -> None: diff --git a/howso/utilities/feature_attributes/warnings.py b/howso/utilities/feature_attributes/warnings.py index 8d217f2c..c4985c59 100644 --- a/howso/utilities/feature_attributes/warnings.py +++ b/howso/utilities/feature_attributes/warnings.py @@ -1,7 +1,33 @@ from abc import ABC from enum import Enum +import inspect +from pathlib import Path import warnings +#: The root of the `howso` package, used to find the frame a warning should be attributed to. +_PACKAGE_ROOT = str(Path(__file__).resolve().parents[2]) + + +def _user_stacklevel() -> int: + """ + Return the `stacklevel` of the nearest frame outside of the `howso` package. + + Emitters run several frames below the public entry point, and that depth differs between + inferrers, so the frame to attribute a warning to is found by walking out of the package + rather than by counting. Falls back to the outermost frame available. + """ + frame = inspect.currentframe() + if frame is None or frame.f_back is None: + # Frame introspection is unavailable on this interpreter + return 1 + # Level 1 is the caller of this helper, i.e. the frame that emits the warning + frame = frame.f_back + level = 1 + while frame.f_back is not None and frame.f_code.co_filename.startswith(_PACKAGE_ROOT): + frame = frame.f_back + level += 1 + return level + class IFAWarningEmitterType(Enum): """IFAWarningEmitter enum.""" @@ -51,7 +77,7 @@ def emit(self): warnings.warn("The following provided `dependent_features` have a large share of values that are unique: " f"{self.features_list}" "Dependent features with many unique values can severely impact the quality of results.", - UserWarning) + UserWarning, stacklevel=_user_stacklevel()) class MissingTZFeaturesWarningEmitter(IFAWarningEmitter): @@ -62,7 +88,7 @@ def emit(self): warnings.warn("The provided or inferred `date_time_formats` for the following " f"features do not include a time zone and will default to UTC: {self.features_list}" "\nTo change the default time zone, please specify the `default_time_zone` " - "argument to `infer_feature_attributes`.", UserWarning) + "argument to `infer_feature_attributes`.", UserWarning, stacklevel=_user_stacklevel()) class UnknownDatetimeFormatWarningEmitter(IFAWarningEmitter): @@ -73,7 +99,7 @@ def emit(self): warnings.warn("The following features were detected as possible datetimes, but we cannot assume " "their formats. Please provide them using `datetime_feature_formats` if desired. " f"Otherwise, these features will be treated as nominal strings: {self.features_list}", - UserWarning) + UserWarning, stacklevel=_user_stacklevel()) class UTCOffsetFeaturesWarningEmitter(IFAWarningEmitter): @@ -84,7 +110,7 @@ def emit(self): warnings.warn(f"The following features are using UTC offsets (%z) for their time zones: {self.features_list}" "\nThis could lead to unexpected results due to daylight savings time. We recommend " "using explicit time zone strings, e.g., \"GMT\", which are represented by the \"%Z\" " - "identifier.", UserWarning) + "identifier.", UserWarning, stacklevel=_user_stacklevel()) class ValueCountsProcessing(IFAWarningEmitter): @@ -94,7 +120,8 @@ def emit(self): """Emit the warning.""" warnings.warn("Could not process some value counts for the following features, likely due to the presence of " f"unhashable values: {self.features_list}\nThis may affect the accuracy and completeness of " - "suggested or computed `preserve_rare_values` configurations`.", UserWarning) + "suggested or computed `preserve_rare_values` configurations`.", UserWarning, + stacklevel=_user_stacklevel()) class FloatPrecisionWarningEmitter(IFAWarningEmitter): @@ -107,7 +134,8 @@ def emit(self): """Emit the warning.""" warnings.warn(f"The following features {self._certainty} floating point values that exceed the " f"maximum supported precision of 64 bits: {self.features_list}" - "\nThese features are trained without a `decimal_places` attribute.", UserWarning) + "\nThese features are trained without a `decimal_places` attribute.", UserWarning, + stacklevel=_user_stacklevel()) class ExcessiveFloatPrecisionWarningEmitter(FloatPrecisionWarningEmitter): @@ -127,8 +155,9 @@ class SimpleWarningEmitter(IFAWarningEmitter): def emit(self): """Emit the warning.""" + stacklevel = _user_stacklevel() for msg in self.features: - warnings.warn(msg, UserWarning) + warnings.warn(msg, UserWarning, stacklevel=stacklevel) class IFAWarningCollector: From 59a99541c89d968f82f10f7fde0809731bf2c4bf Mon Sep 17 00:00:00 2001 From: apbassett <43486400+apbassett@users.noreply.github.com> Date: Mon, 21 Sep 2026 12:01:20 -0400 Subject: [PATCH 5/5] Fix stacklevel --- howso/utilities/feature_attributes/suggestions.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/howso/utilities/feature_attributes/suggestions.py b/howso/utilities/feature_attributes/suggestions.py index 2003315d..7ffcab10 100644 --- a/howso/utilities/feature_attributes/suggestions.py +++ b/howso/utilities/feature_attributes/suggestions.py @@ -352,7 +352,8 @@ def _warn_default_max_distilled_cases(self, addendum: str = "", stack_level: int An additional sentence appended to the warning, describing the consequence for the calling method. stack_level : int, default 4 - The stack level value to pass into `warn` via `stacklevel`. + The stack level value to pass into `warn` via `stacklevel`. The default attributes the + warning to the caller of `apply_suggestion()`; methods a user calls directly pass 3. """ warnings.warn( "The computed case weights for rare value multipliers are likely inaccurate as " @@ -363,7 +364,6 @@ def _warn_default_max_distilled_cases(self, addendum: str = "", stack_level: int "values in the data, since the weighting is calibrated proportionally to the number of cases " "remaining after distillation." + addendum, UserWarning, - # Point past this helper at the caller of the public method that invoked it. stacklevel=stack_level, ) @@ -381,13 +381,13 @@ def apply(self, attributes: Mapping[str, Any]) -> None: def get_config(self, enable_warnings: bool = True) -> FullPreserveRareValuesConfig: """Get the `preserve_rare_values_config` for use in future calls to `infer_feature_attributes`.""" if not self._user_set_mdc and enable_warnings: - self._warn_default_max_distilled_cases() + self._warn_default_max_distilled_cases(stack_level=3) return self._prvc def get_values_map(self) -> PreserveRareValuesMap: """Get the `preserve_rare_values_map` for use in future calls to `infer_feature_attributes.""" if not self._user_set_mdc: - self._warn_default_max_distilled_cases() + self._warn_default_max_distilled_cases(stack_level=3) values_map = {} for feature, config in self._prvc.items(): multipliers = config["protected_values_multipliers"]