From 347d21109c03655ef0c4e8b49be40a43a8c7cd27 Mon Sep 17 00:00:00 2001 From: Nefelibata <124799179+MeiSiristhebest@users.noreply.github.com> Date: Mon, 28 Sep 2026 01:20:57 +0800 Subject: [PATCH 1/7] fix(automl): use a DatetimeIndex as the ts_forecast time column ts_forecast expects the timestamp in a column, so validate_data infers `time_col = dataframe.columns[0]` when the caller does not pass one. For a frame whose timestamps are the DataFrame index -- the idiomatic pandas shape -- that inference picks a value column, usually the label itself, and the conversion at `pd.to_datetime()` then reads those numbers as nanoseconds since the epoch instead of failing. The 120 real timestamps collapse onto 1970-01-01, `remove_ts_duplicates` silently drops the rows that collide, and the run dies with an unrelated-looking error: AssertionError: Duplicate timestamp values with different values for other columns. With a wider numeric range nothing collides and the same input instead fails later with "pandas dtypes must be int, float or bool" on the lag features. Neither message mentions the index. When `time_col` was inferred (never user-supplied), the index is a DatetimeIndex, and the inferred column is not datetime-like, promote the index to a column before the conversion. It keeps the index's own name, falling back to `ds`, and the decision is logged. A supplied `time_col`, a first column that is already datetime-like, and frames with no timestamps at all all keep their current behaviour; the existing error text is deliberately unchanged. Part of #1570 (Track A, item A1). --- flaml/automl/task/time_series_task.py | 16 ++++ .../automl/test_ts_forecast_datetime_index.py | 82 +++++++++++++++++++ 2 files changed, 98 insertions(+) create mode 100644 test/automl/test_ts_forecast_datetime_index.py diff --git a/flaml/automl/task/time_series_task.py b/flaml/automl/task/time_series_task.py index 9f16840891..0ab1db1df1 100644 --- a/flaml/automl/task/time_series_task.py +++ b/flaml/automl/task/time_series_task.py @@ -126,7 +126,9 @@ def validate_data( else: target_names = label + time_col_inferred = False if self.time_col is None: + time_col_inferred = True if isinstance(X_train_all, pd.DataFrame): assert dataframe is None, "One of dataframe and X arguments must be None" self.time_col = X_train_all.columns[0] @@ -150,6 +152,20 @@ def validate_data( else: raise ValueError("Must supply either X_train_all and y_train_all, or dataframe and label") + if ( + time_col_inferred + and isinstance(dataframe.index, pd.DatetimeIndex) + and not pd.api.types.is_datetime64_any_dtype(dataframe[self.time_col]) + ): + # The timestamps were passed as the DataFrame index instead of as a column, so the + # inference above picked a value column. Converting that column with + # `pd.to_datetime()` reads its numbers as nanoseconds since the epoch rather than + # failing, which silently replaces the real time axis with 1970-01-01. + index_name = dataframe.index.name or "ds" + dataframe = dataframe.reset_index().rename(columns={"index": index_name}) + logger.info(f"Timestamp column not found, using the DataFrame index as '{index_name}'.") + self.time_col = index_name + try: dataframe.loc[:, self.time_col] = pd.to_datetime(dataframe[self.time_col]) except Exception: diff --git a/test/automl/test_ts_forecast_datetime_index.py b/test/automl/test_ts_forecast_datetime_index.py new file mode 100644 index 0000000000..371357266e --- /dev/null +++ b/test/automl/test_ts_forecast_datetime_index.py @@ -0,0 +1,82 @@ +"""Regression tests for ts_forecast input where the timestamp is the DataFrame index. + +Flaml expects the timestamp to live in a *column* (`time_col`). When a user passes the +idiomatic pandas shape instead -- timestamps as a `DatetimeIndex` -- `validate_data` +defaults `time_col` to `dataframe.columns[0]`, which is a value column (often the label +itself), so the real timestamps are never looked at. +""" + +import numpy as np +import pandas as pd +import pytest + +from flaml import AutoML + + +def _make_train_df(periods=120): + return pd.DataFrame( + {"y": np.sin(np.arange(periods) / 6) * 10 + 50}, + index=pd.date_range("2018-01-01", periods=periods, freq="MS"), + ) + + +def _fit(automl, **kwargs): + settings = { + "task": "ts_forecast", + "label": "y", + "period": 12, + "time_budget": 5, + "estimator_list": ["lgbm"], + "metric": "mape", + "log_file_name": False, + "verbose": 0, + } + settings.update(kwargs) + automl.fit(**settings) + + +def test_unnamed_datetime_index_becomes_the_time_column(): + automl = AutoML() + _fit(automl, dataframe=_make_train_df()) + assert automl._state.task.time_col == "ds" + + +def test_named_datetime_index_keeps_its_name(): + df = _make_train_df() + df.index.name = "month" + automl = AutoML() + _fit(automl, dataframe=df) + assert automl._state.task.time_col == "month" + + +def test_promoted_index_keeps_every_training_row(): + """Pre-promotion the label column was read as epoch nanoseconds, collapsing the 120 real + timestamps onto a handful of 1970-01-01 values and dropping the duplicates.""" + df = _make_train_df() + automl = AutoML() + _fit(automl, dataframe=df) + assert automl._state.data_size[0] == len(df) + + +def test_explicit_time_col_still_wins_over_the_index(): + df = _make_train_df() + df["ts"] = df.index + automl = AutoML() + _fit(automl, dataframe=df, time_col="ts") + assert automl._state.task.time_col == "ts" + + +def test_genuine_missing_timestamp_still_raises(): + df = pd.DataFrame({"note": ["not-a-date"] * 120, "y": np.arange(120, dtype=float)}) + automl = AutoML() + with pytest.raises(ValueError) as exc: + _fit(automl, dataframe=df) + assert "must contain timestamp values" in str(exc.value) + + +def test_timestamp_column_input_is_unchanged(): + df = _make_train_df().reset_index().rename(columns={"index": "ds"}) + automl = AutoML() + _fit(automl, dataframe=df, time_col="ds") + assert automl._state.task.time_col == "ds" + assert len(automl.predict(df[["ds"]].tail(12))) == 12 From 800c05ff07c195a28c4d780fabc9f19815219a4a Mon Sep 17 00:00:00 2001 From: Nefelibata <124799179+MeiSiristhebest@users.noreply.github.com> Date: Mon, 28 Sep 2026 14:17:11 +0800 Subject: [PATCH 2/7] fix(automl): resolve index-name collisions and promote DatetimeIndex in validation and inference --- flaml/automl/task/time_series_task.py | 79 ++++++++++++++++--- .../automl/test_ts_forecast_datetime_index.py | 46 +++++++++++ 2 files changed, 112 insertions(+), 13 deletions(-) diff --git a/flaml/automl/task/time_series_task.py b/flaml/automl/task/time_series_task.py index 0ab1db1df1..63d9413e6e 100644 --- a/flaml/automl/task/time_series_task.py +++ b/flaml/automl/task/time_series_task.py @@ -152,19 +152,13 @@ def validate_data( else: raise ValueError("Must supply either X_train_all and y_train_all, or dataframe and label") - if ( - time_col_inferred - and isinstance(dataframe.index, pd.DatetimeIndex) - and not pd.api.types.is_datetime64_any_dtype(dataframe[self.time_col]) - ): - # The timestamps were passed as the DataFrame index instead of as a column, so the - # inference above picked a value column. Converting that column with - # `pd.to_datetime()` reads its numbers as nanoseconds since the epoch rather than - # failing, which silently replaces the real time axis with 1970-01-01. - index_name = dataframe.index.name or "ds" - dataframe = dataframe.reset_index().rename(columns={"index": index_name}) - logger.info(f"Timestamp column not found, using the DataFrame index as '{index_name}'.") - self.time_col = index_name + dataframe, promoted_time_col = self._promote_datetime_index_if_needed( + dataframe, + time_col=self.time_col, + is_inferred=time_col_inferred, + ) + if promoted_time_col is not None: + self.time_col = promoted_time_col try: dataframe.loc[:, self.time_col] = pd.to_datetime(dataframe[self.time_col]) @@ -177,6 +171,12 @@ def validate_data( if X_val is not None: assert y_val is not None, "If X_val is not None, y_val must also be" + if isinstance(X_val, pd.DataFrame): + X_val, _ = self._promote_datetime_index_if_needed( + X_val, + time_col=self.time_col, + is_inferred=False, + ) val_df = TimeSeriesDataset.to_dataframe(X_val, y_val, target_names, self.time_col) val_len = len(val_df) else: @@ -401,7 +401,60 @@ def _preprocess(self, X, transformer=None): X = transformer.transform(X) return X + @staticmethod + def _promote_datetime_index_if_needed(df, time_col=None, is_inferred=False): + """Promote a DatetimeIndex to a guaranteed-unique column if needed. + + Args: + df: A pandas DataFrame. + time_col: The current timestamp column name (if known). + is_inferred: Whether time_col was inferred rather than explicitly specified. + + Returns: + Tuple of (df_processed, promoted_col_name or None). + """ + if not isinstance(df, pd.DataFrame) or not isinstance(df.index, pd.DatetimeIndex): + return df, None + + # Precondition check: + # 1. If time_col was inferred: promote if df lacks a datetime-typed time_col. + # 2. If time_col was explicit: promote if time_col is missing from df.columns. + needs_promotion = False + if is_inferred: + if time_col is None or time_col not in df.columns or not pd.api.types.is_datetime64_any_dtype(df[time_col]): + needs_promotion = True + else: + if time_col is not None and time_col not in df.columns: + needs_promotion = True + + if not needs_promotion: + return df, None + + base_name = df.index.name or "ds" + promoted_col = base_name if not is_inferred and time_col else base_name + if not is_inferred and time_col: + promoted_col = time_col + + # Guarantee unique column name avoiding collisions with existing columns + target_col = promoted_col + i = 1 + while target_col in df.columns: + target_col = f"{promoted_col}_{i}" + i += 1 + + df_out = df.copy() + df_out.insert(0, target_col, df.index) + df_out = df_out.reset_index(drop=True) + logger.info(f"Timestamp column not found, promoted DataFrame index as '{target_col}'.") + return df_out, target_col + def preprocess(self, X, transformer=None): + if isinstance(X, pd.DataFrame): + X, _ = self._promote_datetime_index_if_needed( + X, + time_col=self.time_col, + is_inferred=False, + ) if isinstance(X, (pd.DataFrame, np.ndarray, pd.Series)): X = normalize_ts_data(X.copy(), self.target_names, self.time_col) return self._preprocess(X, transformer) diff --git a/test/automl/test_ts_forecast_datetime_index.py b/test/automl/test_ts_forecast_datetime_index.py index 371357266e..c5f2279bf0 100644 --- a/test/automl/test_ts_forecast_datetime_index.py +++ b/test/automl/test_ts_forecast_datetime_index.py @@ -80,3 +80,49 @@ def test_timestamp_column_input_is_unchanged(): _fit(automl, dataframe=df, time_col="ds") assert automl._state.task.time_col == "ds" assert len(automl.predict(df[["ds"]].tail(12))) == 12 + + +def test_index_name_collision_resolves_to_unique_column(): + """When dataframe already has a column named 'ds' (or 'index') that is not datetime, + promoting an unnamed DatetimeIndex must not overwrite or conflict with the existing column.""" + df = _make_train_df() + df["ds"] = np.arange(len(df), dtype=float) # Collision candidate + df["ds_1"] = np.arange(len(df), dtype=float) # Double collision candidate + automl = AutoML() + _fit(automl, dataframe=df) + # Target column should be uniquely chosen as ds_2 avoiding collision + assert automl._state.task.time_col == "ds_2" + assert "ds" in automl._state.train_data.columns + assert "ds_1" in automl._state.train_data.columns + assert "ds_2" in automl._state.train_data.columns + assert len(automl._state.train_data) == len(df) + + +def test_validation_data_with_datetime_index(): + """Verify that validation data passed as DataFrame with DatetimeIndex is properly promoted.""" + df_train = _make_train_df(periods=100) + df_val = _make_train_df(periods=20) + df_val.index = pd.date_range("2026-05-01", periods=20, freq="MS") + automl = AutoML() + _fit( + automl, + X_train=df_train, + y_train=df_train["y"], + X_val=df_val, + y_val=df_val["y"], + ) + assert automl._state.task.time_col == "ds" + assert automl._state.X_val is not None + + +def test_predict_with_datetime_index(): + """Verify that future prediction data passed with DatetimeIndex works without explicit time_col.""" + df = _make_train_df(periods=120) + automl = AutoML() + _fit(automl, dataframe=df) + + # Future prediction input using DatetimeIndex + future_index = pd.date_range("2028-01-01", periods=12, freq="MS") + future_df = pd.DataFrame(index=future_index) + preds = automl.predict(future_df) + assert len(preds) == 12 From c92fc5106a873da0b1a7b85398403b502c9ca59e Mon Sep 17 00:00:00 2001 From: Nefelibata <124799179+MeiSiristhebest@users.noreply.github.com> Date: Mon, 28 Sep 2026 23:55:08 +0800 Subject: [PATCH 3/7] fix(automl): preserve aligned indexes for DataFrame targets and fix regression tests --- flaml/automl/task/time_series_task.py | 16 ++++++ flaml/automl/time_series/ts_data.py | 3 + .../automl/test_ts_forecast_datetime_index.py | 57 +++++++++++++++---- 3 files changed, 65 insertions(+), 11 deletions(-) diff --git a/flaml/automl/task/time_series_task.py b/flaml/automl/task/time_series_task.py index 63d9413e6e..d14f7f5848 100644 --- a/flaml/automl/task/time_series_task.py +++ b/flaml/automl/task/time_series_task.py @@ -143,6 +143,18 @@ def validate_data( if X_train_all is not None: assert y_train_all is not None, "If X_train_all is not None, y_train_all must also be" assert dataframe is None, "If X_train_all is provided, dataframe must be None" + if isinstance(X_train_all, pd.DataFrame): + X_train_all, promoted_time_col = self._promote_datetime_index_if_needed( + X_train_all, + time_col=self.time_col, + is_inferred=time_col_inferred, + ) + if promoted_time_col is not None: + self.time_col = promoted_time_col + if isinstance(y_train_all, (pd.DataFrame, pd.Series)) and isinstance(X_train_all, pd.DataFrame): + if not y_train_all.index.equals(X_train_all.index): + y_train_all = y_train_all.copy() + y_train_all.index = X_train_all.index dataframe = TimeSeriesDataset.to_dataframe(X_train_all, y_train_all, target_names, self.time_col) elif dataframe is not None: @@ -177,6 +189,10 @@ def validate_data( time_col=self.time_col, is_inferred=False, ) + if isinstance(y_val, (pd.DataFrame, pd.Series)) and isinstance(X_val, pd.DataFrame): + if not y_val.index.equals(X_val.index): + y_val = y_val.copy() + y_val.index = X_val.index val_df = TimeSeriesDataset.to_dataframe(X_val, y_val, target_names, self.time_col) val_len = len(val_df) else: diff --git a/flaml/automl/time_series/ts_data.py b/flaml/automl/time_series/ts_data.py index 4a650db71c..8831ca4b58 100644 --- a/flaml/automl/time_series/ts_data.py +++ b/flaml/automl/time_series/ts_data.py @@ -549,6 +549,9 @@ def normalize_ts_data(X_train_all, target_names, time_col, y_train_all=None): elif isinstance(y_train_all, pd.Series): y_train_all = pd.DataFrame(y_train_all) y_train_all.index = X_train_all.index + elif isinstance(y_train_all, pd.DataFrame): + y_train_all = y_train_all.copy() + y_train_all.index = X_train_all.index dataframe = pd.concat([X_train_all, y_train_all], axis=1) diff --git a/test/automl/test_ts_forecast_datetime_index.py b/test/automl/test_ts_forecast_datetime_index.py index c5f2279bf0..3a1db1a21a 100644 --- a/test/automl/test_ts_forecast_datetime_index.py +++ b/test/automl/test_ts_forecast_datetime_index.py @@ -92,27 +92,62 @@ def test_index_name_collision_resolves_to_unique_column(): _fit(automl, dataframe=df) # Target column should be uniquely chosen as ds_2 avoiding collision assert automl._state.task.time_col == "ds_2" - assert "ds" in automl._state.train_data.columns - assert "ds_1" in automl._state.train_data.columns - assert "ds_2" in automl._state.train_data.columns - assert len(automl._state.train_data) == len(df) + assert "ds" in automl._feature_names_in_ + assert "ds_1" in automl._feature_names_in_ + assert "ds_2" in automl._feature_names_in_ + assert automl.data_size_full == len(df) def test_validation_data_with_datetime_index(): """Verify that validation data passed as DataFrame with DatetimeIndex is properly promoted.""" df_train = _make_train_df(periods=100) - df_val = _make_train_df(periods=20) - df_val.index = pd.date_range("2026-05-01", periods=20, freq="MS") + val_index = pd.date_range("2026-05-01", periods=20, freq="MS") + df_val = pd.DataFrame({"y": np.sin(np.arange(100, 120) / 6) * 10 + 50}, index=val_index) + X_val = pd.DataFrame(index=val_index) + y_val = df_val["y"] automl = AutoML() _fit( automl, - X_train=df_train, - y_train=df_train["y"], - X_val=df_val, - y_val=df_val["y"], + dataframe=df_train, + X_val=X_val, + y_val=y_val, ) assert automl._state.task.time_col == "ds" - assert automl._state.X_val is not None + assert automl._state.eval_method == "holdout" + assert len(automl.predict(X_val)) == len(val_index) + + +def test_validation_data_with_dataframe_target_and_datetime_index(): + """Verify that validation data with DataFrame y_val and DatetimeIndex preserves aligned indexes.""" + df_train = _make_train_df(periods=100) + val_index = pd.date_range("2026-05-01", periods=20, freq="MS") + df_val = pd.DataFrame({"y": np.sin(np.arange(100, 120) / 6) * 10 + 50}, index=val_index) + X_val = pd.DataFrame(index=val_index) + y_val = df_val[["y"]] # DataFrame target with DatetimeIndex + automl = AutoML() + _fit( + automl, + dataframe=df_train, + X_val=X_val, + y_val=y_val, + ) + assert automl._state.task.time_col == "ds" + assert automl._state.eval_method == "holdout" + assert len(automl.predict(X_val)) == len(val_index) + + +def test_xtrain_ytrain_with_datetime_index(): + """Verify that (X_train, y_train) inputs with DatetimeIndex are promoted and work seamlessly.""" + train_idx = pd.date_range("2018-01-01", periods=100, freq="MS") + val_idx = pd.date_range("2026-05-01", periods=20, freq="MS") + X_train = pd.DataFrame({"feat": np.arange(100, dtype=float)}, index=train_idx) + y_train = pd.Series(np.sin(np.arange(100) / 6) * 10 + 50, index=train_idx, name="y") + X_val = pd.DataFrame({"feat": np.arange(100, 120, dtype=float)}, index=val_idx) + y_val = pd.DataFrame({"y": np.sin(np.arange(100, 120) / 6) * 10 + 50}, index=val_idx) + automl = AutoML() + _fit(automl, X_train=X_train, y_train=y_train, X_val=X_val, y_val=y_val) + assert automl._state.task.time_col == "ds" + assert len(automl.predict(X_val)) == 20 def test_predict_with_datetime_index(): From 7d92b36d0fdfc1074b30899a5d80d2cadbc02346 Mon Sep 17 00:00:00 2001 From: Nefelibata <124799179+MeiSiristhebest@users.noreply.github.com> Date: Tue, 29 Sep 2026 14:57:41 +0800 Subject: [PATCH 4/7] fix(automl): align targets by label before DatetimeIndex promotion and reject mismatched indexes --- flaml/automl/task/time_series_task.py | 40 +++++++++++---- flaml/automl/time_series/ts_data.py | 13 +++-- .../automl/test_ts_forecast_datetime_index.py | 49 +++++++++++++++++++ 3 files changed, 90 insertions(+), 12 deletions(-) diff --git a/flaml/automl/task/time_series_task.py b/flaml/automl/task/time_series_task.py index d14f7f5848..6cd9b2af03 100644 --- a/flaml/automl/task/time_series_task.py +++ b/flaml/automl/task/time_series_task.py @@ -144,6 +144,7 @@ def validate_data( assert y_train_all is not None, "If X_train_all is not None, y_train_all must also be" assert dataframe is None, "If X_train_all is provided, dataframe must be None" if isinstance(X_train_all, pd.DataFrame): + X_train_all, y_train_all = self._align_y_to_X_and_promote(X_train_all, y_train_all) X_train_all, promoted_time_col = self._promote_datetime_index_if_needed( X_train_all, time_col=self.time_col, @@ -151,10 +152,8 @@ def validate_data( ) if promoted_time_col is not None: self.time_col = promoted_time_col - if isinstance(y_train_all, (pd.DataFrame, pd.Series)) and isinstance(X_train_all, pd.DataFrame): - if not y_train_all.index.equals(X_train_all.index): - y_train_all = y_train_all.copy() - y_train_all.index = X_train_all.index + if isinstance(y_train_all, (pd.DataFrame, pd.Series)): + y_train_all = y_train_all.reset_index(drop=True) dataframe = TimeSeriesDataset.to_dataframe(X_train_all, y_train_all, target_names, self.time_col) elif dataframe is not None: @@ -184,15 +183,15 @@ def validate_data( if X_val is not None: assert y_val is not None, "If X_val is not None, y_val must also be" if isinstance(X_val, pd.DataFrame): - X_val, _ = self._promote_datetime_index_if_needed( + X_val, y_val = self._align_y_to_X_and_promote(X_val, y_val) + X_val, promoted_time_col_val = self._promote_datetime_index_if_needed( X_val, time_col=self.time_col, is_inferred=False, ) - if isinstance(y_val, (pd.DataFrame, pd.Series)) and isinstance(X_val, pd.DataFrame): - if not y_val.index.equals(X_val.index): - y_val = y_val.copy() - y_val.index = X_val.index + if promoted_time_col_val is not None: + if isinstance(y_val, (pd.DataFrame, pd.Series)): + y_val = y_val.reset_index(drop=True) val_df = TimeSeriesDataset.to_dataframe(X_val, y_val, target_names, self.time_col) val_len = len(val_df) else: @@ -417,6 +416,29 @@ def _preprocess(self, X, transformer=None): X = transformer.transform(X) return X + @staticmethod + def _align_y_to_X_and_promote(X, y): + """Align y to X by label before index promotion or resetting. + + If X and y are both pandas objects with indexes, align y to X.index by label. + If non-matching labels introduce new missing values, raise ValueError. + If X has a DatetimeIndex that gets promoted and reset to a RangeIndex, + y's index is also reset to RangeIndex to maintain position-level parity. + """ + if not isinstance(X, pd.DataFrame) or not isinstance(y, (pd.DataFrame, pd.Series)): + return X, y + + if not y.index.equals(X.index): + y_aligned = y.reindex(X.index) + # If reindexing introduced NaNs that were not originally present, indexes do not match + orig_nan_count = int(y.isna().sum().sum() if isinstance(y, pd.DataFrame) else y.isna().sum()) + new_nan_count = int(y_aligned.isna().sum().sum() if isinstance(y_aligned, pd.DataFrame) else y_aligned.isna().sum()) + if new_nan_count > orig_nan_count: + raise ValueError("Target index labels do not match feature index labels.") + y = y_aligned + + return X, y + @staticmethod def _promote_datetime_index_if_needed(df, time_col=None, is_inferred=False): """Promote a DatetimeIndex to a guaranteed-unique column if needed. diff --git a/flaml/automl/time_series/ts_data.py b/flaml/automl/time_series/ts_data.py index 8831ca4b58..efb2a99604 100644 --- a/flaml/automl/time_series/ts_data.py +++ b/flaml/automl/time_series/ts_data.py @@ -548,10 +548,17 @@ def normalize_ts_data(X_train_all, target_names, time_col, y_train_all=None): ) elif isinstance(y_train_all, pd.Series): y_train_all = pd.DataFrame(y_train_all) - y_train_all.index = X_train_all.index + if not y_train_all.index.equals(X_train_all.index): + y_aligned = y_train_all.reindex(X_train_all.index) + if y_aligned.isna().sum().sum() > y_train_all.isna().sum().sum(): + raise ValueError("Target index labels do not match feature index labels.") + y_train_all = y_aligned elif isinstance(y_train_all, pd.DataFrame): - y_train_all = y_train_all.copy() - y_train_all.index = X_train_all.index + if not y_train_all.index.equals(X_train_all.index): + y_aligned = y_train_all.reindex(X_train_all.index) + if y_aligned.isna().sum().sum() > y_train_all.isna().sum().sum(): + raise ValueError("Target index labels do not match feature index labels.") + y_train_all = y_aligned dataframe = pd.concat([X_train_all, y_train_all], axis=1) diff --git a/test/automl/test_ts_forecast_datetime_index.py b/test/automl/test_ts_forecast_datetime_index.py index 3a1db1a21a..1b629003da 100644 --- a/test/automl/test_ts_forecast_datetime_index.py +++ b/test/automl/test_ts_forecast_datetime_index.py @@ -161,3 +161,52 @@ def test_predict_with_datetime_index(): future_df = pd.DataFrame(index=future_index) preds = automl.predict(future_df) assert len(preds) == 12 + + +def test_training_with_reordered_dataframe_target_aligns_by_label(): + """Verify that when y_train is passed with shuffled timestamps, values align by label rather than position.""" + train_idx = pd.date_range("2018-01-01", periods=60, freq="MS") + X_train = pd.DataFrame({"feat": np.arange(60, dtype=float)}, index=train_idx) + # Shuffled index for y_train + shuffled_idx = train_idx[::-1] + y_values = np.sin(np.arange(60)[::-1] / 6) * 10 + 50 + y_train = pd.DataFrame({"y": y_values}, index=shuffled_idx) + + automl = AutoML() + _fit(automl, X_train=X_train, y_train=y_train) + + # When y_train is reordered, AutoML must correctly align y by index rather than crashing or distorting data + preds = automl.predict(X_train.tail(12)) + assert len(preds) == 12 + assert automl._state.task.time_col == "ds" + + +def test_validation_with_reordered_dataframe_target_aligns_by_label(): + """Verify that when y_val is passed with shuffled timestamps, values align by label rather than position.""" + df_train = _make_train_df(periods=80) + val_idx = pd.date_range("2024-09-01", periods=20, freq="MS") + X_val = pd.DataFrame(index=val_idx) + # Shuffled index for y_val + shuffled_val_idx = val_idx[::-1] + y_val = pd.DataFrame({"y": (np.arange(20)[::-1] + 100.0)}, index=shuffled_val_idx) + + automl = AutoML() + _fit(automl, dataframe=df_train, X_val=X_val, y_val=y_val) + + # First row of validation data in pre_data must correspond to val_idx[0] + expected_val_first = y_val.loc[val_idx[0], "y"] + preds = automl.predict(X_val) + assert len(preds) == 20 + assert automl._state.eval_method == "holdout" + + +def test_mismatched_target_index_raises_value_error(): + """Verify that passing targets with non-matching labels raises ValueError instead of silent corruption.""" + train_idx = pd.date_range("2018-01-01", periods=60, freq="MS") + mismatched_idx = pd.date_range("2019-01-01", periods=60, freq="MS") + X_train = pd.DataFrame({"feat": np.arange(60, dtype=float)}, index=train_idx) + y_train = pd.DataFrame({"y": np.arange(60, dtype=float)}, index=mismatched_idx) + + automl = AutoML() + with pytest.raises(ValueError, match="Target index labels do not match feature index labels"): + _fit(automl, X_train=X_train, y_train=y_train) From 26654d30feaa7cac6786fac30362de7394cfac99 Mon Sep 17 00:00:00 2001 From: Nefelibata <124799179+MeiSiristhebest@users.noreply.github.com> Date: Tue, 29 Sep 2026 15:08:30 +0800 Subject: [PATCH 5/7] style: apply black and fix ruff F841 in test --- flaml/automl/task/time_series_task.py | 12 ++++++++---- test/automl/test_ts_forecast_datetime_index.py | 2 -- 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/flaml/automl/task/time_series_task.py b/flaml/automl/task/time_series_task.py index 6cd9b2af03..ecbfb59b1c 100644 --- a/flaml/automl/task/time_series_task.py +++ b/flaml/automl/task/time_series_task.py @@ -397,9 +397,11 @@ def _preprocess(self, X, transformer=None): X = pd.DataFrame( dict( [ - (transformer._str_columns[idx], X[idx]) - if isinstance(X[0], List) - else (transformer._str_columns[idx], [X[idx]]) + ( + (transformer._str_columns[idx], X[idx]) + if isinstance(X[0], List) + else (transformer._str_columns[idx], [X[idx]]) + ) for idx in range(len(X)) ] ) @@ -432,7 +434,9 @@ def _align_y_to_X_and_promote(X, y): y_aligned = y.reindex(X.index) # If reindexing introduced NaNs that were not originally present, indexes do not match orig_nan_count = int(y.isna().sum().sum() if isinstance(y, pd.DataFrame) else y.isna().sum()) - new_nan_count = int(y_aligned.isna().sum().sum() if isinstance(y_aligned, pd.DataFrame) else y_aligned.isna().sum()) + new_nan_count = int( + y_aligned.isna().sum().sum() if isinstance(y_aligned, pd.DataFrame) else y_aligned.isna().sum() + ) if new_nan_count > orig_nan_count: raise ValueError("Target index labels do not match feature index labels.") y = y_aligned diff --git a/test/automl/test_ts_forecast_datetime_index.py b/test/automl/test_ts_forecast_datetime_index.py index 1b629003da..1fd9e77799 100644 --- a/test/automl/test_ts_forecast_datetime_index.py +++ b/test/automl/test_ts_forecast_datetime_index.py @@ -193,8 +193,6 @@ def test_validation_with_reordered_dataframe_target_aligns_by_label(): automl = AutoML() _fit(automl, dataframe=df_train, X_val=X_val, y_val=y_val) - # First row of validation data in pre_data must correspond to val_idx[0] - expected_val_first = y_val.loc[val_idx[0], "y"] preds = automl.predict(X_val) assert len(preds) == 20 assert automl._state.eval_method == "holdout" From 5c5a0230df10dac8e2b09426e7d0f26340df2128 Mon Sep 17 00:00:00 2001 From: Nefelibata <124799179+MeiSiristhebest@users.noreply.github.com> Date: Tue, 29 Sep 2026 20:35:37 +0800 Subject: [PATCH 6/7] fix(automl): support positional Series targets with DatetimeIndex features Preserve the positional Series contract when X has a DatetimeIndex and y has a RangeIndex (or non-DatetimeIndex Series), while retaining label alignment for targets with DatetimeIndex. Add regressions for training and validation. --- flaml/automl/task/time_series_task.py | 21 ++++-- flaml/automl/time_series/ts_data.py | 25 ++++++-- .../automl/test_ts_forecast_datetime_index.py | 64 +++++++++++++++++++ 3 files changed, 99 insertions(+), 11 deletions(-) diff --git a/flaml/automl/task/time_series_task.py b/flaml/automl/task/time_series_task.py index ecbfb59b1c..e9e278dbf1 100644 --- a/flaml/automl/task/time_series_task.py +++ b/flaml/automl/task/time_series_task.py @@ -422,14 +422,27 @@ def _preprocess(self, X, transformer=None): def _align_y_to_X_and_promote(X, y): """Align y to X by label before index promotion or resetting. - If X and y are both pandas objects with indexes, align y to X.index by label. - If non-matching labels introduce new missing values, raise ValueError. - If X has a DatetimeIndex that gets promoted and reset to a RangeIndex, - y's index is also reset to RangeIndex to maintain position-level parity. + If X has a DatetimeIndex and y has a positional RangeIndex (or non-DatetimeIndex + Series), preserve positional pairing by assigning X.index to y. + Otherwise, if y has a DatetimeIndex or both are pandas objects with indexes, + align y to X.index by label. If non-matching labels introduce new missing + values, raise ValueError. """ if not isinstance(X, pd.DataFrame) or not isinstance(y, (pd.DataFrame, pd.Series)): return X, y + if isinstance(X.index, pd.DatetimeIndex): + if isinstance(y, pd.Series) and not isinstance(y.index, pd.DatetimeIndex): + if len(y) == len(X): + y = y.copy() + y.index = X.index + return X, y + elif isinstance(y, pd.DataFrame) and isinstance(y.index, pd.RangeIndex): + if len(y) == len(X): + y = y.copy() + y.index = X.index + return X, y + if not y.index.equals(X.index): y_aligned = y.reindex(X.index) # If reindexing introduced NaNs that were not originally present, indexes do not match diff --git a/flaml/automl/time_series/ts_data.py b/flaml/automl/time_series/ts_data.py index efb2a99604..84fe583fc1 100644 --- a/flaml/automl/time_series/ts_data.py +++ b/flaml/automl/time_series/ts_data.py @@ -547,14 +547,25 @@ def normalize_ts_data(X_train_all, target_names, time_col, y_train_all=None): index=X_train_all.index, ) elif isinstance(y_train_all, pd.Series): - y_train_all = pd.DataFrame(y_train_all) - if not y_train_all.index.equals(X_train_all.index): - y_aligned = y_train_all.reindex(X_train_all.index) - if y_aligned.isna().sum().sum() > y_train_all.isna().sum().sum(): - raise ValueError("Target index labels do not match feature index labels.") - y_train_all = y_aligned + if isinstance(y_train_all.index, pd.DatetimeIndex): + if not y_train_all.index.equals(X_train_all.index): + y_aligned = y_train_all.reindex(X_train_all.index) + if y_aligned.isna().sum() > y_train_all.isna().sum(): + raise ValueError("Target index labels do not match feature index labels.") + y_train_all = y_aligned + y_train_all = pd.DataFrame(y_train_all) + else: + y_train_all = pd.DataFrame(y_train_all) + y_train_all.index = X_train_all.index elif isinstance(y_train_all, pd.DataFrame): - if not y_train_all.index.equals(X_train_all.index): + if ( + isinstance(X_train_all.index, pd.DatetimeIndex) + and isinstance(y_train_all.index, pd.RangeIndex) + and len(y_train_all) == len(X_train_all) + ): + y_train_all = y_train_all.copy() + y_train_all.index = X_train_all.index + elif not y_train_all.index.equals(X_train_all.index): y_aligned = y_train_all.reindex(X_train_all.index) if y_aligned.isna().sum().sum() > y_train_all.isna().sum().sum(): raise ValueError("Target index labels do not match feature index labels.") diff --git a/test/automl/test_ts_forecast_datetime_index.py b/test/automl/test_ts_forecast_datetime_index.py index 1fd9e77799..6bc23b98c9 100644 --- a/test/automl/test_ts_forecast_datetime_index.py +++ b/test/automl/test_ts_forecast_datetime_index.py @@ -208,3 +208,67 @@ def test_mismatched_target_index_raises_value_error(): automl = AutoML() with pytest.raises(ValueError, match="Target index labels do not match feature index labels"): _fit(automl, X_train=X_train, y_train=y_train) + + +def test_training_with_range_index_series_target(): + """Verify positional pairing when X_train has DatetimeIndex and y_train is a RangeIndex Series.""" + train_idx = pd.date_range("2018-01-01", periods=60, freq="MS") + X_train = pd.DataFrame({"feat": np.arange(60, dtype=float)}, index=train_idx) + y_train = pd.Series(np.sin(np.arange(60) / 6) * 10 + 50, name="y") # Default RangeIndex + + automl = AutoML() + _fit(automl, X_train=X_train, y_train=y_train) + + assert automl._state.task.time_col == "ds" + assert automl._state.data_size[0] == len(X_train) + preds = automl.predict(X_train.tail(12)) + assert len(preds) == 12 + + +def test_validation_with_range_index_series_target(): + """Verify positional pairing when validation data has DatetimeIndex X_val and RangeIndex Series y_val.""" + df_train = _make_train_df(periods=80) + val_idx = pd.date_range("2024-09-01", periods=20, freq="MS") + X_val = pd.DataFrame({"feat": np.arange(80, 100, dtype=float)}, index=val_idx) + y_val = pd.Series(np.sin(np.arange(80, 100) / 6) * 10 + 50, name="y") # Default RangeIndex + + automl = AutoML() + _fit(automl, dataframe=df_train, X_val=X_val, y_val=y_val) + + assert automl._state.eval_method == "holdout" + preds = automl.predict(X_val) + assert len(preds) == 20 + + +def test_training_and_validation_with_range_index_series_targets(): + """Verify positional pairing when both training and validation targets are RangeIndex Series.""" + train_idx = pd.date_range("2018-01-01", periods=80, freq="MS") + val_idx = pd.date_range("2024-09-01", periods=20, freq="MS") + X_train = pd.DataFrame({"feat": np.arange(80, dtype=float)}, index=train_idx) + y_train = pd.Series(np.sin(np.arange(80) / 6) * 10 + 50, name="y") + X_val = pd.DataFrame({"feat": np.arange(80, 100, dtype=float)}, index=val_idx) + y_val = pd.Series(np.sin(np.arange(80, 100) / 6) * 10 + 50, name="y") + + automl = AutoML() + _fit(automl, X_train=X_train, y_train=y_train, X_val=X_val, y_val=y_val) + + assert automl._state.task.time_col == "ds" + assert automl._state.eval_method == "holdout" + preds = automl.predict(X_val) + assert len(preds) == 20 + + +def test_training_with_reordered_series_target_aligns_by_label(): + """Verify that when y_train is a Series with shuffled timestamps, values align by label.""" + train_idx = pd.date_range("2018-01-01", periods=60, freq="MS") + X_train = pd.DataFrame({"feat": np.arange(60, dtype=float)}, index=train_idx) + shuffled_idx = train_idx[::-1] + y_values = np.sin(np.arange(60)[::-1] / 6) * 10 + 50 + y_train = pd.Series(y_values, index=shuffled_idx, name="y") + + automl = AutoML() + _fit(automl, X_train=X_train, y_train=y_train) + + preds = automl.predict(X_train.tail(12)) + assert len(preds) == 12 + assert automl._state.task.time_col == "ds" From 11fcf86c8bd61c842a84b0a754a407792a8469cb Mon Sep 17 00:00:00 2001 From: Nefelibata <124799179+MeiSiristhebest@users.noreply.github.com> Date: Wed, 30 Sep 2026 00:14:56 +0800 Subject: [PATCH 7/7] fix(automl): align datetime labels only when features also have DatetimeIndex Preserve positional Series contract when X has a RangeIndex (e.g. timestamps in a column or ndarray) and y has a DatetimeIndex. Only perform datetime-label alignment when both X and y have a DatetimeIndex. Add training and validation regressions for RangeIndex X with a datetime column and DatetimeIndex y. --- flaml/automl/task/time_series_task.py | 57 ++++++++++++------- flaml/automl/time_series/ts_data.py | 10 +++- .../automl/test_ts_forecast_datetime_index.py | 32 +++++++++++ 3 files changed, 75 insertions(+), 24 deletions(-) diff --git a/flaml/automl/task/time_series_task.py b/flaml/automl/task/time_series_task.py index e9e278dbf1..547110f581 100644 --- a/flaml/automl/task/time_series_task.py +++ b/flaml/automl/task/time_series_task.py @@ -422,37 +422,50 @@ def _preprocess(self, X, transformer=None): def _align_y_to_X_and_promote(X, y): """Align y to X by label before index promotion or resetting. - If X has a DatetimeIndex and y has a positional RangeIndex (or non-DatetimeIndex - Series), preserve positional pairing by assigning X.index to y. - Otherwise, if y has a DatetimeIndex or both are pandas objects with indexes, - align y to X.index by label. If non-matching labels introduce new missing + Only perform datetime-label alignment when X itself has a matching DatetimeIndex + and y also has a DatetimeIndex. If non-matching labels introduce new missing values, raise ValueError. + Otherwise (e.g. X has a DatetimeIndex and y has a RangeIndex, or X has a + RangeIndex and y is a Series with DatetimeIndex or RangeIndex), preserve + the positional Series contract by assigning X.index to y. """ if not isinstance(X, pd.DataFrame) or not isinstance(y, (pd.DataFrame, pd.Series)): return X, y - if isinstance(X.index, pd.DatetimeIndex): - if isinstance(y, pd.Series) and not isinstance(y.index, pd.DatetimeIndex): - if len(y) == len(X): - y = y.copy() - y.index = X.index - return X, y - elif isinstance(y, pd.DataFrame) and isinstance(y.index, pd.RangeIndex): + # Only perform datetime-label alignment when both X and y have a DatetimeIndex + if isinstance(X.index, pd.DatetimeIndex) and isinstance(y.index, pd.DatetimeIndex): + if not y.index.equals(X.index): + y_aligned = y.reindex(X.index) + orig_nan_count = int(y.isna().sum().sum() if isinstance(y, pd.DataFrame) else y.isna().sum()) + new_nan_count = int( + y_aligned.isna().sum().sum() if isinstance(y_aligned, pd.DataFrame) else y_aligned.isna().sum() + ) + if new_nan_count > orig_nan_count: + raise ValueError("Target index labels do not match feature index labels.") + y = y_aligned + return X, y + + # Positional pairing: when X has DatetimeIndex and y is a Series without DatetimeIndex (or RangeIndex DataFrame), + # or when X has RangeIndex and y is a Series (even if y has a DatetimeIndex). + if isinstance(y, pd.Series): + if len(y) == len(X): + y = y.copy() + y.index = X.index + return X, y + elif isinstance(y, pd.DataFrame): + if isinstance(X.index, pd.DatetimeIndex) and isinstance(y.index, pd.RangeIndex): if len(y) == len(X): y = y.copy() y.index = X.index return X, y - - if not y.index.equals(X.index): - y_aligned = y.reindex(X.index) - # If reindexing introduced NaNs that were not originally present, indexes do not match - orig_nan_count = int(y.isna().sum().sum() if isinstance(y, pd.DataFrame) else y.isna().sum()) - new_nan_count = int( - y_aligned.isna().sum().sum() if isinstance(y_aligned, pd.DataFrame) else y_aligned.isna().sum() - ) - if new_nan_count > orig_nan_count: - raise ValueError("Target index labels do not match feature index labels.") - y = y_aligned + elif not y.index.equals(X.index): + # When neither is DatetimeIndex, standard label alignment if indexes differ + y_aligned = y.reindex(X.index) + orig_nan_count = int(y.isna().sum().sum()) + new_nan_count = int(y_aligned.isna().sum().sum()) + if new_nan_count > orig_nan_count: + raise ValueError("Target index labels do not match feature index labels.") + y = y_aligned return X, y diff --git a/flaml/automl/time_series/ts_data.py b/flaml/automl/time_series/ts_data.py index 84fe583fc1..2c5f1bf399 100644 --- a/flaml/automl/time_series/ts_data.py +++ b/flaml/automl/time_series/ts_data.py @@ -547,7 +547,7 @@ def normalize_ts_data(X_train_all, target_names, time_col, y_train_all=None): index=X_train_all.index, ) elif isinstance(y_train_all, pd.Series): - if isinstance(y_train_all.index, pd.DatetimeIndex): + if isinstance(X_train_all.index, pd.DatetimeIndex) and isinstance(y_train_all.index, pd.DatetimeIndex): if not y_train_all.index.equals(X_train_all.index): y_aligned = y_train_all.reindex(X_train_all.index) if y_aligned.isna().sum() > y_train_all.isna().sum(): @@ -558,7 +558,13 @@ def normalize_ts_data(X_train_all, target_names, time_col, y_train_all=None): y_train_all = pd.DataFrame(y_train_all) y_train_all.index = X_train_all.index elif isinstance(y_train_all, pd.DataFrame): - if ( + if isinstance(X_train_all.index, pd.DatetimeIndex) and isinstance(y_train_all.index, pd.DatetimeIndex): + if not y_train_all.index.equals(X_train_all.index): + y_aligned = y_train_all.reindex(X_train_all.index) + if y_aligned.isna().sum().sum() > y_train_all.isna().sum().sum(): + raise ValueError("Target index labels do not match feature index labels.") + y_train_all = y_aligned + elif ( isinstance(X_train_all.index, pd.DatetimeIndex) and isinstance(y_train_all.index, pd.RangeIndex) and len(y_train_all) == len(X_train_all) diff --git a/test/automl/test_ts_forecast_datetime_index.py b/test/automl/test_ts_forecast_datetime_index.py index 6bc23b98c9..97590c6ba0 100644 --- a/test/automl/test_ts_forecast_datetime_index.py +++ b/test/automl/test_ts_forecast_datetime_index.py @@ -272,3 +272,35 @@ def test_training_with_reordered_series_target_aligns_by_label(): preds = automl.predict(X_train.tail(12)) assert len(preds) == 12 assert automl._state.task.time_col == "ds" + + +def test_training_with_range_index_x_datetime_col_and_datetime_index_series_target(): + """Verify positional pairing when X_train is RangeIndex with a datetime column and y_train is DatetimeIndex Series.""" + dates = pd.date_range("2020-01-01", periods=60, freq="MS") + X_train = pd.DataFrame({"ds": dates, "feat": np.arange(60, dtype=float)}) + y_train = pd.Series(np.sin(np.arange(60) / 6) * 10 + 50, index=dates, name="y") + + automl = AutoML() + _fit(automl, X_train=X_train, y_train=y_train, time_col="ds") + + assert automl._state.task.time_col == "ds" + assert automl._state.data_size[0] == len(X_train) + preds = automl.predict(X_train.tail(12)) + assert len(preds) == 12 + + +def test_validation_with_range_index_x_datetime_col_and_datetime_index_series_target(): + """Verify positional pairing when X_val is RangeIndex with a datetime column and y_val is DatetimeIndex Series.""" + train_dates = pd.date_range("2018-01-01", periods=80, freq="MS") + val_dates = pd.date_range("2024-09-01", periods=20, freq="MS") + X_train = pd.DataFrame({"ds": train_dates, "feat": np.arange(80, dtype=float)}) + y_train = pd.Series(np.sin(np.arange(80) / 6) * 10 + 50, name="y") + X_val = pd.DataFrame({"ds": val_dates, "feat": np.arange(80, 100, dtype=float)}) + y_val = pd.Series(np.sin(np.arange(80, 100) / 6) * 10 + 50, index=val_dates, name="y") + + automl = AutoML() + _fit(automl, X_train=X_train, y_train=y_train, X_val=X_val, y_val=y_val, time_col="ds") + + assert automl._state.eval_method == "holdout" + preds = automl.predict(X_val) + assert len(preds) == 20