diff --git a/flaml/automl/time_series/ts_model.py b/flaml/automl/time_series/ts_model.py index 445cf337b8..1207f7eabe 100644 --- a/flaml/automl/time_series/ts_model.py +++ b/flaml/automl/time_series/ts_model.py @@ -334,6 +334,9 @@ def predict(self, X, **kwargs): class StatsModelsEstimator(TimeSeriesEstimator): def predict(self, X, **kwargs) -> pd.Series: + if isinstance(X, int) and self.train_end_date is not None: + # Forecast the requested number of periods right after the fitted training data + X = create_forward_frame(self.frequency, X, self.train_end_date, self.time_col) X = self.enrich(X) if self._model is None or self._model is False: return np.ones(X if isinstance(X, int) else X.shape[0]) @@ -384,10 +387,19 @@ def predict(self, X, **kwargs) -> pd.Series: forecast = forecast.iloc[positions] elif self.train_end_date is not None and end > self.train_end_date: raise ValueError("Prediction timestamps cannot span both training and future periods.") - elif exog is not None: - forecast = self._model.predict(start=start, end=end, exog=exog, **kwargs) else: - forecast = self._model.predict(start=start, end=end, **kwargs) + if exog is not None: + forecast = self._model.predict(start=start, end=end, exog=exog, **kwargs) + else: + forecast = self._model.predict(start=start, end=end, **kwargs) + # The model predicts every period from start to end; keep only the requested timestamps + positions = pd.DatetimeIndex(forecast.index).get_indexer(pd.DatetimeIndex(X[self.time_col])) + if (positions < 0).any() or (positions[1:] <= positions[:-1]).any(): + raise ValueError( + "Prediction timestamps must be unique, increasing, and aligned with the training frequency." + ) + if len(positions) != len(forecast): + forecast = forecast.iloc[positions] else: raise ValueError( "X needs to be either a pandas Dataframe with dates as the first column" @@ -736,34 +748,53 @@ def fit(self, X_train, y_train=None, budget=None, **kwargs): return train_time +class SeasonalRandomWalk: + """Point forecasts of a seasonal random walk, where each value repeats the one a season earlier.""" + + def __init__(self, y: pd.Series, season: int, frequency: str): + self.season = season + self.frequency = frequency + self.last_date = y.index[-1] + self.last_cycle = y.to_numpy()[-season:] + # In-sample predictions; the first season has no earlier value and uses the first observation + self.fittedvalues = y.shift(season).fillna(y.iloc[0]) + + def forecast(self, steps: int, **kwargs) -> pd.Series: + start = self.last_date + pd.tseries.frequencies.to_offset(self.frequency) + index = pd.date_range(start=start, periods=steps, freq=self.frequency) + return pd.Series(np.resize(self.last_cycle, steps), index=index) + + def predict(self, start, end, **kwargs) -> pd.Series: + return self.fittedvalues.loc[start:end] + + class SeasonalNaive(SimpleForecaster): smoothing_level = 1.0 - def predict(self, X, **kwargs): - if isinstance(X, int): - forecasts = [] - for i in range(X): - forecast = self._model.forecast(steps=self.season)[0] - forecasts.append(forecast) - return pd.Series(forecasts) - else: - return super().predict(X, **kwargs) + def fit(self, X_train, y_train=None, budget=None, **kwargs): + season = self.params.get("season", 1) + if isinstance(season, bool) or not isinstance(season, (int, np.integer)) or season < 1: + raise ValueError(f"season must be a positive integer, got {season!r}.") + if season == 1: + return super().fit(X_train, y_train, budget=budget, **kwargs) + current_time = time.time() + self.season = int(season) + train_df, target_col = self.joint_preprocess(X_train, y_train) + if len(train_df) < self.season: + raise ValueError( + f"SeasonalNaive needs at least season={self.season} training observations, got {len(train_df)}." + ) + self._model = SeasonalRandomWalk(train_df[target_col], self.season, self.frequency) + return time.time() - current_time class Naive(SimpleForecaster): - smoothing_level = 0.0 + smoothing_level = 1.0 @classmethod def _search_space(cls, data: TimeSeriesDataset, task: Task, pred_horizon: int, **params): return {} - def predict(self, X, **kwargs): - if isinstance(X, int): - last_observation = self._model.params["initial_level"] - return pd.Series([last_observation] * X) - else: - return super().predict(X, **kwargs) - class SeasonalAverage(SimpleForecaster): def fit(self, X_train, y_train=None, budget=None, **kwargs): @@ -771,7 +802,7 @@ def fit(self, X_train, y_train=None, budget=None, **kwargs): start_time = time.time() - self.season = kwargs.get("season", 1) # seasonality period + self.season = kwargs.get("season", self.params.get("season", 1)) # seasonality period train_df, target_col = self.joint_preprocess(X_train, y_train) selection_res = ar_select_order(train_df[target_col], maxlag=self.season) diff --git a/test/automl/test_forecast.py b/test/automl/test_forecast.py index e5c262c3c5..153e525859 100644 --- a/test/automl/test_forecast.py +++ b/test/automl/test_forecast.py @@ -170,6 +170,94 @@ def test_average_forecasters_set_training_boundary(estimator_name): assert estimator.train_end_date == dates[-1] +def _weekly_dataset(): + from flaml.automl.time_series import TimeSeriesDataset + + t = np.arange(84) + weekly = np.array([0.0, 5.0, 2.0, -3.0, 8.0, -6.0, 1.0]) + dates = pd.date_range("2024-01-01", periods=84, freq="D") + train_data = pd.DataFrame({"ds": dates, "y": 100 + 0.1 * t + weekly[t % 7]}) + future = pd.DataFrame({"ds": pd.date_range(dates[-1] + pd.Timedelta(days=1), periods=14, freq="D")}) + return TimeSeriesDataset(train_data, time_col="ds", target_names="y"), train_data["y"], future + + +def test_naive_forecasts_last_observation(): + from flaml.automl.time_series import Naive + + dataset, y, future = _weekly_dataset() + estimator = Naive() + estimator.fit(dataset) + + np.testing.assert_allclose(estimator.predict(future), y.iloc[-1]) + np.testing.assert_allclose(estimator.predict(3), y.iloc[-1]) + + +def test_seasonal_naive_repeats_last_season(): + from flaml.automl.time_series import SeasonalNaive + + dataset, y, future = _weekly_dataset() + estimator = SeasonalNaive(season=7) + estimator.fit(dataset) + + last_season = y.iloc[-7:].to_numpy() + np.testing.assert_allclose(estimator.predict(future), np.tile(last_season, 2)) + np.testing.assert_allclose(estimator.predict(7), last_season) + + +def test_seasonal_naive_predicts_in_sample_one_season_back(): + from flaml.automl.time_series import SeasonalNaive + + dataset, y, _ = _weekly_dataset() + estimator = SeasonalNaive(season=7) + estimator.fit(dataset) + + in_sample = dataset.train_data[["ds"]].iloc[20:40] + np.testing.assert_allclose(estimator.predict(in_sample), y.iloc[13:33]) + + sparse = dataset.train_data[["ds"]].iloc[[20, 22, 30]] + np.testing.assert_allclose(estimator.predict(sparse), y.iloc[[13, 15, 23]]) + + +def test_seasonal_naive_integer_horizon_starts_after_training_data(): + from flaml.automl.time_series import SeasonalNaive, TimeSeriesDataset + + _, y, _ = _weekly_dataset() + data = pd.DataFrame({"ds": pd.date_range("2024-01-01", periods=len(y), freq="D"), "y": y}) + # 10 held-out rows, not a multiple of the season, so a forecast that started after them would be out of phase + dataset = TimeSeriesDataset(data.iloc[:70], time_col="ds", target_names="y", test_data=data.iloc[70:80]) + estimator = SeasonalNaive(season=7) + estimator.fit(dataset) + + np.testing.assert_allclose(estimator.predict(7), y.iloc[63:70]) + + +@pytest.mark.parametrize("season", [0, -3, 2.5]) +def test_seasonal_naive_rejects_invalid_season(season): + from flaml.automl.time_series import SeasonalNaive + + dataset, _, _ = _weekly_dataset() + with pytest.raises(ValueError, match="season must be a positive integer"): + SeasonalNaive(season=season).fit(dataset) + + +def test_seasonal_naive_requires_a_full_season(): + from flaml.automl.time_series import SeasonalNaive + + dataset, _, _ = _weekly_dataset() + with pytest.raises(ValueError, match="at least season=100"): + SeasonalNaive(season=100).fit(dataset) + + +def test_seasonal_average_uses_tuned_season(): + from flaml.automl.time_series import SeasonalAverage + + dataset, _, _ = _weekly_dataset() + estimator = SeasonalAverage(season=7) + estimator.fit(dataset) + + assert estimator.season == 7 + + def test_numpy(): X_train = np.arange("2014-01", "2021-01", dtype="datetime64[M]") y_train = np.random.random(size=len(X_train))