Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
73 changes: 52 additions & 21 deletions flaml/automl/time_series/ts_model.py
Original file line number Diff line number Diff line change
Expand Up @@ -334,6 +334,9 @@ def predict(self, X, **kwargs):

class StatsModelsEstimator(TimeSeriesEstimator):
def predict(self, X, **kwargs) -> pd.Series:
if isinstance(X, int) and self.train_end_date is not None:
# Forecast the requested number of periods right after the fitted training data
X = create_forward_frame(self.frequency, X, self.train_end_date, self.time_col)
X = self.enrich(X)
if self._model is None or self._model is False:
return np.ones(X if isinstance(X, int) else X.shape[0])
Expand Down Expand Up @@ -384,10 +387,19 @@ def predict(self, X, **kwargs) -> pd.Series:
forecast = forecast.iloc[positions]
elif self.train_end_date is not None and end > self.train_end_date:
raise ValueError("Prediction timestamps cannot span both training and future periods.")
elif exog is not None:
forecast = self._model.predict(start=start, end=end, exog=exog, **kwargs)
else:
forecast = self._model.predict(start=start, end=end, **kwargs)
if exog is not None:
forecast = self._model.predict(start=start, end=end, exog=exog, **kwargs)
else:
forecast = self._model.predict(start=start, end=end, **kwargs)
# The model predicts every period from start to end; keep only the requested timestamps
positions = pd.DatetimeIndex(forecast.index).get_indexer(pd.DatetimeIndex(X[self.time_col]))
if (positions < 0).any() or (positions[1:] <= positions[:-1]).any():
raise ValueError(
"Prediction timestamps must be unique, increasing, and aligned with the training frequency."
)
if len(positions) != len(forecast):
forecast = forecast.iloc[positions]
else:
raise ValueError(
"X needs to be either a pandas Dataframe with dates as the first column"
Expand Down Expand Up @@ -736,42 +748,61 @@ def fit(self, X_train, y_train=None, budget=None, **kwargs):
return train_time


class SeasonalRandomWalk:
"""Point forecasts of a seasonal random walk, where each value repeats the one a season earlier."""

def __init__(self, y: pd.Series, season: int, frequency: str):
self.season = season
self.frequency = frequency
self.last_date = y.index[-1]
self.last_cycle = y.to_numpy()[-season:]
# In-sample predictions; the first season has no earlier value and uses the first observation
self.fittedvalues = y.shift(season).fillna(y.iloc[0])

def forecast(self, steps: int, **kwargs) -> pd.Series:
start = self.last_date + pd.tseries.frequencies.to_offset(self.frequency)
index = pd.date_range(start=start, periods=steps, freq=self.frequency)
return pd.Series(np.resize(self.last_cycle, steps), index=index)

def predict(self, start, end, **kwargs) -> pd.Series:
return self.fittedvalues.loc[start:end]


class SeasonalNaive(SimpleForecaster):
smoothing_level = 1.0

def predict(self, X, **kwargs):
if isinstance(X, int):
forecasts = []
for i in range(X):
forecast = self._model.forecast(steps=self.season)[0]
forecasts.append(forecast)
return pd.Series(forecasts)
else:
return super().predict(X, **kwargs)
def fit(self, X_train, y_train=None, budget=None, **kwargs):
season = self.params.get("season", 1)
if isinstance(season, bool) or not isinstance(season, (int, np.integer)) or season < 1:
raise ValueError(f"season must be a positive integer, got {season!r}.")
if season == 1:
return super().fit(X_train, y_train, budget=budget, **kwargs)
current_time = time.time()
self.season = int(season)
train_df, target_col = self.joint_preprocess(X_train, y_train)
if len(train_df) < self.season:
raise ValueError(
f"SeasonalNaive needs at least season={self.season} training observations, got {len(train_df)}."
)
self._model = SeasonalRandomWalk(train_df[target_col], self.season, self.frequency)
return time.time() - current_time


class Naive(SimpleForecaster):
smoothing_level = 0.0
smoothing_level = 1.0

@classmethod
def _search_space(cls, data: TimeSeriesDataset, task: Task, pred_horizon: int, **params):
return {}

def predict(self, X, **kwargs):
if isinstance(X, int):
last_observation = self._model.params["initial_level"]
return pd.Series([last_observation] * X)
else:
return super().predict(X, **kwargs)


class SeasonalAverage(SimpleForecaster):
def fit(self, X_train, y_train=None, budget=None, **kwargs):
from statsmodels.tsa.ar_model import AutoReg, ar_select_order

start_time = time.time()

self.season = kwargs.get("season", 1) # seasonality period
self.season = kwargs.get("season", self.params.get("season", 1)) # seasonality period
train_df, target_col = self.joint_preprocess(X_train, y_train)
selection_res = ar_select_order(train_df[target_col], maxlag=self.season)

Expand Down
88 changes: 88 additions & 0 deletions test/automl/test_forecast.py
Original file line number Diff line number Diff line change
Expand Up @@ -170,6 +170,94 @@ def test_average_forecasters_set_training_boundary(estimator_name):
assert estimator.train_end_date == dates[-1]


def _weekly_dataset():
from flaml.automl.time_series import TimeSeriesDataset

t = np.arange(84)
weekly = np.array([0.0, 5.0, 2.0, -3.0, 8.0, -6.0, 1.0])
dates = pd.date_range("2024-01-01", periods=84, freq="D")
train_data = pd.DataFrame({"ds": dates, "y": 100 + 0.1 * t + weekly[t % 7]})
future = pd.DataFrame({"ds": pd.date_range(dates[-1] + pd.Timedelta(days=1), periods=14, freq="D")})
return TimeSeriesDataset(train_data, time_col="ds", target_names="y"), train_data["y"], future


def test_naive_forecasts_last_observation():
from flaml.automl.time_series import Naive

dataset, y, future = _weekly_dataset()
estimator = Naive()
estimator.fit(dataset)

np.testing.assert_allclose(estimator.predict(future), y.iloc[-1])
np.testing.assert_allclose(estimator.predict(3), y.iloc[-1])


def test_seasonal_naive_repeats_last_season():
from flaml.automl.time_series import SeasonalNaive

dataset, y, future = _weekly_dataset()
estimator = SeasonalNaive(season=7)
estimator.fit(dataset)

last_season = y.iloc[-7:].to_numpy()
np.testing.assert_allclose(estimator.predict(future), np.tile(last_season, 2))
np.testing.assert_allclose(estimator.predict(7), last_season)


def test_seasonal_naive_predicts_in_sample_one_season_back():
from flaml.automl.time_series import SeasonalNaive

dataset, y, _ = _weekly_dataset()
estimator = SeasonalNaive(season=7)
estimator.fit(dataset)

in_sample = dataset.train_data[["ds"]].iloc[20:40]
np.testing.assert_allclose(estimator.predict(in_sample), y.iloc[13:33])

sparse = dataset.train_data[["ds"]].iloc[[20, 22, 30]]
np.testing.assert_allclose(estimator.predict(sparse), y.iloc[[13, 15, 23]])


def test_seasonal_naive_integer_horizon_starts_after_training_data():
from flaml.automl.time_series import SeasonalNaive, TimeSeriesDataset

_, y, _ = _weekly_dataset()
data = pd.DataFrame({"ds": pd.date_range("2024-01-01", periods=len(y), freq="D"), "y": y})
# 10 held-out rows, not a multiple of the season, so a forecast that started after them would be out of phase
dataset = TimeSeriesDataset(data.iloc[:70], time_col="ds", target_names="y", test_data=data.iloc[70:80])
estimator = SeasonalNaive(season=7)
estimator.fit(dataset)

np.testing.assert_allclose(estimator.predict(7), y.iloc[63:70])


@pytest.mark.parametrize("season", [0, -3, 2.5])
def test_seasonal_naive_rejects_invalid_season(season):
from flaml.automl.time_series import SeasonalNaive

dataset, _, _ = _weekly_dataset()
with pytest.raises(ValueError, match="season must be a positive integer"):
SeasonalNaive(season=season).fit(dataset)


def test_seasonal_naive_requires_a_full_season():
from flaml.automl.time_series import SeasonalNaive

dataset, _, _ = _weekly_dataset()
with pytest.raises(ValueError, match="at least season=100"):
SeasonalNaive(season=100).fit(dataset)


def test_seasonal_average_uses_tuned_season():
from flaml.automl.time_series import SeasonalAverage

dataset, _, _ = _weekly_dataset()
estimator = SeasonalAverage(season=7)
estimator.fit(dataset)

assert estimator.season == 7


def test_numpy():
X_train = np.arange("2014-01", "2021-01", dtype="datetime64[M]")
y_train = np.random.random(size=len(X_train))
Expand Down
Loading