Source code for autogluon.cloud.predictor.timeseries_cloud_predictor

from __future__ import annotations

import logging
from pathlib import Path
from typing import Any

import pandas as pd
from typing_extensions import deprecated

from ..backend.constant import SAGEMAKER, TIMESERIES_SAGEMAKER
from ..endpoint.timeseries_endpoint import TimeSeriesEndpoint
from ..utils.constants import DEFAULT_FRAMEWORK_VERSION, DEFAULT_VOLUME_SIZE
from ..utils.sagemaker_api import reject_legacy_kwargs
from .cloud_predictor import _DEPRECATED_REAL_TIME, CloudPredictor

logger = logging.getLogger(__name__)


[docs] class TimeSeriesCloudPredictor(CloudPredictor[TimeSeriesEndpoint]): """Train and deploy AutoGluon time series forecasting models on Amazon SageMaker. Wraps :class:`autogluon.timeseries.TimeSeriesPredictor` (`docs <https://auto.gluon.ai/stable/api/autogluon.timeseries.TimeSeriesPredictor.html>`_) and runs ``fit``, ``predict``, and endpoint deployment as managed SageMaker jobs. """ predictor_file_name = "TimeSeriesCloudPredictor.pkl" backend_map = {SAGEMAKER: TIMESERIES_SAGEMAKER} _endpoint_cls = TimeSeriesEndpoint @property def predictor_type(self): """ Type of the underneath AutoGluon Predictor """ return "timeseries" def _get_local_predictor_cls(self): from autogluon.timeseries import TimeSeriesPredictor return TimeSeriesPredictor
[docs] @reject_legacy_kwargs def fit( self, train_data: str | Path | pd.DataFrame | None = None, *, predictor_init_args: dict[str, Any], predictor_fit_args: dict[str, Any] | None = None, tuning_data: str | Path | pd.DataFrame | None = None, static_features: str | Path | pd.DataFrame | None = None, id_column: str = "item_id", timestamp_column: str = "timestamp", framework_version: str = DEFAULT_FRAMEWORK_VERSION, job_name: str | None = None, instance_type: str = "ml.m5.2xlarge", instance_count: int = 1, volume_size: int = DEFAULT_VOLUME_SIZE, custom_image_uri: str | None = None, wait: bool = True, backend_overrides: dict[str, dict[str, Any]] | None = None, known_covariates: str | Path | pd.DataFrame | None = None, **kwargs, ) -> TimeSeriesCloudPredictor: """ Fit the predictor in a SageMaker training job. Parameters ---------- train_data: str | pathlib.Path | pd.DataFrame Training time series in long format, as a ``pd.DataFrame`` or local/S3 path to a data file. See the `TimeSeriesPredictor.fit docs <https://auto.gluon.ai/stable/api/autogluon.timeseries.TimeSeriesPredictor.fit.html>`_ for the expected format. predictor_init_args: dict Arguments forwarded to ``TimeSeriesPredictor()``. See the `TimeSeriesPredictor docs <https://auto.gluon.ai/stable/api/autogluon.timeseries.TimeSeriesPredictor.html>`_ for available options (e.g. ``target``, ``prediction_length``, ``freq``, ``eval_metric``, ``quantile_levels``, ``known_covariates_names``). predictor_fit_args: dict | None, default = None Additional fit args forwarded to ``TimeSeriesPredictor.fit()``. See the `TimeSeriesPredictor.fit docs <https://auto.gluon.ai/stable/api/autogluon.timeseries.TimeSeriesPredictor.fit.html>`_ for available options. Must NOT contain ``train_data`` or ``tuning_data`` — pass those as explicit arguments above. tuning_data: str | pathlib.Path | pd.DataFrame | None, default = None Optional tuning data in long format, as a ``pd.DataFrame`` or local/S3 path to a data file. known_covariates: str | pathlib.Path | pd.DataFrame | None, default = None Values of the known covariates. Must be provided if ``known_covariates_names`` was specified in ``predictor_init_args``. static_features: str | pathlib.Path | pd.DataFrame | None, default = None Static (time-independent) features describing each individual time series. id_column: str, default = "item_id" Name of the column with the unique identifier of each time series (item). timestamp_column: str, default = "timestamp" Name of the column with the observation timestamps. framework_version: str, optional AutoGluon version, e.g. "1.6". Training uses the official AutoGluon DLC image for this version. If `custom_image_uri` is set, this argument will be ignored. job_name: str, default = None Name of the launched training job. If None, CloudPredictor creates one with prefix ``ag-cloud-timeseries``. instance_type: str, default = 'ml.m5.2xlarge' Instance type the predictor will be trained on with SageMaker. instance_count: int, default = 1 Number of instances used to fit the predictor. volume_size: int, default = 100 Size in GB of the EBS volume to use for storing input data during training. Must be large enough to store training data if File Mode is used (which is the default). custom_image_uri: str | None, default = None Custom container image URI. If set, ``framework_version`` is ignored. wait: bool, default = True Whether the call should wait until the job completes To be noticed, the function won't return immediately because there are some preparations needed prior fit. Use `get_fit_job_status` to get job status. backend_overrides: dict[str, dict[str, Any]] | None, default = None Raw SageMaker request fields for settings without a dedicated argument. * Keys: request names from the *SageMaker API* section below. * Values: request fields in PascalCase, as in the SageMaker API and boto3. Deep-merged over the request built by AutoGluon-Cloud; lists and other non-dict values replace the generated ones. * Example: ``{"CreateTrainingJob": {"RetryStrategy": {"MaximumRetryAttempts": 2}}}`` Returns ------- TimeSeriesCloudPredictor The fitted predictor (``self``). SageMaker API ------------- * :sm-api:`CreateTrainingJob`: trains the predictor on ``instance_count`` x ``instance_type`` and writes the artifact to ``cloud_output_path``. """ assert not self.backend.is_fit, ( "Predictor is already fit! To fit additional models, create a new `CloudPredictor`" ) # `extra_ag_args` is an internal channel for `fit_predict`; it is intentionally not part of the public signature. extra_ag_args = kwargs.pop("extra_ag_args", None) if kwargs: raise TypeError(f"fit() got unexpected keyword arguments: {sorted(kwargs)}") predictor_fit_args = {} if predictor_fit_args is None else dict(predictor_fit_args) data_channels = { "train_data": train_data, "tuning_data": tuning_data, "known_covariates": known_covariates, "static_features": static_features, } for key in ("train_data", "tuning_data", "known_covariates"): if key in predictor_fit_args: raise TypeError( f"`{key}` can no longer be passed via `predictor_fit_args`. " f"Pass `{key}` as an explicit argument to `fit()` instead." ) if data_channels["train_data"] is None: raise TypeError("fit() missing required argument: 'train_data'") self.backend.fit( predictor_init_args=predictor_init_args, predictor_fit_args=predictor_fit_args, data_channels=data_channels, id_column=id_column, timestamp_column=timestamp_column, framework_version=framework_version, job_name=job_name, instance_type=instance_type, instance_count=instance_count, volume_size=volume_size, custom_image_uri=custom_image_uri, wait=wait, backend_overrides=backend_overrides, extra_ag_args=extra_ag_args, ) return self
@deprecated(_DEPRECATED_REAL_TIME, category=None) def predict_real_time( self, data: str | pd.DataFrame, static_features: str | pd.DataFrame | None = None, known_covariates: pd.DataFrame | None = None, accept: str = "application/x-parquet", **kwargs, ) -> pd.DataFrame: """ Predict with the deployed SageMaker endpoint. A deployed SageMaker endpoint is required. This is intended to provide a low latency inference. If you want to inference on a large dataset, use `predict()` instead. :meta private: .. deprecated:: Use ``predict()`` of the endpoint returned by :meth:`deploy` instead. ``data`` must use the same ``id_column`` / ``timestamp_column`` names that were passed to ``fit()``. Parameters ---------- data: str | pd.DataFrame Historical time series to forecast from, in long format, as a ``pd.DataFrame`` or local/S3 path to a data file. static_features: pd.DataFrame | None Static (time-independent) features describing each individual time series. known_covariates: pd.DataFrame | None Future values of the known covariates over the forecast horizon. Must be provided if ``known_covariates_names`` was specified at fit time. accept: str, default = application/x-parquet Type of accept output content. Valid options are application/x-parquet, text/csv, application/json **kwargs: Any Additional args that you would pass to `predict` calls of an AutoGluon logic Returns ------- pd.DataFrame Predict results in ``pd.DataFrame`` SageMaker API ------------- * :sm-runtime-api:`InvokeEndpoint`: sends the data to the endpoint and returns the predictions. The payload is limited to 6 MB (4 MB for serverless endpoints). """ self._warn_deprecated_real_time("predict_real_time") return self.backend.predict_real_time( test_data=data, static_features=static_features, known_covariates=known_covariates, accept=accept, inference_kwargs=kwargs, ) def predict_proba_real_time(self, **kwargs) -> pd.DataFrame: """ :meta private: """ raise ValueError(f"{self.__class__.__name__} does not support predict_proba operation.")
[docs] @reject_legacy_kwargs def predict( self, data: str | pd.DataFrame, static_features: str | pd.DataFrame | None = None, known_covariates: str | pd.DataFrame | None = None, predictor_path: str | None = None, framework_version: str | None = None, job_name: str | None = None, instance_type: str = "ml.m5.2xlarge", instance_count: int = 1, custom_image_uri: str | None = None, wait: bool = True, predictions_path: str | None = None, backend_overrides: dict[str, dict[str, Any]] | None = None, ) -> pd.DataFrame | None: """ Predict using SageMaker batch transform. When minimizing latency isn't a concern, then the batch transform functionality may be easier, more scalable, and more appropriate. If you want to minimize latency, deploy an endpoint with `deploy()` instead. To learn more: https://docs.aws.amazon.com/sagemaker/latest/dg/batch-transform.html ``data`` must use the same ``id_column`` / ``timestamp_column`` names that were passed to ``fit()``. Parameters ---------- data: str | pd.DataFrame Historical time series to forecast from, in long format, as a ``pd.DataFrame`` or local/S3 path to a data file. static_features: str | pd.DataFrame | None Static (time-independent) features describing each individual time series. known_covariates: str | pd.DataFrame | None Future values of the known covariates over the forecast horizon. Must be provided if ``known_covariates_names`` was specified at fit time. predictor_path: str Path to the predictor tarball you want to use to predict. Path can be both a local path or a S3 location. If None, will use the most recent trained predictor trained with `fit()`. framework_version: str, optional AutoGluon version, e.g. "1.6". Inference uses the official AutoGluon DLC image for this version. Defaults to the version used by `fit()`. If `custom_image_uri` is set, this argument will be ignored. job_name: str, default = None Name of the launched training job. If None, CloudPredictor creates one with prefix ``ag-cloud-timeseries``. instance_count: int, default = 1, Number of instances used to do batch transform. instance_type: str, default = 'ml.m5.2xlarge' Instance to be used for batch transform. wait: bool, default = True Whether to wait for batch transform to complete. To be noticed, the function won't return immediately because there are some preparations needed prior transform. predictions_path: str | None, default = None S3 prefix under which the batch transform job writes its results (``<predictions_path>/<input file>.out``). Defaults to ``{cloud_output_path}/batch_transform/<timestamp>/results``. backend_overrides: dict[str, dict[str, Any]] | None, default = None Raw SageMaker request fields for settings without a dedicated argument. * Keys: request names from the *SageMaker API* section below. * Values: request fields in PascalCase, as in the SageMaker API and boto3. Deep-merged over the request built by AutoGluon-Cloud; lists and other non-dict values replace the generated ones. * Example: ``{"CreateTransformJob": {"BatchStrategy": "SingleRecord", "MaxPayloadInMB": 20}}`` SageMaker API ------------- * :sm-api:`CreateModel`: registers the predictor artifact and inference image as a SageMaker model. * :sm-api:`CreateTransformJob`: runs batch inference on ``instance_count`` x ``instance_type``. Results are written to ``predictions_path``. The model is deleted when the job finishes. With ``wait=False`` it is kept; delete it with :sm-api:`DeleteModel`. """ return self.backend.predict( test_data=data, static_features=static_features, known_covariates=known_covariates, predictor_path=predictor_path, framework_version=framework_version, job_name=job_name, instance_type=instance_type, instance_count=instance_count, custom_image_uri=custom_image_uri, wait=wait, predictions_path=predictions_path, backend_overrides=backend_overrides, )
def predict_proba( self, **kwargs, ) -> pd.DataFrame | None: """ :meta private: """ raise ValueError(f"{self.__class__.__name__} does not support predict_proba operation.")
[docs] @reject_legacy_kwargs def fit_predict( self, train_data: str | Path | pd.DataFrame, *, predictor_init_args: dict[str, Any], predictor_fit_args: dict[str, Any] | None = None, known_covariates: str | Path | pd.DataFrame | None = None, static_features: str | Path | pd.DataFrame | None = None, id_column: str = "item_id", timestamp_column: str = "timestamp", predictions_path: str | None = None, framework_version: str = DEFAULT_FRAMEWORK_VERSION, job_name: str | None = None, instance_type: str = "ml.m5.2xlarge", instance_count: int = 1, volume_size: int = DEFAULT_VOLUME_SIZE, custom_image_uri: str | None = None, wait: bool = True, backend_overrides: dict[str, dict[str, Any]] | None = None, ) -> pd.DataFrame | None: """ Fit and predict in a single SageMaker training job. Predictions are generated inside the training container against ``train_data`` (the standard time-series forecasting flow where the last ``prediction_length`` steps of each series are forecast) and written directly to S3. Parameters ---------- train_data: str | pathlib.Path | pd.DataFrame Historical time series to train on and forecast from, in long format, as a ``pd.DataFrame`` or local/S3 path to a data file. predictor_init_args: dict Arguments forwarded to ``TimeSeriesPredictor()``. Must include ``prediction_length``. See the `TimeSeriesPredictor docs <https://auto.gluon.ai/stable/api/autogluon.timeseries.TimeSeriesPredictor.html>`_ for available options. predictor_fit_args: dict | None, default = None Additional fit args forwarded to ``TimeSeriesPredictor.fit()``. See the `TimeSeriesPredictor.fit docs <https://auto.gluon.ai/stable/api/autogluon.timeseries.TimeSeriesPredictor.fit.html>`_ for available options. Must NOT contain ``train_data``, ``tuning_data``, or ``known_covariates`` — pass those as explicit arguments above. known_covariates: str | pathlib.Path | pd.DataFrame | None, default = None Future values of the known covariates over the forecast horizon. Must be provided if ``known_covariates_names`` was specified in ``predictor_init_args``. static_features: str | pathlib.Path | pd.DataFrame | None, default = None Static (time-independent) features describing each individual time series. id_column: str, default = "item_id" Name of the column with the unique identifier of each time series (item). timestamp_column: str, default = "timestamp" Name of the column with the observation timestamps. predictions_path: str | None S3 URL where predictions will be written by the training container (e.g. ``s3://my-bucket/runs/2024-05-01/predictions.csv``). The container's SageMaker execution role must have ``s3:PutObject`` permission for this location. Defaults to ``{cloud_output_path}/{job_name}/predictions.csv``. Predictions use AutoGluon's canonical column names ``item_id`` and ``timestamp``, regardless of the ``id_column`` / ``timestamp_column`` passed in. framework_version: str, optional AutoGluon version, e.g. "1.6". Training uses the official AutoGluon DLC image for this version. If `custom_image_uri` is set, this argument will be ignored. job_name: str, default = None Name of the launched training job. If None, CloudPredictor creates one with prefix ``ag-cloud-timeseries``. instance_type: str, default = 'ml.m5.2xlarge' Instance type the predictor will be trained on with SageMaker. instance_count: int, default = 1 Number of instances used to fit the predictor. volume_size: int, default = 100 Size in GB of the EBS volume to use for storing input data during training. custom_image_uri: str | None, default = None Custom container image URI. If set, ``framework_version`` is ignored. wait: bool, default = True Whether the call should wait until the job completes. backend_overrides: dict[str, dict[str, Any]] | None, default = None Raw SageMaker request fields for settings without a dedicated argument. * Keys: request names from the *SageMaker API* section below. * Values: request fields in PascalCase, as in the SageMaker API and boto3. Deep-merged over the request built by AutoGluon-Cloud; lists and other non-dict values replace the generated ones. * Example: ``{"CreateTrainingJob": {"RetryStrategy": {"MaximumRetryAttempts": 2}}}`` Returns ------- pd.DataFrame | None Predictions as a ``pd.DataFrame``. Returns ``None`` when ``wait`` is False. SageMaker API ------------- * :sm-api:`CreateTrainingJob`: trains the predictor and predicts in the same job on ``instance_count`` x ``instance_type``. Predictions are written to ``predictions_path``. """ extra_ag_args = {"predict_after_fit": True} if predictions_path is not None: extra_ag_args["predictions_path"] = predictions_path self.fit( train_data=train_data, known_covariates=known_covariates, static_features=static_features, predictor_init_args=predictor_init_args, predictor_fit_args=predictor_fit_args, id_column=id_column, timestamp_column=timestamp_column, framework_version=framework_version, job_name=job_name, instance_type=instance_type, instance_count=instance_count, volume_size=volume_size, custom_image_uri=custom_image_uri, wait=wait, backend_overrides=backend_overrides, extra_ag_args=extra_ag_args, ) if not wait: logger.info( "fit_predict job launched asynchronously. Use `get_fit_job_status()` " "to poll, then `get_fit_predict_results()` to fetch predictions." ) return None return self.get_fit_predict_results()
[docs] def get_fit_predict_results(self) -> pd.DataFrame: """ Retrieve predictions produced by a completed ``fit_predict`` job. Returns ------- pd.DataFrame Predictions for the forecast horizon. """ return self.backend.get_fit_predict_results()