Code import - branch 8
This commit is contained in:
234
models/xgboost/xgboost_data_model.py
Normal file
234
models/xgboost/xgboost_data_model.py
Normal file
@@ -0,0 +1,234 @@
|
||||
"""
|
||||
Standardized preprocessor for time series models used with XGBoost wrappers.
|
||||
|
||||
Args:
|
||||
- None
|
||||
|
||||
Return:
|
||||
None
|
||||
"""
|
||||
|
||||
import pandas as pd
|
||||
from sklearn.base import BaseEstimator, TransformerMixin
|
||||
from sklearn.preprocessing import MinMaxScaler, StandardScaler
|
||||
|
||||
|
||||
class TimeSeriesPreprocessor(BaseEstimator, TransformerMixin):
|
||||
def __init__(
|
||||
self,
|
||||
target: str | None = None,
|
||||
tags_list: list[str] | None = None,
|
||||
scaler_method: str | None = None,
|
||||
window_size: int = 0,
|
||||
use_filtering: bool = False,
|
||||
transform_mode: str = 'all',
|
||||
arpr_config: dict[str, str] | None = None,
|
||||
) -> None:
|
||||
"""
|
||||
Initialize the time series preprocessor for feature engineering and scaling.
|
||||
|
||||
Args:
|
||||
- target (str | None): Name of the target column. If None, it must be
|
||||
set before calling `fit`.
|
||||
- tags_list (list[str] | None): Explicit list of feature columns. If
|
||||
None, inferred during `fit` by excluding target (and CI column when
|
||||
AR/PR filtering is enabled).
|
||||
- scaler_method (str | None): Scaling strategy, either `'MinMax'`,
|
||||
`'Standard'` or None for no scaling.
|
||||
- window_size (int): Rolling window size used to generate lagged
|
||||
statistics for the target. Zero disables rolling features.
|
||||
- use_filtering (bool): Whether to filter rows based on a CI column
|
||||
during `fit` when `arpr_config` is provided.
|
||||
- transform_mode (str): Prediction mode, `'all'` to keep all rows or
|
||||
`'latest'` to return only the last row.
|
||||
- arpr_config (dict[str, str] | None): Optional configuration mapping
|
||||
`'ci_col'`, `'ar_col'` and `'pr_col'` names used for AR/PR substitution.
|
||||
|
||||
Return:
|
||||
None
|
||||
"""
|
||||
self.target = target
|
||||
self.tags_list = tags_list
|
||||
self.scaler_method = scaler_method
|
||||
self.window_size = window_size
|
||||
self.use_filtering = use_filtering
|
||||
self.transform_mode = transform_mode
|
||||
|
||||
if arpr_config is None:
|
||||
arpr_config = {}
|
||||
|
||||
self.ci_col = arpr_config.get('ci_col')
|
||||
self.ar_col = arpr_config.get('ar_col')
|
||||
self.pr_col = arpr_config.get('pr_col')
|
||||
|
||||
if self.ci_col is None or self.ar_col is None or self.pr_col is None:
|
||||
self.ar_filter = False
|
||||
else:
|
||||
self.ar_filter = True
|
||||
|
||||
# Internal state
|
||||
self.scaler = None
|
||||
|
||||
def _adjust_time_index(self, x: pd.DataFrame) -> pd.DataFrame:
|
||||
"""
|
||||
Standardize the time index so it is compatible with XGBoost expectations.
|
||||
|
||||
Args:
|
||||
- x (pd.DataFrame): Input DataFrame that may contain a time column.
|
||||
|
||||
Return:
|
||||
pd.DataFrame: Copy of the input with the index set to `Timestamp` or
|
||||
`timestamp` when present; otherwise the original index is preserved.
|
||||
"""
|
||||
df = x.copy()
|
||||
time_cols = ['Timestamp', 'timestamp']
|
||||
for col in time_cols:
|
||||
if col in df.columns:
|
||||
df[col] = pd.to_datetime(df[col])
|
||||
df = df.set_index(col)
|
||||
break
|
||||
return df
|
||||
|
||||
def _generate_rolling_features(self, df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""
|
||||
Generate rolling statistics for the target column.
|
||||
|
||||
The method computes rolling mean, max, min and standard deviation for
|
||||
the configured target when `window_size` is greater than zero.
|
||||
|
||||
Args:
|
||||
- df (pd.DataFrame): Input DataFrame indexed by time and containing
|
||||
the target column.
|
||||
|
||||
Return:
|
||||
pd.DataFrame: DataFrame with new rolling feature columns added. When
|
||||
the window is larger than the number of rows, the result may be empty
|
||||
after `dropna`.
|
||||
"""
|
||||
|
||||
if self.target is None:
|
||||
raise ValueError('target is not set. Set it before calling fit.')
|
||||
|
||||
if self.window_size > 0:
|
||||
df = df.sort_index()
|
||||
if df.shape[0] < self.window_size:
|
||||
# If we don't have enough data for the window, we might have issues
|
||||
# but we'll try to calculate what we can.
|
||||
pass
|
||||
|
||||
df['rolling_mean'] = df[self.target].rolling(window=self.window_size).mean()
|
||||
df['rolling_max'] = df[self.target].rolling(window=self.window_size).max()
|
||||
df['rolling_min'] = df[self.target].rolling(window=self.window_size).min()
|
||||
df['rolling_std'] = df[self.target].rolling(window=self.window_size).std()
|
||||
df = df.dropna()
|
||||
return df
|
||||
|
||||
def _apply_ar_substitution(self, df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""
|
||||
Apply AR/PR substitution based on the configured CI (confidence) column.
|
||||
|
||||
When AR/PR filtering is enabled and the CI column is present, rows
|
||||
where `ci_col == 0` have the AR column replaced by the PR column.
|
||||
|
||||
Args:
|
||||
- df (pd.DataFrame): Input DataFrame containing AR, PR and CI columns.
|
||||
|
||||
Return:
|
||||
pd.DataFrame: DataFrame with AR values updated where CI indicates PR
|
||||
substitution, or unchanged when filtering is disabled or misconfigured.
|
||||
"""
|
||||
if self.ar_filter:
|
||||
ar_col = self.ar_col
|
||||
pr_col = self.pr_col
|
||||
ci_col = self.ci_col
|
||||
|
||||
if all([ar_col, pr_col, ci_col]) and ci_col in df.columns:
|
||||
mask = df[ci_col] == 0
|
||||
if mask.any():
|
||||
df.loc[mask, ar_col] = df.loc[mask, pr_col]
|
||||
return df
|
||||
|
||||
def fit(self, x: pd.DataFrame) -> None:
|
||||
"""
|
||||
Fit the preprocessor on the full dataset.
|
||||
|
||||
This step may filter rows, generate rolling features, infer `tags_list`
|
||||
when it is not provided, and fit the underlying scaler if configured.
|
||||
|
||||
Args:
|
||||
- x (pd.DataFrame): Full dataset containing the target and feature
|
||||
columns, and optionally CI/AR/PR columns.
|
||||
|
||||
Return:
|
||||
None
|
||||
"""
|
||||
if self.target is None:
|
||||
raise ValueError('target is not set. Set it before calling fit.')
|
||||
|
||||
df = self._adjust_time_index(x)
|
||||
|
||||
# Optional filtering during fit
|
||||
if self.use_filtering and self.ar_filter:
|
||||
ci_col = self.ci_col
|
||||
if ci_col in df.columns:
|
||||
df = df[df[ci_col] == 1]
|
||||
|
||||
# Generate rolling features if needed
|
||||
df = self._generate_rolling_features(df)
|
||||
|
||||
# Infer tags_list if not provided
|
||||
if self.tags_list is None:
|
||||
exclude_cols = [self.target]
|
||||
if self.ar_filter and self.ci_col is not None:
|
||||
exclude_cols.append(self.ci_col)
|
||||
self.tags_list = [c for c in df.columns if c not in exclude_cols]
|
||||
|
||||
# Fit scaler
|
||||
if self.scaler_method == 'MinMax':
|
||||
self.scaler = MinMaxScaler()
|
||||
elif self.scaler_method == 'Standard':
|
||||
self.scaler = StandardScaler()
|
||||
|
||||
if self.scaler:
|
||||
# Only fit on tags_list_
|
||||
self.scaler.fit(df[self.tags_list])
|
||||
|
||||
def transform(self, x: pd.DataFrame) -> pd.DataFrame:
|
||||
"""
|
||||
Transform input data using the fitted preprocessor.
|
||||
|
||||
The transformation applies the same steps used during `fit` (time
|
||||
indexing, AR/PR substitution and rolling features), then selects the
|
||||
feature columns and optionally applies scaling and latest-row selection.
|
||||
|
||||
Args:
|
||||
- x (pd.DataFrame): Input data to transform.
|
||||
|
||||
Return:
|
||||
pd.DataFrame: Transformed feature matrix ready to be passed to the
|
||||
model wrapper.
|
||||
"""
|
||||
if self.tags_list is None:
|
||||
raise ValueError('tags_list is not set. Call fit first.')
|
||||
|
||||
df = self._adjust_time_index(x)
|
||||
|
||||
# Apply AR/PR substitution if configured
|
||||
df = self._apply_ar_substitution(df)
|
||||
|
||||
# Generate rolling features
|
||||
df = self._generate_rolling_features(df)
|
||||
|
||||
# Select features
|
||||
df = df[self.tags_list]
|
||||
|
||||
# Apply scaling
|
||||
if self.scaler:
|
||||
data_scaled = self.scaler.transform(df)
|
||||
df = pd.DataFrame(data=data_scaled, columns=self.tags_list, index=df.index)
|
||||
|
||||
# Transform mode: only latest sample
|
||||
if self.transform_mode == 'latest':
|
||||
df = df.tail(1)
|
||||
|
||||
return df
|
||||
Reference in New Issue
Block a user