diff --git a/benchmark_utils/download_hf.py b/benchmark_utils/download_hf.py new file mode 100644 index 0000000..e3c81eb --- /dev/null +++ b/benchmark_utils/download_hf.py @@ -0,0 +1,24 @@ +"""Hugging Face Hub download helper shared by the HF-backed datasets.""" + +from pathlib import Path + +from huggingface_hub import snapshot_download + + +def snapshot_hf_files(repo_id: str, subdir: str, pattern: str) -> "list[str]": + """Download ``/`` files from a HF dataset repo and + return their sorted local paths. + + Tries the local HF cache first (``local_files_only``) so cached runs + skip the Hub round-trip and work offline; falls back to a network + snapshot when the cache misses or lacks the requested files. + """ + kwargs = dict(repo_type="dataset", allow_patterns=f"{subdir}/{pattern}") + try: + root = snapshot_download(repo_id, local_files_only=True, **kwargs) + if not any((Path(root) / subdir).glob(pattern)): + # Cached snapshot lacks these files — handled below. + raise FileNotFoundError + except FileNotFoundError: + root = snapshot_download(repo_id, **kwargs) + return sorted(str(p) for p in (Path(root) / subdir).glob(pattern)) diff --git a/benchmark_utils/download.py b/benchmark_utils/download_pooch.py similarity index 100% rename from benchmark_utils/download.py rename to benchmark_utils/download_pooch.py diff --git a/benchmark_utils/forecasting_constants.py b/benchmark_utils/forecasting_constants.py new file mode 100644 index 0000000..de64412 --- /dev/null +++ b/benchmark_utils/forecasting_constants.py @@ -0,0 +1,162 @@ +"""Shared forecasting constants: frequency / seasonality tables and metrics. + +Two sources name frequencies differently: + - aeon (used by Monash) uses words: "yearly", "weekly", "minutely", ... + - GIFT-Eval (and pandas) use offset aliases: "Y", "W-SUN", "5T", ... + +This module exposes a single canonical (freq, seasonality) lookup keyed on +the canonical pandas-style base alias (e.g. "Y", "W", "D"), plus two +adapters that normalize each source onto that canonical key. +""" + +import re + +# Metrics reported by every forecasting dataset (names from +# benchmark_utils.metrics.ALL_METRICS). +FORECASTING_METRICS = ( + "mae", "mse", "rmse", "mase", "smape", + "crps", "wql", "mcis", "pinball", "skill_score_ratio", +) + +# Canonical base alias → (display_freq, MASE seasonality, default forecast horizon) +_BASE = { + "Y": ("Y", 1, 6), + "Q": ("Q", 4, 8), + "M": ("M", 12, 12), + "W": ("W", 52, 13), + "D": ("D", 7, 14), + "H": ("H", 24, 24), + "T": ("T", 1440, 60), # minutes + "S": ("S", 1, 60), +} + +# aeon's spelled-out names → pandas offset alias. Sub-hourly words map to +# multiplied aliases so the seasonality accounts for the step size. +_AEON_TO_ALIAS = { + "yearly": "Y", + "quarterly": "Q", + "monthly": "M", + "weekly": "W", + "daily": "D", + "hourly": "H", + "half_hourly": "30T", + "minutely": "T", + "10_minutes": "10T", + "seconds": "S", + "4_seconds": "4S", +} + + +def from_aeon(freq_word: str) -> tuple[str, int, int]: + """Look up (freq, seasonality, default_horizon) from an aeon freq word. + + Unknown words default to daily. + """ + return from_pandas(_AEON_TO_ALIAS.get(freq_word, "D")) + + +# Pandas offset aliases: capture the leading multiplier and the unit, +# ignoring any anchor suffix (e.g. "5T" → (5, "T"), "W-SUN" → (1, "W")). +_PANDAS_ALIAS_RE = re.compile(r"^(\d*)([A-Za-z]+)") +_NORMALIZE_BASE = { + # Newer pandas spellings → legacy single-letter aliases used in _BASE. + "YE": "Y", "YS": "Y", "A": "Y", "AS": "Y", + "QE": "Q", "QS": "Q", + "ME": "M", "MS": "M", + "min": "T", "MIN": "T", +} + + +def from_pandas(freq_alias: str) -> tuple[str, int, int]: + """Look up (freq, seasonality, default_horizon) from a pandas freq alias. + + Anchors ("W-SUN", "QS-OCT") are stripped before lookup. A multiplier + scales the step size, so the seasonality is divided by it: at "15T" + one day is 1440/15 = 96 steps, not 1440. The original alias is + returned as freq so calendar-building consumers (``pd.date_range``) + keep the true sampling rate. Unknown aliases default to daily. + """ + if not freq_alias: + return _BASE["D"] + m = _PANDAS_ALIAS_RE.match(freq_alias.split("-", 1)[0]) + if not m: + return _BASE["D"] + mult = int(m.group(1)) if m.group(1) else 1 + base = _NORMALIZE_BASE.get(m.group(2), m.group(2)[:1].upper()) + if base not in _BASE: + return _BASE["D"] + _, seasonality, default_h = _BASE[base] + return freq_alias, max(1, seasonality // max(mult, 1)), default_h + + +# --------------------------------------------------------------------------- +# GIFT-Eval term resolution +# +# Mirrors the canonical table in the upstream time-series repo: prediction +# length is a function of pandas freq, then scaled by a term multiplier +# (short=1, medium=10, long=15). Used by datasets/gifteval.py so reported +# numbers line up with the GIFT-Eval leaderboard. +# --------------------------------------------------------------------------- + +GIFT_EVAL_PRED_LENGTH_MAP: dict[str, int] = { + "M": 12, "MS": 12, + "W": 8, "W-SUN": 8, "W-MON": 8, + "D": 30, + "H": 48, "6H": 48, + "T": 48, "5T": 48, "10T": 48, "15T": 48, "30T": 48, + "S": 60, "4S": 60, + "Q": 8, "Q-DEC": 8, + "A": 4, "A-DEC": 4, + "Y": 4, +} + +# M4-competition horizons differ from the generic table; upstream +# gift-eval selects this map whenever "m4" is in the dataset name. +M4_PRED_LENGTH_MAP: dict[str, int] = { + "A": 6, "Y": 6, + "Q": 8, + "M": 18, + "W": 13, + "D": 14, + "H": 48, +} + +GIFT_EVAL_TERM_MULTIPLIER: dict[str, int] = { + "short": 1, + "medium": 10, + "long": 15, +} + + +def gift_eval_prediction_length( + freq: str, term: str, dataset_name: str = "" +) -> int: + """Resolve the GIFT-Eval prediction length for a (freq, term) pair. + + ``freq`` is a pandas-style alias (e.g. ``"5T"``, ``"1H"``, ``"W-SUN"``). + Lookup falls back through: exact match → strip leading "1" multiplier + ("1H" → "H") → collapse any multi-X alias to its base X ("10S" → "S", + "30T" → "T") → default 48. ``term`` must be one of ``"short"``, + ``"medium"``, ``"long"``. When ``dataset_name`` contains "m4", the + M4-competition horizons are used, mirroring upstream gift-eval. + """ + if term not in GIFT_EVAL_TERM_MULTIPLIER: + raise ValueError( + f"term must be one of {list(GIFT_EVAL_TERM_MULTIPLIER)}; got {term!r}" + ) + pred_length_map = ( + M4_PRED_LENGTH_MAP if "m4" in dataset_name + else GIFT_EVAL_PRED_LENGTH_MAP + ) + base = pred_length_map.get(freq) + if base is None: + m = _PANDAS_ALIAS_RE.match(freq.split("-", 1)[0]) + if m: + head = m.group(2) + # Normalize new pandas spellings ("QE"→"Q", "ME"→"M", ...) + # before falling back through the map. + head = _NORMALIZE_BASE.get(head, head) + base = pred_length_map.get(head) + if base is None: + base = 48 + return base * GIFT_EVAL_TERM_MULTIPLIER[term] diff --git a/benchmark_utils/windowing.py b/benchmark_utils/windowing.py index e3bf3db..6ceff56 100644 --- a/benchmark_utils/windowing.py +++ b/benchmark_utils/windowing.py @@ -67,3 +67,50 @@ def make_forecasting_splits( targets.append(np.stack(ys, axis=0)) # (n_cutoffs, H, C) return series_full, cutoff_indexes, targets + + +def build_forecasting_data( + series: List[np.ndarray], + prediction_length: int, + n_windows: int = 1, + debug: bool = False, +) -> dict: + """Build the shared train/test split fields of a forecasting data dict. + + Keeps everything but the last ``prediction_length * n_windows`` steps + of each series as training context, then delegates the evaluation + windows to :func:`make_forecasting_splits` (a single window in debug + mode). Series shorter than ``prediction_length + 1`` are dropped. + + ``y_train`` is ``None``: forecasting is self-supervised, so solvers + that fine-tune carve their own (context, target) windows out of + ``X_train`` — handing out a fixed pair would either leak the test + windows or duplicate a slice of ``X_train``. + + Returns a dict with ``X_train``, ``y_train``, ``X_test``, ``y_test`` + and ``cutoff_indexes``; datasets add their task-specific fields + (metrics, freq, seasonality, ...) on top. + """ + test_len = prediction_length * n_windows + X_train, full_series = [], [] + for ts in series: + if ts.shape[0] < prediction_length + 1: + continue + X_train.append(ts[:max(1, ts.shape[0] - test_len)]) + full_series.append(ts) + + if not full_series: + raise ValueError("All series are shorter than prediction_length.") + + X_test, cutoff_indexes, y_test = make_forecasting_splits( + full_series, + prediction_length=prediction_length, + n_windows=1 if debug else n_windows, + ) + return dict( + X_train=X_train, + y_train=None, + X_test=X_test, + y_test=y_test, + cutoff_indexes=cutoff_indexes, + ) diff --git a/datasets/ecg.py b/datasets/ecg.py index 42560e3..45e595c 100644 --- a/datasets/ecg.py +++ b/datasets/ecg.py @@ -20,7 +20,7 @@ import pandas as pd from benchopt import BaseDataset -from benchmark_utils.download import fetch_tsb_uad, load_data_tsb_uad +from benchmark_utils.download_pooch import fetch_tsb_uad, load_data_tsb_uad from benchmark_utils.metrics import AD_METRICS diff --git a/datasets/fev.py b/datasets/fev.py new file mode 100644 index 0000000..6875ea2 --- /dev/null +++ b/datasets/fev.py @@ -0,0 +1,274 @@ +"""AutoGluon fev_datasets forecasting benchmark +(huggingface.co/datasets/autogluon/fev_datasets). + +The HF repo organizes data either: + - per-freq: ``//train-*.parquet`` + (e.g. ``ETT/1H``, ``LOOP_SEATTLE/5T``) + - flat: ``/train-*.parquet`` + (e.g. ``australian_tourism``) + - or with an arbitrary subdir that is NOT a freq (e.g. ``boomlet/`` + where ```` is a series id, not a frequency). + +We accept the directory path directly as ``dataset_name`` (e.g. +``"ETT/1H"``, ``"australian_tourism"``) and infer the actual freq from +each series' ``timestamp`` column rather than parsing the path. + +Each parquet row is one series; columns vary: + - Always: ``id``, ``timestamp`` + - Univariate: a ``target`` column (list of floats) + - Multivariate (e.g. ``ETT``): no ``target`` column — each channel is + its own column (``HUFL``, ..., ``OT``). Channel columns are stacked + on the last axis to form ``(T, C)``. + +Rolling-window splits match :mod:`datasets.monash`. The default +``prediction_length`` is the freq-based heuristic from +:func:`benchmark_utils.forecasting_constants.from_pandas`; FEV does not publish a +per-dataset horizon spec, so we don't try to mirror one. Pass +``prediction_length=N`` explicitly to override. +""" + +import warnings + +import numpy as np +import pandas as pd +from benchopt import BaseDataset + +from benchmark_utils.covariates import Covariates +from benchmark_utils.download_hf import snapshot_hf_files +from benchmark_utils.forecasting_constants import FORECASTING_METRICS, from_pandas +from benchmark_utils.windowing import build_forecasting_data + + +_METADATA_COLS = ("id", "timestamp") + + +# Canonical list of FEV evaluation configs — directory paths inside +# https://huggingface.co/datasets/autogluon/fev_datasets that contain at +# least one ``train-*.parquet`` file. Surfaced via +# ``get_parameter_choices`` so that ``dataset_name=all`` and ``benchopt +# info -v`` work. +FEV_DATASETS: tuple[str, ...] = ( + "ETT/15T", "ETT/1D", "ETT/1H", "ETT/1W", + "LOOP_SEATTLE/1D", "LOOP_SEATTLE/1H", "LOOP_SEATTLE/5T", + "M_DENSE/1D", "M_DENSE/1H", + "SZ_TAXI/15T", "SZ_TAXI/1H", + "australian_tourism", + "bizitobs_l2c/1H", "bizitobs_l2c/5T", + "boomlet/1062", "boomlet/1209", "boomlet/1225", "boomlet/1230", + "boomlet/1282", "boomlet/1487", "boomlet/1631", "boomlet/1676", + "boomlet/1855", "boomlet/1975", "boomlet/2187", + "boomlet/285", "boomlet/619", "boomlet/772", "boomlet/963", + "ecdc_ili", + "entsoe/15T", "entsoe/1H", "entsoe/30T", + "epf_be", "epf_de", "epf_fr", "epf_np", "epf_pjm", + "ercot/1D", "ercot/1H", "ercot/1M", "ercot/1W", + "favorita_stores/1D", "favorita_stores/1M", "favorita_stores/1W", + "favorita_transactions/1D", "favorita_transactions/1M", + "favorita_transactions/1W", + "fred_md_2025", "fred_qd_2025", + "gvar", "hermes", + "hierarchical_sales/1D", "hierarchical_sales/1W", + "hospital", + "hospital_admissions/1D", "hospital_admissions/1W", + "jena_weather/10T", "jena_weather/1D", "jena_weather/1H", + "kdd_cup_2022/10T", "kdd_cup_2022/1D", "kdd_cup_2022/30T", + "m5/1D", "m5/1M", "m5/1W", + "proenfo_bull", "proenfo_cockatoo", + "proenfo_gfc12", "proenfo_gfc14", "proenfo_gfc17", + "proenfo_hog", "proenfo_pdb", + "redset/15T", "redset/1H", "redset/5T", + "restaurant", + "rohlik_orders/1D", "rohlik_orders/1W", + "rohlik_sales/1D", "rohlik_sales/1W", + "rossmann/1D", "rossmann/1W", + "solar/1D", "solar/1W", + "solar_with_weather/15T", "solar_with_weather/1H", + "uci_air_quality/1D", "uci_air_quality/1H", + "uk_covid_nation/1D", "uk_covid_nation/1W", + "uk_covid_utla/1D", "uk_covid_utla/1W", + "us_consumption/1M", "us_consumption/1Q", "us_consumption/1Y", + "walmart", + "world_co2_emissions", "world_life_expectancy", "world_tourism", +) + + +def _infer_freq(timestamps) -> str: + """Best-effort freq inference from a series' timestamp column. + + Tries ``pd.infer_freq`` on the first points, then the most common + consecutive delta (robust to isolated gaps). Warns and falls back to + ``"D"`` when nothing can be inferred. + """ + try: + idx = pd.DatetimeIndex(timestamps[:20]) + freq = pd.infer_freq(idx[:5]) if len(idx) >= 3 else None + if freq: + return freq + deltas = pd.Series(idx).diff().dropna() + if len(deltas): + step = deltas.mode().iloc[0] + return pd.tseries.frequencies.to_offset(step).freqstr + except Exception: + pass + warnings.warn( + "Could not infer the sampling frequency from timestamps; " + "defaulting to daily ('D')." + ) + return "D" + + +def _is_numeric_array_col(df, c) -> bool: + """True if column ``c`` holds non-empty numeric array-likes. + + Decided from the first informative entry among the first 5 rows, so + a null or empty first row does not drop a valid channel. + """ + for v in df[c].head(5): + if v is None or isinstance(v, (str, bytes)) or not hasattr(v, "__len__"): + continue + if len(v) == 0: + continue + return isinstance(v[0], (int, float, np.integer, np.floating)) + return False + + +def _select_channel_cols(df) -> "list[str]": + """Columns to stack as forecast channels. + + An explicit ``target`` column is the sole target — other numeric + array columns are exogenous covariates (e.g. the epf_* load + forecasts), out of scope for the MVP. Without ``target`` (e.g. ETT), + every numeric array column is a channel. + """ + if "target" in df.columns: + return ["target"] + return [ + c for c in df.columns + if c not in _METADATA_COLS and _is_numeric_array_col(df, c) + ] + + +class Dataset(BaseDataset): + """AutoGluon fev forecasting dataset. + + Parameters + ---------- + dataset_name : str + Directory path inside the HF repo. Per-freq paths look like + ``"ETT/1H"`` / ``"LOOP_SEATTLE/5T"``; flat paths like + ``"australian_tourism"`` / ``"hospital"``. See ``FEV_DATASETS`` + for the full list (also discoverable via ``benchopt info -v``). + prediction_length : int or None + Explicit override. ``None`` → resolved from the inferred freq + via :func:`benchmark_utils.forecasting_constants.from_pandas` (same heuristic + used by Monash). FEV does not publish its own per-dataset + horizon matrix, so we don't try to align with a leaderboard + spec here. + n_windows : int + Number of rolling evaluation windows per series. + max_series : int or None + Optional cap on the number of series. + debug : bool + If True, keep only the first 5 series. + """ + + name = "FEV" + + requirements = ["pip::huggingface-hub"] + + parameters = { + "dataset_name": ["LOOP_SEATTLE/1H"], + "prediction_length": [None], + "n_windows": [1], + "max_series": [None], + "debug": [False], + } + + # Cache prepare() by dataset_name only — the other knobs shape the + # in-memory view, not the downloaded files. + prepare_cache_ignore = ( + "prediction_length", "n_windows", "max_series", "debug", + ) + + @classmethod + def get_all_parameter_values(cls, name): + if name == "dataset_name": + return list(FEV_DATASETS) + return None + + def prepare(self): + """Pre-download parquet shards for this config into HF's cache.""" + self._snapshot() + + def _snapshot(self) -> "list[str]": + """Snapshot-download parquet files for this dataset_name and + return their local paths (cache-first, see ``snapshot_hf_files``).""" + return snapshot_hf_files( + "autogluon/fev_datasets", self.dataset_name, "*.parquet" + ) + + def get_data(self): + parquet_files = self._snapshot() + if not parquet_files: + raise ValueError( + f"No parquet found at {self.dataset_name!r} in " + "autogluon/fev_datasets. Valid choices are in FEV_DATASETS." + ) + + df = pd.concat( + [pd.read_parquet(f) for f in parquet_files], + ignore_index=True, + ) + + if self.debug: + df = df.head(5) + elif self.max_series is not None: + df = df.head(int(self.max_series)) + + if df.empty: + raise ValueError(f"{self.dataset_name!r} contained 0 series.") + + channel_cols = _select_channel_cols(df) + if not channel_cols: + raise ValueError( + f"{self.dataset_name!r} has no channel columns " + f"(only {_METADATA_COLS} present)." + ) + + # Infer freq from the first series' timestamps — same for the + # whole config (FEV groups by freq at the directory level for + # nested configs, and flat configs are single-freq). + inferred_freq = _infer_freq(df.iloc[0]["timestamp"]) + canonical_freq, seasonality, default_h = from_pandas(inferred_freq) + + pred_len = self.prediction_length + if pred_len is None: + pred_len = int(default_h) + + # Build (T, C) series. Each row's per-channel array has the same + # length (T_i); stack on the last axis. + series_list = [] + for _, row in df.iterrows(): + channels = [np.asarray(row[c], dtype=np.float32) for c in channel_cols] + T = channels[0].shape[0] + if any(ch.shape[0] != T for ch in channels): + continue + series_list.append(np.stack(channels, axis=-1)) + + if not series_list: + raise ValueError("All series were skipped (inconsistent channel lengths).") + + return dict( + **build_forecasting_data( + series_list, + prediction_length=pred_len, + n_windows=self.n_windows, + debug=self.debug, + ), + covariates=Covariates(), + task="forecasting", + metrics=list(FORECASTING_METRICS), + prediction_length=pred_len, + freq=canonical_freq, + seasonality=seasonality, + ) diff --git a/datasets/gifteval.py b/datasets/gifteval.py new file mode 100644 index 0000000..67cf71e --- /dev/null +++ b/datasets/gifteval.py @@ -0,0 +1,351 @@ +"""GIFT-Eval forecasting benchmark dataset (Salesforce/GiftEval on HF). + +Parametrization +--------------- +The class exposes two orthogonal parameters that drive the leaderboard +matrix: + +* ``dataset_name`` — one of 55 canonical ``/`` paths (e.g. + ``"m4_weekly/W"``, ``"loop_seattle/H"``). The full list is derived + from the leaderboard CSV and discoverable via ``benchopt info -v``. +* ``term`` — one of ``short`` / ``medium`` / ``long``, controlling the + forecast horizon (×1, ×10, ×15 of the per-freq base). + +Both are surfaced via ``get_all_parameter_values`` so that +``-d "GiftEval[dataset_name=all,term=short]"`` and ``benchopt info -v`` +work. + +Canonical-combo gating +---------------------- +GIFT-Eval scores only **97** of the 55 × 3 = 165 possible ``(path, +term)`` combinations on its public leaderboard. The 34 short-only paths +do not define ``medium`` / ``long``. The canonical set is derived +from the leaderboard's results CSV (fetched from the HF Space +``Salesforce/GIFT-Eval`` by ``prepare`` or on first use, and stored as a +small CSV under benchopt's data path) and gates runs at the dataset +level: when +``(dataset_name, term)`` is not canonical, ``get_data()`` short-circuits +and returns a placeholder dict carrying a ``_skip_reason`` field. +:meth:`Objective.skip` (see ``objective.py``) honors that field and +skips the combo cleanly. + +So: + +* ``-d "GiftEval[dataset_name=all,term=short]"`` → 55 canonical runs. +* ``-d "GiftEval[dataset_name=all,term=long]"`` → 55 attempts, + 21 canonical runs, 34 skipped. +* ``-d "GiftEval[dataset_name=all,term=all]"`` → 165 attempts, + 97 canonical runs, 68 skipped. + +Leaderboard names vs HF directory names +--------------------------------------- +The leaderboard uses lowercase, paper-style identifiers (e.g. +``loop_seattle/H``, ``m_dense/D``, ``car_parts/M``) while the HF repo +``Salesforce/GiftEval`` uses mixed-case directory names that don't +always match (``LOOP_SEATTLE/H``, ``M_DENSE/D``, +``car_parts_with_missing/``). We accept leaderboard names — that's what +appears in the paper, the leaderboard, and the gift-eval README — and +translate to HF paths internally via :data:`_LEADERBOARD_TO_HF`. Cases: + + * Pure case difference: ``loop_seattle`` → ``LOOP_SEATTLE``, + ``m_dense`` → ``M_DENSE``, ``sz_taxi`` → ``SZ_TAXI``. + * Missing-data suffix: ``car_parts`` → ``car_parts_with_missing``, + ``kdd_cup_2018`` → ``kdd_cup_2018_with_missing``, + ``temperature_rain`` → ``temperature_rain_with_missing``. + * Rename: ``saugeen`` → ``saugeenday``. + * Leaderboard adds a freq segment for HF-flat datasets: leaderboard + ``m4_yearly/A`` → HF flat ``m4_yearly`` (the freq is implicit in the + data, not the path). Likewise for the other ``m4_*``, + ``car_parts/M``, ``covid_deaths/D``, ``hospital/M``, + ``restaurant/D``, ``temperature_rain/D``, + ``bizitobs_application/10S``, ``bizitobs_service/10S``. + +Schema +------ +Each HF entry exposes ``item_id``, ``start``, ``freq``, ``target``. +``target`` is a flat ``List[float]`` for univariate configs and a +``List[List[float]]`` of shape ``(C, T)`` for multivariate ones (e.g. +``bitbrains_*``, ``electricity/*``, ``ett1/*``, ``ett2/*``, +``jena_weather/*``, ``solar/*``). Both shapes are handled — multivariate +entries are transposed to the repo's ``(T, C)`` contract. + +Cutoffs and windows +------------------- +We don't comply with GIFT-Eval's prescribed test cutoff; we use the same +rolling-window logic as Monash via +:func:`benchmark_utils.windowing.make_forecasting_splits`. The +``prediction_length`` for a given (freq, term) follows GIFT-Eval's +canonical ``base × multiplier`` rule via +:func:`benchmark_utils.forecasting_constants.gift_eval_prediction_length`. + +Data contract output mirrors :mod:`datasets.monash`. +""" + +import csv +from pathlib import Path + +import numpy as np +from benchopt import BaseDataset, config + +from benchmark_utils.covariates import Covariates +from benchmark_utils.download_hf import snapshot_hf_files +from benchmark_utils.forecasting_constants import ( + FORECASTING_METRICS, + from_pandas, + gift_eval_prediction_length, +) +from benchmark_utils.windowing import build_forecasting_data + + +# --------------------------------------------------------------------------- +# Canonical (dataset_name, term) table — derived from the GIFT-Eval +# leaderboard Space (results/seasonal_naive/all_results.csv: 55 paths, +# 97 combos, 34 short-only). Fetched once, stored as a small CSV under +# benchopt's data path by ``prepare`` (or on first use, e.g. when +# expanding ``dataset_name=all``). +# --------------------------------------------------------------------------- +_LEADERBOARD_REPO = "Salesforce/GIFT-Eval" +_LEADERBOARD_FILE = "results/seasonal_naive/all_results.csv" +_leaderboard_cache: "dict[str, tuple[str, ...]] | None" = None + +GIFTEVAL_TERMS: tuple[str, ...] = ("short", "medium", "long") + + +def _leaderboard_csv_path() -> Path: + return Path(config.get_data_path(key="gifteval")) / "leaderboard_combos.csv" + + +def _build_leaderboard_csv(path: Path): + """Fetch the leaderboard results CSV and store the (dataset_name, term) + combos it scores.""" + from huggingface_hub import hf_hub_download + + src = hf_hub_download(_LEADERBOARD_REPO, _LEADERBOARD_FILE, repo_type="space") + with open(src) as f: + # "dataset" column is "//". + combos = [row["dataset"].rsplit("/", 1) for row in csv.DictReader(f)] + path.parent.mkdir(parents=True, exist_ok=True) + with open(path, "w", newline="") as f: + writer = csv.writer(f) + writer.writerow(["dataset_name", "term"]) + writer.writerows(combos) + + +def _parse_leaderboard_csv(path: Path) -> "dict[str, tuple[str, ...]]": + table: dict = {} + with open(path) as f: + for row in csv.DictReader(f): + table.setdefault(row["dataset_name"], []).append(row["term"]) + return {name: tuple(terms) for name, terms in table.items()} + + +def _leaderboard() -> "dict[str, tuple[str, ...]]": + """dataset_name → terms it defines on the leaderboard (disk-cached).""" + global _leaderboard_cache + if _leaderboard_cache is None: + csv_path = _leaderboard_csv_path() + if not csv_path.exists(): + _build_leaderboard_csv(csv_path) + _leaderboard_cache = _parse_leaderboard_csv(csv_path) + return _leaderboard_cache + + +# --------------------------------------------------------------------------- +# Leaderboard ```` → HF top-level directory name. Only entries that +# differ from the lowercase identity mapping appear here. +# --------------------------------------------------------------------------- +_LEADERBOARD_TO_HF: dict[str, str] = { + "loop_seattle": "LOOP_SEATTLE", + "m_dense": "M_DENSE", + "sz_taxi": "SZ_TAXI", + "car_parts": "car_parts_with_missing", + "kdd_cup_2018": "kdd_cup_2018_with_missing", + "temperature_rain": "temperature_rain_with_missing", + "saugeen": "saugeenday", +} + + +# --------------------------------------------------------------------------- +# Datasets that live as a single arrow file directly under the dataset +# name (no per-freq subdir on HF). The leaderboard still adds a freq +# segment to their paths (e.g. ``m4_yearly/A``, ``hospital/M``), which we +# strip before locating the file. +# --------------------------------------------------------------------------- +_HF_FLAT_DATASETS: frozenset[str] = frozenset({ + "bizitobs_application", "bizitobs_service", + "car_parts_with_missing", "covid_deaths", "hospital", + "m4_daily", "m4_hourly", "m4_monthly", "m4_quarterly", + "m4_weekly", "m4_yearly", + "restaurant", "temperature_rain_with_missing", +}) + + +def _hf_arrow_directory(leaderboard_path: str) -> str: + """Resolve a leaderboard ``/`` path to the actual HF + directory containing the arrow file. + + Examples + -------- + ``"m4_weekly/W"`` → ``"m4_weekly"`` (HF-flat, drops freq) + ``"loop_seattle/H"`` → ``"LOOP_SEATTLE/H"`` (case-renamed) + ``"car_parts/M"`` → ``"car_parts_with_missing"`` (HF-flat + suffix) + """ + leaderboard_name, _, freq_segment = leaderboard_path.partition("/") + hf_name = _LEADERBOARD_TO_HF.get(leaderboard_name, leaderboard_name) + if hf_name in _HF_FLAT_DATASETS: + return hf_name + if freq_segment: + return f"{hf_name}/{freq_segment}" + return hf_name + + +def _skip_placeholder(reason: str) -> dict: + """Flag the combo for skipping via ``Objective.skip``, which benchopt + calls before ``set_data`` — no other data field is needed.""" + return dict(_skip_reason=reason) + + +class Dataset(BaseDataset): + """GIFT-Eval forecasting dataset (loaded from HF Salesforce/GiftEval). + + Parameters + ---------- + dataset_name : str + One of 55 canonical leaderboard paths — ``/``, e.g. + ``"m4_weekly/W"``, ``"loop_seattle/H"``. See + ``benchopt info -v``. + term : str + ``"short"`` / ``"medium"`` / ``"long"``. Combos not on the + leaderboard are skipped (placeholder + objective + ``skip``), so ``dataset_name=all, term=long`` runs only the 21 + paths that define ``long``. + prediction_length : int or None + Explicit override. ``None`` → resolved from (freq, term) via + :func:`benchmark_utils.forecasting_constants.gift_eval_prediction_length`. + n_windows : int + Number of rolling evaluation windows per series. + max_series : int or None + Optional cap on the number of series. + debug : bool + If True, keep only the first 5 series for fast iteration. + """ + + name = "GiftEval" + + requirements = ["pip::datasets", "pip::huggingface-hub"] + + parameters = { + "dataset_name": ["m4_weekly/W"], + "term": ["short"], + "prediction_length": [None], + "n_windows": [1], + "max_series": [None], + "debug": [False], + } + + # ``prepare()`` depends on ``dataset_name`` only — ``term`` and the + # other knobs shape the in-memory view, not the downloaded files. + prepare_cache_ignore = ( + "term", "prediction_length", "n_windows", "max_series", "debug", + ) + + @classmethod + def get_all_parameter_values(cls, name): + if name == "dataset_name": + return sorted(_leaderboard()) + if name == "term": + return list(GIFTEVAL_TERMS) + return None + + def prepare(self): + """Build the leaderboard-combos CSV and pre-download the arrow + shards for this config into HF's cache.""" + _leaderboard() + self._snapshot() + + def _snapshot(self) -> "list[str]": + """Snapshot-download the arrow files for this dataset and return + their local paths (cache-first, see ``snapshot_hf_files``). + + data-*.arrow only: the hub repo also holds stray HF map-cache + shards (cache-*.arrow, e.g. under electricity/15T) with + duplicated rows that must not be loaded. + """ + return snapshot_hf_files( + "Salesforce/GiftEval", + _hf_arrow_directory(self.dataset_name), + "data-*.arrow", + ) + + def get_data(self): + # Short-circuit non-canonical combos so heavy parsing doesn't run. + if self.term not in _leaderboard().get(self.dataset_name, ()): + return _skip_placeholder( + f"non-canonical GIFT-Eval combo: {self.dataset_name!r} does " + f"not define term {self.term!r} on the leaderboard" + ) + + from datasets import Dataset as HFDataset, concatenate_datasets + + arrow_files = self._snapshot() + if not arrow_files: + raise ValueError( + f"No Arrow file found for GIFT-Eval dataset " + f"{self.dataset_name!r}. See `benchopt info` for valid choices." + ) + + parts = [HFDataset.from_file(f) for f in arrow_files] + ds = parts[0] if len(parts) == 1 else concatenate_datasets(parts) + + # Slice on the Arrow table (zero-copy) before decoding anything. + n_keep = 5 if self.debug else self.max_series + if n_keep is not None: + ds = ds.select(range(min(int(n_keep), len(ds)))) + + if len(ds) == 0: + raise ValueError( + f"GIFT-Eval dataset {self.dataset_name!r} returned 0 series." + ) + + # Frequency / seasonality — every series in a GIFT-Eval subset + # shares the same freq, so taking it from the first entry is safe. + pandas_freq = ds[0].get("freq") or "D" + freq, seasonality, _ = from_pandas(pandas_freq) + + pred_len = self.prediction_length + if pred_len is None: + pred_len = gift_eval_prediction_length( + pandas_freq, self.term, dataset_name=self.dataset_name + ) + + # Build (T, C) series with columnar numpy access — avoids + # round-tripping every float through python objects. Univariate + # entries arrive as flat arrays (ndim=1); multivariate as (C, T). + series_list = [] + for values in ds.with_format("numpy")["target"]: + values = np.asarray(values, dtype=np.float32) + if values.ndim == 1: + series_list.append(values.reshape(-1, 1)) # (T, 1) + elif values.ndim == 2: + series_list.append(values.T) # (C,T)→(T,C) + + if not series_list: + raise ValueError( + f"All entries in GIFT-Eval dataset {self.dataset_name!r} " + "had unsupported target shapes." + ) + + return dict( + **build_forecasting_data( + series_list, + prediction_length=pred_len, + n_windows=self.n_windows, + debug=self.debug, + ), + covariates=Covariates(), # GIFT-Eval HF schema has no covariates + task="forecasting", + metrics=list(FORECASTING_METRICS), + prediction_length=pred_len, + freq=freq, + seasonality=seasonality, + ) diff --git a/datasets/mitdb.py b/datasets/mitdb.py index fe00fb9..111fb4f 100644 --- a/datasets/mitdb.py +++ b/datasets/mitdb.py @@ -26,7 +26,7 @@ class 4 Q — Unknown / pacemaker artefact import numpy as np from benchopt import BaseDataset -from benchmark_utils.download import fetch_mitdb +from benchmark_utils.download_pooch import fetch_mitdb # AAMI beat-type grouping (MIT-BIH annotation symbol → class index) BEAT_CLASS = { diff --git a/datasets/monash.py b/datasets/monash.py index a607f64..fb21368 100644 --- a/datasets/monash.py +++ b/datasets/monash.py @@ -36,29 +36,8 @@ from benchopt import BaseDataset from benchmark_utils.covariates import Covariates -from benchmark_utils.windowing import make_forecasting_splits - -# Map aeon frequency strings → pandas-style freq codes and MASE seasonality -_FREQ_MAP = { - "yearly": ("Y", 1), - "quarterly": ("Q", 4), - "monthly": ("M", 12), - "weekly": ("W", 52), - "daily": ("D", 7), - "hourly": ("H", 24), - "minutely": ("T", 1440), - "seconds": ("S", 1), -} - -_DEFAULT_HORIZON = { - "Y": 6, - "Q": 8, - "M": 12, - "W": 13, - "D": 14, - "H": 24, - "T": 60, -} +from benchmark_utils.forecasting_constants import FORECASTING_METRICS, from_aeon +from benchmark_utils.windowing import build_forecasting_data class Dataset(BaseDataset): @@ -88,6 +67,20 @@ class Dataset(BaseDataset): "debug": [False], } + # Only dataset_name decides what aeon downloads; the other knobs + # affect the in-memory split, not the file on disk. + prepare_cache_ignore = ("prediction_length", "n_windows", "debug") + + def prepare(self): + """Warm aeon's local cache for this dataset (download if missing). + + aeon writes the ``.tsf`` to + ``~/.aeon/datasets/local_data//.tsf`` on first use; + we call it once and discard the parsed result so the cache layer + in :func:`load_forecasting` handles the actual download. + """ + load_forecasting(self.dataset_name, return_metadata=False) + def get_data(self): df, meta = load_forecasting(self.dataset_name, return_metadata=True) # df columns: series_name, start_timestamp, series_value @@ -95,13 +88,11 @@ def get_data(self): # contain_missing_values, contain_equal_length aeon_freq = meta.get("frequency", "yearly") - freq, seasonality = _FREQ_MAP.get(aeon_freq, ("D", 1)) + freq, seasonality, default_h = from_aeon(aeon_freq) pred_len = self.prediction_length if pred_len is None: - pred_len = int( - meta.get("forecast_horizon") or _DEFAULT_HORIZON.get(freq, 10) - ) + pred_len = int(meta.get("forecast_horizon") or default_h) series_list = [] rows = df.iterrows() if not self.debug else list(df.iterrows())[:5] @@ -112,47 +103,16 @@ def get_data(self): if not series_list: raise ValueError(f"No series found for dataset {self.dataset_name!r}.") - # Training portion: everything except the last test windows - test_len = pred_len * self.n_windows - X_train, y_train_list, full_series = [], [], [] - for ts in series_list: - if ts.shape[0] < pred_len + 1: - continue - train_end = max(1, ts.shape[0] - test_len) - X_train.append(ts[:train_end]) - y_train_list.append(ts[train_end : train_end + pred_len]) - full_series.append(ts) - - if not full_series: - raise ValueError("All series are shorter than prediction_length.") - - n_windows = 1 if self.debug else self.n_windows - X_test, cutoff_indexes, y_test = make_forecasting_splits( - full_series, - prediction_length=pred_len, - n_windows=n_windows, - ) - return dict( - X_train=X_train, - y_train=y_train_list, - X_test=X_test, - y_test=y_test, - cutoff_indexes=cutoff_indexes, + **build_forecasting_data( + series_list, + prediction_length=pred_len, + n_windows=self.n_windows, + debug=self.debug, + ), covariates=Covariates(), task="forecasting", - metrics=[ - "mae", - "mse", - "rmse", - "mase", - "smape", - "crps", - "wql", - "mcis", - "pinball", - "skill_score_ratio", - ], + metrics=list(FORECASTING_METRICS), prediction_length=pred_len, freq=freq, seasonality=seasonality, diff --git a/datasets/svdb.py b/datasets/svdb.py index aa78f49..f615a39 100644 --- a/datasets/svdb.py +++ b/datasets/svdb.py @@ -1,7 +1,7 @@ from benchopt import BaseDataset -from benchmark_utils.download import fetch_tsb_uad, load_data_tsb_uad +from benchmark_utils.download_pooch import fetch_tsb_uad, load_data_tsb_uad from benchmark_utils.metrics import AD_METRICS diff --git a/datasets/yahoo.py b/datasets/yahoo.py index 4c66b7b..404e120 100644 --- a/datasets/yahoo.py +++ b/datasets/yahoo.py @@ -1,6 +1,6 @@ from benchopt import BaseDataset -from benchmark_utils.download import fetch_tsb_uad, load_data_tsb_uad +from benchmark_utils.download_pooch import fetch_tsb_uad, load_data_tsb_uad from benchmark_utils.metrics import AD_METRICS diff --git a/objective.py b/objective.py index 7c88290..dbe8ef2 100644 --- a/objective.py +++ b/objective.py @@ -18,7 +18,9 @@ Task-specific shapes -------------------- -forecasting X_test List[(T_i, C)] full series — adapter uses +forecasting y_train None — solvers carve fine-tuning + windows out of X_train themselves + X_test List[(T_i, C)] full series — adapter uses ``x[:cutoff]`` as history cutoff_indexes List[List[int]] jagged per-series cutoffs y_test List[(n_cutoffs, H, C)] @@ -105,6 +107,20 @@ def set_data( self.metrics = metrics self.meta = meta # freq, prediction_length, n_classes, … + def skip(self, **data): + """Honor a ``_skip_reason`` field set by the dataset. + + Datasets that want to filter their own parameter grid (e.g. + :mod:`datasets.gifteval` skipping non-leaderboard (path, term) + combos) return ``dict(_skip_reason="...")`` from ``get_data()``. + benchopt calls this hook *before* ``set_data`` and returns early + on skip, so no other data field is needed or consumed. + """ + reason = data.get("_skip_reason") + if reason: + return True, reason + return False, None + # ------------------------------------------------------------------ # Passed to the solver # ------------------------------------------------------------------ diff --git a/tests/benchmark_utils/test_forecasting_constants.py b/tests/benchmark_utils/test_forecasting_constants.py new file mode 100644 index 0000000..b404e2d --- /dev/null +++ b/tests/benchmark_utils/test_forecasting_constants.py @@ -0,0 +1,70 @@ +"""Conversion tests for the shared frequency / seasonality tables.""" + +import pytest + +from benchmark_utils.forecasting_constants import ( + from_aeon, + from_pandas, + gift_eval_prediction_length, +) + + +@pytest.mark.parametrize( + "alias, expected", + [ + ("H", ("H", 24, 24)), + ("D", ("D", 7, 14)), + ("W-SUN", ("W-SUN", 52, 13)), + ("QS-OCT", ("QS-OCT", 4, 8)), + ("YE", ("YE", 1, 6)), + # Multipliers scale the seasonality; freq keeps the true step. + ("5T", ("5T", 288, 60)), + ("15T", ("15T", 96, 60)), + ("30T", ("30T", 48, 60)), + ("15min", ("15min", 96, 60)), + ("6H", ("6H", 4, 24)), + ("10S", ("10S", 1, 60)), + # Unknown or empty aliases fall back to daily. + ("", ("D", 7, 14)), + ("??", ("D", 7, 14)), + ], +) +def test_from_pandas(alias, expected): + assert from_pandas(alias) == expected + + +@pytest.mark.parametrize( + "word, expected", + [ + ("yearly", ("Y", 1, 6)), + ("monthly", ("M", 12, 12)), + ("hourly", ("H", 24, 24)), + # Sub-hourly aeon words resolve through multiplied aliases. + ("half_hourly", ("30T", 48, 60)), + ("10_minutes", ("10T", 144, 60)), + ("4_seconds", ("4S", 1, 60)), + ("unknown_word", ("D", 7, 14)), + ], +) +def test_from_aeon(word, expected): + assert from_aeon(word) == expected + + +def test_gift_eval_prediction_length_terms(): + assert gift_eval_prediction_length("H", "short") == 48 + assert gift_eval_prediction_length("5T", "medium") == 480 + assert gift_eval_prediction_length("H", "long") == 720 + # Multiplied aliases missing from the map fall back to their base. + assert gift_eval_prediction_length("20T", "short") == 48 + with pytest.raises(ValueError): + gift_eval_prediction_length("H", "weekly") + + +def test_gift_eval_prediction_length_m4(): + # m4 datasets follow the M4-competition horizons, not the generic map. + assert gift_eval_prediction_length("A", "short", "m4_yearly/A") == 6 + assert gift_eval_prediction_length("Q", "short", "m4_quarterly/Q") == 8 + assert gift_eval_prediction_length("M", "short", "m4_monthly/M") == 18 + assert gift_eval_prediction_length("W", "short", "m4_weekly/W") == 13 + assert gift_eval_prediction_length("D", "short", "m4_daily/D") == 14 + assert gift_eval_prediction_length("H", "short", "m4_hourly/H") == 48 diff --git a/tests/benchmark_utils/test_windowing.py b/tests/benchmark_utils/test_windowing.py new file mode 100644 index 0000000..f99cfb6 --- /dev/null +++ b/tests/benchmark_utils/test_windowing.py @@ -0,0 +1,47 @@ +"""Split-consistency tests for the shared forecasting data builder.""" + +import numpy as np +import pytest + +from benchmark_utils.windowing import build_forecasting_data + + +def _series(T): + return np.arange(T, dtype=np.float32).reshape(-1, 1) + + +def test_split_shapes(): + d = build_forecasting_data([_series(100)], prediction_length=10, n_windows=2) + assert d["X_train"][0].shape == (80, 1) + assert d["y_train"] is None + assert d["cutoff_indexes"][0] == [80, 90] + assert d["y_test"][0].shape == (2, 10, 1) + + +def test_train_does_not_overlap_test_windows(): + ts = _series(100) + d = build_forecasting_data([ts], prediction_length=10, n_windows=2) + # Training history ends where the first evaluation window starts. + assert d["X_train"][0].shape[0] <= d["cutoff_indexes"][0][0] + assert np.array_equal(d["X_train"][0], ts[:80]) + + +def test_short_series_dropped(): + d = build_forecasting_data( + [_series(5), _series(50)], prediction_length=10, n_windows=1 + ) + assert len(d["X_train"]) == len(d["X_test"]) == 1 + + +def test_all_series_too_short_raises(): + with pytest.raises(ValueError, match="shorter than prediction_length"): + build_forecasting_data([_series(5)], prediction_length=10) + + +def test_debug_uses_single_window(): + d = build_forecasting_data( + [_series(100)], prediction_length=10, n_windows=2, debug=True + ) + # Train cut still reserves n_windows, but only one window is evaluated. + assert d["X_train"][0].shape == (80, 1) + assert d["cutoff_indexes"][0] == [90] diff --git a/tests/datasets/test_fev.py b/tests/datasets/test_fev.py new file mode 100644 index 0000000..538bee8 --- /dev/null +++ b/tests/datasets/test_fev.py @@ -0,0 +1,63 @@ +"""Tests for the FEV dataset helpers (freq inference, channel selection).""" + +import inspect +from pathlib import Path + +import numpy as np +import pandas as pd +import pytest +from benchopt.benchmark import Benchmark + +BENCHMARK_DIR = Path(__file__).parents[2] +Dataset, = Benchmark(BENCHMARK_DIR).check_dataset_patterns( + ["FEV"], class_only=True +) +fev = inspect.getmodule(Dataset) + + +class TestInferFreq: + def test_regular_hourly(self): + ts = pd.date_range("2024-01-01", periods=10, freq="h") + assert fev._infer_freq(ts).upper().startswith("H") + + def test_gap_in_first_points_uses_delta_mode(self): + # One missing stamp among the first 5 breaks pd.infer_freq; the + # most-common-delta fallback should still find hourly. + ts = pd.date_range("2024-01-01", periods=10, freq="h").delete(2) + assert fev._infer_freq(ts).upper().startswith("H") + + def test_uninferable_warns_and_defaults_to_daily(self): + with pytest.warns(UserWarning, match="defaulting to daily"): + assert fev._infer_freq(pd.DatetimeIndex([])) == "D" + + +class TestChannelSelection: + def test_target_column_is_sole_channel(self): + # epf_*-style schema: target + numeric covariate columns. + df = pd.DataFrame({ + "id": ["a"], + "timestamp": [np.array(["2024-01-01"], dtype="datetime64[ns]")], + "target": [np.array([1.0, 2.0])], + "Load Forecast": [np.array([3.0, 4.0])], + }) + assert fev._select_channel_cols(df) == ["target"] + + def test_no_target_stacks_numeric_array_cols(self): + # ETT-style schema: one column per channel, no target. + df = pd.DataFrame({ + "id": ["a"], + "timestamp": [np.array(["2024-01-01"], dtype="datetime64[ns]")], + "HUFL": [np.array([1.0])], + "OT": [np.array([2.0])], + "type": ["metadata-string"], + "holidays": [np.array(["xmas"], dtype=object)], + }) + assert fev._select_channel_cols(df) == ["HUFL", "OT"] + + def test_empty_first_row_does_not_drop_channel(self): + df = pd.DataFrame({ + "id": ["a", "b"], + "timestamp": [None, None], + "OT": [np.array([]), np.array([1.0, 2.0])], + }) + assert fev._select_channel_cols(df) == ["OT"] diff --git a/tests/datasets/test_gifteval.py b/tests/datasets/test_gifteval.py new file mode 100644 index 0000000..e8cfef0 --- /dev/null +++ b/tests/datasets/test_gifteval.py @@ -0,0 +1,46 @@ +"""Tests for the GIFT-Eval leaderboard-combos loading (offline).""" + +import inspect +from pathlib import Path + +from benchopt.benchmark import Benchmark + +BENCHMARK_DIR = Path(__file__).parents[2] +Dataset, = Benchmark(BENCHMARK_DIR).check_dataset_patterns( + ["GiftEval"], class_only=True +) +gifteval = inspect.getmodule(Dataset) + + +def test_parse_leaderboard_csv(tmp_path): + csv_path = tmp_path / "leaderboard_combos.csv" + csv_path.write_text( + "dataset_name,term\n" + "m4_weekly/W,short\n" + "electricity/15T,short\n" + "electricity/15T,medium\n" + "electricity/15T,long\n" + ) + table = gifteval._parse_leaderboard_csv(csv_path) + assert table == { + "m4_weekly/W": ("short",), + "electricity/15T": ("short", "medium", "long"), + } + + +def test_non_canonical_combo_skips(monkeypatch): + monkeypatch.setattr( + gifteval, "_leaderboard_cache", {"m4_weekly/W": ("short",)} + ) + ds = Dataset.get_instance(dataset_name="m4_weekly/W", term="long") + data = ds.get_data() + assert set(data) == {"_skip_reason"} + assert "does not define term 'long'" in data["_skip_reason"] + + +def test_hf_arrow_directory(): + assert gifteval._hf_arrow_directory("m4_weekly/W") == "m4_weekly" + assert gifteval._hf_arrow_directory("loop_seattle/H") == "LOOP_SEATTLE/H" + assert ( + gifteval._hf_arrow_directory("car_parts/M") == "car_parts_with_missing" + )