"""New, bounded validation experiment; NOT the original study or a live backtest.

Orders and future predictor scenarios are chosen using each training fold only.
Uses current-vintage snapshots; historical publication lags remain unresolved.
Run: python validate.py /path/to/inflation-forecast
"""
import argparse
import json

import numpy as np
import pmdarima as pm
from sklearn.model_selection import TimeSeriesSplit

from audit import FEATURES, TARGET, future_scenario, prepare, score


def main():
    from pathlib import Path
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("repository", type=Path)
    args = parser.parse_args()
    _, data = prepare(args.repository / "makro_veri_seti_kapsamli.csv")
    rows = []
    for fold, (train, test) in enumerate(TimeSeriesSplit(n_splits=3, test_size=17).split(data), 1):
        history, held_out = data.iloc[train], data.iloc[test]
        y_train, x_train = history[TARGET], history[FEATURES]
        # Deliberately small NONSEASONAL candidate set; no claim of global optimum.
        # d is tested on training data. The selected fitted model is used directly.
        model = pm.auto_arima(
            y_train, X=x_train, seasonal=False, start_p=0, start_q=0,
            max_p=2, max_q=2, max_d=1, information_criterion="aic",
            stepwise=True, maxiter=200, suppress_warnings=False,
            error_action="warn", with_intercept="auto",
        )
        if not model.arima_res_.mle_retvals.get("converged"):
            raise RuntimeError(f"Fold {fold} selected model did not converge.")
        future_x = future_scenario(x_train, held_out.index)
        prediction = model.predict(n_periods=len(test), X=future_x)
        candidates = {
            "train_selected_arimax": prediction,
            "training_mean": np.repeat(y_train.mean(), len(test)),
            "last_value": np.repeat(y_train.iloc[-1], len(test)),
            "seasonal_naive": np.resize(y_train.iloc[-12:].to_numpy(), len(test)),
        }
        rows.append({"fold": fold, "train_n": len(train), "test_n": len(test),
                     "train_end": str(history.index[-1].date()),
                     "test_start": str(held_out.index[0].date()),
                     "test_end": str(held_out.index[-1].date()),
                     "order": model.order, "with_intercept": model.with_intercept,
                     "converged": True,
                     "metrics": {key: score(held_out[TARGET], pred) for key, pred in candidates.items()}})
    print(json.dumps({
        "experiment": "Training-only order selection; training-only exogenous scenario; 17-month horizon",
        "folds": rows,
        "mean_fold_metrics": {
            name: {metric: float(np.mean([r["metrics"][name][metric] for r in rows]))
                   for metric in ("mae_pp", "rmse_pp")}
            for name in rows[0]["metrics"]},
        "limitations": "Small nonseasonal search. Three folds; not a final untouched holdout. "
                       "Snapshot data, no original publication vintages. Not evidence of causal effects."
    }, ensure_ascii=False, indent=2, allow_nan=False))


if __name__ == "__main__":
    main()
