"""Personal learning experiment. Synthetic data only; not a real business forecast.
Run: python experiment.py (requires numpy and pandas).
"""
import json
from pathlib import Path
import numpy as np
import pandas as pd

def run():
    rng = np.random.default_rng(42)
    days = np.arange(180)
    dates = pd.date_range("2025-01-01", periods=len(days))
    # The generator deliberately includes a trend and weekly pattern.
    # A linear model can recover these assumptions; this is not external validation.
    demand = 100 + 0.25 * days + 14 * np.sin(2 * np.pi * days / 7) + rng.normal(0, 6, len(days))
    features = np.column_stack([np.ones(len(days)), days,
                               np.sin(2 * np.pi * days / 7), np.cos(2 * np.pi * days / 7)])
    split = 140
    weights = np.linalg.lstsq(features[:split], demand[:split], rcond=None)[0]
    predicted = features[split:] @ weights
    # Both baselines use training data only; no test observations are fed back.
    mean_baseline = np.full(len(days) - split, demand[:split].mean())
    weekly_baseline = np.array([demand[:split][days[:split] % 7 == day % 7].mean() for day in days[split:]])
    actual = demand[split:]
    metrics = {}
    for name, values in [("training_mean", mean_baseline), ("weekday_mean", weekly_baseline), ("linear_trend_and_week", predicted)]:
        metrics[name] = {"mae": round(float(np.abs(actual - values).mean()), 3),
                         "rmse": round(float(np.sqrt(np.mean((actual - values) ** 2))), 3)}
    output = Path(__file__).parent / "results"
    output.mkdir(exist_ok=True)
    report = {"dataset": "Synthetic daily demand, seed 42", "training_days": split,
              "held_out_days": len(days) - split, "metrics": metrics,
              "limitations": ["The model matches the known data generator.",
                              "One chronological holdout; no rolling validation.",
                              "No real-world demand, deployment or business impact claims."]}
    (output / "metrics.json").write_text(json.dumps(report, indent=2) + "\n")
    pd.DataFrame({"date": dates[split:], "actual": actual, "prediction": predicted,
                  "training_mean": mean_baseline, "weekday_mean": weekly_baseline}).to_csv(output / "holdout.csv", index=False)
    print(json.dumps(report, indent=2))

if __name__ == "__main__":
    run()
