import os import pandas as pd import numpy as np import matplotlib.pyplot as plt import seaborn as sns # Load solar data data = pd.read_csv( "../assets/datasets/time_series_solar.csv", parse_dates=["Datetime"], index_col="Datetime", ) series = data["Incoming Solar"] # Resample to daily series_daily = series.resample("D").sum() # Create random nan values in first 2 years, remove 60% of values sample_with_nan = series_daily.head(365 * 2).copy() size_na = int(0.6 * len(sample_with_nan)) idx = np.random.choice(a=range(len(sample_with_nan)), size=size_na, replace=False) sample_with_nan[idx] = np.nan # Gather imputations average_value = sample_with_nan.mean() imp_mean = sample_with_nan.fillna(average_value) # Fill na with average imp_ffill = sample_with_nan.ffill() # Fill na with previous value imp_bfill = sample_with_nan.bfill() # Fill na with next value # Plot all on one chart plt.rcParams['figure.figsize'] = [12, 6] sns.set_theme(style='darkgrid') fig, (ax0, ax1, ax2, ax3) = plt.subplots(4, sharex=True) fig.suptitle('Time series imputation methods') ax0.plot(sample_with_nan) ax0.set_title('Original series with missing data') ax1.plot(imp_mean) ax1.set_title('Series with mean imputation') ax2.plot(imp_ffill) ax2.set_title('Series with ffill imputation (Most common)') ax3.plot(imp_bfill) ax3.set_title('Series with bfill imputation') plt.tight_layout() os.makedirs("assets", exist_ok=True) plt.savefig("assets/missing_data_plot.png")