| 1 | import os |
| 2 | import pandas as pd |
| 3 | import numpy as np |
| 4 | import matplotlib.pyplot as plt |
| 5 | import seaborn as sns |
| 6 | |
| 7 | |
| 8 | data = pd.read_csv( |
| 9 | "../assets/datasets/time_series_solar.csv", |
| 10 | parse_dates=["Datetime"], |
| 11 | index_col="Datetime", |
| 12 | ) |
| 13 | series = data["Incoming Solar"] |
| 14 | |
| 15 | |
| 16 | series_daily = series.resample("D").sum() |
| 17 | |
| 18 | |
| 19 | sample_with_nan = series_daily.head(365 * 2).copy() |
| 20 | size_na = int(0.6 * len(sample_with_nan)) |
| 21 | |
| 22 | idx = np.random.choice(a=range(len(sample_with_nan)), size=size_na, replace=False) |
| 23 | |
| 24 | sample_with_nan[idx] = np.nan |
| 25 | |
| 26 | |
| 27 | average_value = sample_with_nan.mean() |
| 28 | imp_mean = sample_with_nan.fillna(average_value) |
| 29 | imp_ffill = sample_with_nan.ffill() |
| 30 | imp_bfill = sample_with_nan.bfill() |
| 31 | |
| 32 | |
| 33 | plt.rcParams['figure.figsize'] = [12, 6] |
| 34 | |
| 35 | sns.set_theme(style='darkgrid') |
| 36 | |
| 37 | fig, (ax0, ax1, ax2, ax3) = plt.subplots(4, sharex=True) |
| 38 | fig.suptitle('Time series imputation methods') |
| 39 | |
| 40 | ax0.plot(sample_with_nan) |
| 41 | ax0.set_title('Original series with missing data') |
| 42 | ax1.plot(imp_mean) |
| 43 | ax1.set_title('Series with mean imputation') |
| 44 | ax2.plot(imp_ffill) |
| 45 | ax2.set_title('Series with ffill imputation (Most common)') |
| 46 | ax3.plot(imp_bfill) |
| 47 | ax3.set_title('Series with bfill imputation') |
| 48 | |
| 49 | plt.tight_layout() |
| 50 | |
| 51 | os.makedirs("assets", exist_ok=True) |
| 52 | plt.savefig("assets/missing_data_plot.png") |