preprocess the app flow dataset for long range forecasting
(csv_path)
| 461 | |
| 462 | |
| 463 | def preprocess_flow(csv_path): |
| 464 | """preprocess the app flow dataset for long range forecasting""" |
| 465 | data_frame = pd.read_csv(csv_path, names=['app_name', 'zone', 'time', 'value'], parse_dates=True) |
| 466 | grouped_data = list(data_frame.groupby(["app_name", "zone"])) |
| 467 | # covariates = gen_covariates(data_frame.index, 3) |
| 468 | all_data = [] |
| 469 | min_length = 10000 |
| 470 | for i in range(len(grouped_data)): |
| 471 | single_df = grouped_data[i][1].drop(labels=['app_name', 'zone'], axis=1).sort_values(by="time", ascending=True) |
| 472 | times = pd.to_datetime(single_df.time) |
| 473 | single_df['weekday'] = times.dt.dayofweek / 7 |
| 474 | single_df['hour'] = times.dt.hour / 24 |
| 475 | single_df['month'] = times.dt.month / 12 |
| 476 | temp_data = single_df.values[:, 1:] |
| 477 | if (temp_data[:, 0] == 0).sum() / len(temp_data) > 0.2 or len(temp_data) < 3000: |
| 478 | continue |
| 479 | |
| 480 | if len(temp_data) < min_length: |
| 481 | min_length = len(temp_data) |
| 482 | |
| 483 | all_data.append(temp_data) |
| 484 | |
| 485 | all_data = np.array([data[len(data)-min_length:, :] for data in all_data]).transpose(1, 0, 2).astype(np.float32) |
| 486 | train_end = min(int(0.8 * min_length), min_length - 1000) |
| 487 | covariates = all_data.copy() |
| 488 | covariates[:, :, :-1] = covariates[:, :, 1:] |
| 489 | |
| 490 | return all_data[:, :, 0], covariates, train_end |
| 491 | |
| 492 | |
| 493 | """Single step dataloader""" |
nothing calls this directly
no outgoing calls
no test coverage detected