Repository navigation
Expand file tree
/
Copy pathpreprocess.py
More file actions
58 lines (46 loc) · 1.49 KB
/
Copy pathpreprocess.py
File metadata and controls
58 lines (46 loc) · 1.49 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
#!/usr/bin/env python
"""
Build train.csv and test.csv from raw GPS and constellation data.
Run this once before any training, evaluation, or ablation:
python preprocess.py
Output
------
data/train.csv — aggregated, feature-engineered, labeled training rows
data/test.csv — same for the cross-test split
"""
import pandas as pd
from const import (
DEFAULT_TRAIN_CONFIG,
DEFAULT_TEST_CONFIG,
EXCLUDED_TESTS,
TRAIN_CSV,
TEST_CSV,
)
from utils import read_raw_csvs, create_agg_fe_labeled_from_df
def build_split(config, gps_df, const_df):
frames = []
for c in config:
df = create_agg_fe_labeled_from_df(
gps_df=gps_df,
constellation_df=const_df,
day=c["day"],
hour_1=c["hour_1"],
hour_2=c["hour_2"],
agg_size="1S",
loc_id=c["loc_id"],
)
frames.append(df[~df["test_id"].isin(EXCLUDED_TESTS)])
return pd.concat(frames, ignore_index=True)
def main():
print("Reading raw CSV data …")
gps_df, const_df = read_raw_csvs()
print("Building train split …")
train_df = build_split(DEFAULT_TRAIN_CONFIG, gps_df, const_df)
train_df.to_csv(TRAIN_CSV, index=False)
print(f" {len(train_df):,} rows → {TRAIN_CSV}")
print("Building test split …")
test_df = build_split(DEFAULT_TEST_CONFIG, gps_df, const_df)
test_df.to_csv(TEST_CSV, index=False)
print(f" {len(test_df):,} rows → {TEST_CSV}")
if __name__ == "__main__":
main()