with '#' will be ignored, and an empty message aborts the commit.

Minor changes, run autoencoder on ml node
This commit is contained in:
Lars Bogner
2025-11-20 14:15:06 +01:00
parent 4b071e68b5
commit 1538b0acf7
6 changed files with 133 additions and 20 deletions
@@ -110,7 +110,7 @@
],
"metadata": {
"kernelspec": {
"display_name": "adsbpy",
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
@@ -124,7 +124,7 @@
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.13.9"
"version": "3.11.13"
}
},
"nbformat": 4,
@@ -0,0 +1,29 @@
# %%
import torch
import torch.nn.functional as F
import pandas as pd
import numpy as np
from aiRNN import preprocessors
# %%
df = pd.read_csv("../all_second_lines.csv", header=None, names=["icao", "r", "t", "timestamp", "lat", "lon", "alt", "ias"])
_ = preprocessors.create_category_mappings(df)
# %%
df = preprocessors.map_categories(df, _)
# %%
X_a_1 = F.one_hot(torch.tensor(df["R_1_IDX"].values)).float()
X_a_2 = F.one_hot(torch.tensor(df["R_2_IDX"].values)).float()
X_t = F.one_hot(torch.tensor(df["T_IDX"].values)).float()
# %%
print(X_a_1.shape, X_a_2.shape, X_t.shape)
# %%
preprocessors.train_autoencoder(X_a_1=X_a_1, X_a_2=X_a_2, X_b=X_t, num_epochs=500, learning_rate=0.011)
# %%
+1 -1
View File
@@ -3,7 +3,7 @@ name = "adsbpy"
version = "0.1.0"
description = "Add your description here"
readme = "README.md"
requires-python = ">=3.13"
requires-python = ">=3.11"
dependencies = [
"aiofiles>=25.1.0",
"aiohttp>=3.13.2",
Binary file not shown.
+39 -3
View File
@@ -1,10 +1,12 @@
import pandas as pd
import torch
import torch.nn.functional as F
from torch.utils.data import Dataset, DataLoader
import numpy as np
import pathlib
from typing import Sequence
import random
from aiRNN import preprocessors
class BaseDataset(Dataset):
def __init__(
@@ -15,7 +17,7 @@ class BaseDataset(Dataset):
n_windows_per_file,
step=1,
feature_columns=("lat", "lon", "alt", "ias"),
context_columns=("r","t"),
context_columns=(f"type_encoding_{i}" for i in range(4), "last_lat", "last_lon", "last_alt", "last_ias"),
time_columns=("timestamp",),
target_columns=("lat", "lon", "alt"),
device="cpu",
@@ -38,7 +40,6 @@ class BaseDataset(Dataset):
# Total samples = windows_per_file * number_of_files
self.total_windows = n_windows_per_file * len(self.filepaths)
for col_list in [self.feature_cols, self.context_cols, self.time_cols, self.target_cols]:
assert all(
col in sample.columns for col in col_list
@@ -68,6 +69,28 @@ class BaseDataset(Dataset):
def _modify_df(self, df):
"""Hook for subclasses to modify dataframe before slicing windows."""
df.loc[:, ["lat", "lon", "alt"]] = preprocessors.norm_coords(
df["lat"].values, df["lon"].values, df["alt"].values
)
df.loc[:, "ias"] = preprocessors.norm_ias(df["ias"].values)
df.loc[:, "timestamp"] = preprocessors.norm_time(df["timestamp"].values)
for col in ["lat", "lon", "alt", "ias"]:
df.loc[:, col] = preprocessors.fillna_with_mean(df[col])
mapped_df = preprocessors.map_categories(df, preprocessors.category_mappings)
X_a_1 = F.one_hot(torch.tensor(mapped_df["R_1_IDX"].values)).float()
X_a_2 = F.one_hot(torch.tensor(mapped_df["R_2_IDX"].values)).float()
X_t = F.one_hot(torch.tensor(mapped_df["T_IDX"].values)).float()
df.loc[:, [f"type_encoding_{i}" for i in range(4)]] = preprocessors.encode_features(
X_a_1, X_a_2, X_t
)
last_row = df.iloc[-1][["lat", "lon", "alt", "ias"]]
df.loc[:, "last_lat"] = last_row["lat"]
df.loc[:, "last_lon"] = last_row["lon"]
df.loc[:, "last_alt"] = last_row["alt"]
df.loc[:, "last_ias"] = last_row["ias"]
return df
@@ -108,6 +131,7 @@ class EvenlySpacedDataset(BaseDataset):
try:
start_idx, end_idx = self._get_start_end_indices(w, n_rows)
window = df.iloc[start_idx:end_idx:self.step]
window = self._modify_df(window)
self.data.append(window)
except ValueError:
self.total_windows -= 1
@@ -184,6 +208,7 @@ class EvenlySpacedStreamingDataset(BaseDataset):
# Read only the required rows
skip = list(set(range(1, start_idx + 1)))
df = pd.read_csv(fp, skiprows=skip, nrows=self.window_size * self.step).iloc[::self.step]
df = self._modify_df(df)
# Slice into input / decoder-input / targets
X_feat = df.iloc[: self.n_input][self.feature_cols].to_numpy(np.float32)
@@ -272,4 +297,15 @@ def get_datasets(
)
return train_dataset, val_dataset, test_dataset
class EncodingDataset(Dataset):
def __init__(self, X_a_1, X_a_2, X_b):
super().__init__()
self.X_a_1 = X_a_1
self.X_a_2 = X_a_2
self.X_b = X_b
def __len__(self):
return self.X_a_1.shape[0]
def __getitem__(self, idx):
return self.X_a_1[idx], self.X_a_2[idx], self.X_b[idx]
+62 -14
View File
@@ -2,12 +2,14 @@ import torch
from torch import nn
import pathlib
import json
from aiRNN.dataloader import EncodingDataset
MIN_LAT, MAX_LAT = 57.0, 72.0
MIN_LON, MAX_LON = 3.0, 32.0
MIN_ALT, MAX_ALT = -500.0, 50000.0 # in feet
MIN_IAS, MAX_IAS = 0.0, 800.0 # in knots
TIME_NORM_FACTOR = 1/3600.0 # normalize time to hours
def min_max_normalize(value, min_val, max_val):
return (value - min_val) / (max_val - min_val)
@@ -33,6 +35,14 @@ def norm_ias(ias):
def denorm_ias(ias_norm):
return min_max_denormalize(ias_norm, MIN_IAS, MAX_IAS)
def norm_time(timestamp_series):
min_time = timestamp_series.min()
return (timestamp_series - min_time) * TIME_NORM_FACTOR
def fillna_with_mean(series):
mean_value = series.mean()
return series.fillna(mean_value)
class AutoEncoder(nn.Module):
def __init__(self, input_a_1_size, input_a_2_size, input_b_size, latent_size):
super(AutoEncoder, self).__init__()
@@ -85,27 +95,62 @@ if auto_encoder_path.exists():
def train_autoencoder(X_a_1, X_a_2, X_b, num_epochs=100, learning_rate=1e-3):
criterion = nn.CrossEntropyLoss()
optimizer = torch.optim.SGD(autoencoder.parameters(), lr=learning_rate)
optimizer = torch.optim.Adam(autoencoder.parameters(), lr=learning_rate)
ds = EncodingDataset(X_a_1, X_a_2, X_b)
dataloader = torch.utils.data.DataLoader(ds, batch_size=512, shuffle=True)
# -----------------------------
# Auto-termination parameters
# -----------------------------
patience = 10
best_loss = float('inf')
epochs_no_improve = 0
# -----------------------------
# Training loop
# -----------------------------
for epoch in range(num_epochs):
autoencoder.train()
optimizer.zero_grad()
rec_A_1, rec_A_2, rec_B = autoencoder(X_a_1, X_a_2, X_b)
loss_A_1 = criterion(rec_A_1, torch.argmax(X_a_1, dim=1))
loss_A_2 = criterion(rec_A_2, torch.argmax(X_a_2, dim=1))
loss_B = criterion(rec_B, torch.argmax(X_b, dim=1))
loss = loss_A_1 + loss_A_2 + loss_B
loss.backward()
optimizer.step()
if (epoch + 1) % 10 == 0:
print(f"Epoch [{epoch+1}/{num_epochs}], Loss: {loss.item():.4f}")
epoch_loss = 0.0
for Xa1, Xa2, Xb in dataloader:
optimizer.zero_grad()
rec_A_1, rec_A_2, rec_B = autoencoder(Xa1, Xa2, Xb)
# Targets from the one-hot vectors
yA1 = torch.argmax(Xa1, dim=1)
yA2 = torch.argmax(Xa2, dim=1)
yB = torch.argmax(Xb, dim=1)
loss = (
criterion(rec_A_1, yA1) +
criterion(rec_A_2, yA2) +
criterion(rec_B, yB)
)
loss.backward()
optimizer.step()
epoch_loss += loss.item() * Xa1.size(0)
epoch_loss /= len(ds)
print(f"Epoch [{epoch+1}/{num_epochs}] - Loss: {epoch_loss:.4f}")
# --- Auto-termination ---
if epoch_loss < best_loss:
best_loss = epoch_loss
epochs_no_improve = 0
else:
epochs_no_improve += 1
if epochs_no_improve >= patience:
print("Training stopped early — loss plateaued.")
break
torch.save(autoencoder.state_dict(), auto_encoder_path)
def encode_features(X_a, X_b):
def encode_features(X_a_1, X_a_2, X_b):
autoencoder.eval()
with torch.no_grad():
X = torch.cat((X_a, X_b), dim=1)
X = torch.cat((X_a_1, X_a_2, X_b), dim=1)
latent = autoencoder.encoder(X)
return latent
@@ -132,7 +177,9 @@ def create_category_mappings(df):
return category_mappings
def map_categories(df, category_mappings):
df = df.loc[:, ["r", "t"]].dropna().reset_index(drop=True)
if category_mappings is None:
raise ValueError("Category mappings not provided.")
df = df.loc[:, ["r", "t"]].copy().dropna().reset_index(drop=True)
df.loc[:, "r_1"] = df["r"].str.split("-").str[0]
df.loc[:, "r_2"] = df["r"].str.split("-").str[1].fillna("Empty")
@@ -147,4 +194,5 @@ if mappings_path.exists():
with open(mappings_path, "r") as f:
category_mappings = json.load(f)
else:
category_mappings = None
print("Category mappings file not found. Please create mappings using 'create_category_mappings' function.")