Files
UAV-CO2/src/gasflux/pre_processing.py

147 lines
5.4 KiB
Python
Raw Normal View History

"""Functions that organise the data into standard columns in pandas dataframes. Conversion functions (e.g. WGS84 to UTM)
are here but transformations take place in processing.py"""
import numpy as np
import pandas as pd
2026-04-20 09:43:21 +08:00
from pyproj import Transformer
from .processing import circ_median
def data_tests(df: pd.DataFrame):
assert df["ch4"].min() > 1.6, "ch4 values are too low"
assert df.index.is_monotonic_increasing, "data is not sorted by time"
assert df.index.is_unique, "data has duplicate timestamps"
assert df["ch4"].isna().sum() == 0, "ch4 has missing values"
assert df["windspeed"].min() >= 0, "windspeed values are negative"
assert df["windspeed"].max() < 20, "windspeed values are too high"
if df["windspeed"].max() > 15:
print("Warning: windspeed is greater than 15 m/s, perhaps due to errors in the data.")
# make timestamp column from UTCs, Month, Day, Year
def timestamp_from_four_columns(df):
df["Year"] = df["Year"] + 2000
df["time"] = pd.to_datetime(df["UTCs"], unit="s")
df["date"] = pd.to_datetime(df[["Year", "Month", "Day"]])
df["timestamp"] = pd.to_datetime(df["date"].dt.date.astype(str) + " " + df["time"].dt.time.astype(str))
df.index = df["timestamp"]
df.drop(
["Year", "Month", "Day", "time", "date", "timestamp", "UTCs"],
axis=1,
inplace=True,
)
return df
# add UTM from latitudes and longitudes
def add_utm(df: pd.DataFrame) -> pd.DataFrame:
2026-04-20 09:43:21 +08:00
"""
Convert WGS84 coordinates to UTM using pyproj.
This function replaces the geopandas implementation with a lighter pyproj-based solution.
"""
# Make a copy to avoid modifying the original DataFrame
df = df.copy()
# Get the average longitude to determine the UTM zone
# For simplicity, we'll use the first valid longitude to determine the zone
# In production, you might want to use the centroid or handle multiple zones
valid_lons = df["longitude"].dropna()
if len(valid_lons) == 0:
raise ValueError("No valid longitude values found")
# Calculate UTM zone from longitude
# UTM zones are 6 degrees wide, starting from -180
lon = valid_lons.iloc[0] # Use first valid longitude
zone_number = int((lon + 180) / 6) + 1
# Determine if it's northern or southern hemisphere
# Use first valid latitude
valid_lats = df["latitude"].dropna()
if len(valid_lats) == 0:
raise ValueError("No valid latitude values found")
lat = valid_lats.iloc[0]
hemisphere = 'north' if lat >= 0 else 'south'
# Create UTM CRS string
utm_crs = f"EPSG:326{zone_number:02d}" if hemisphere == 'north' else f"EPSG:327{zone_number:02d}"
# Create transformer from WGS84 to UTM
transformer = Transformer.from_crs("EPSG:4326", utm_crs, always_xy=True)
# Transform coordinates
utm_easting, utm_northing = transformer.transform(
df["longitude"].values,
df["latitude"].values
)
2026-04-20 09:43:21 +08:00
# Add UTM coordinates to DataFrame
df["utm_easting"] = utm_easting
df["utm_northing"] = utm_northing
return df
# add columns for drone course azimuth and elevation
def add_course(df, rolling_window=1):
df["hor_distance"] = np.sqrt((df["utm_northing"].diff()) ** 2 + (df["utm_easting"].diff()) ** 2)
df["vert_distance"] = df["height_ato"].diff()
df["vert_distance"] = pd.to_numeric(df["vert_distance"], errors="coerce")
df["hor_distance"] = pd.to_numeric(df["hor_distance"], errors="coerce")
df["course_azimuth"] = (
(np.degrees(np.arctan2(df["utm_easting"].diff(), df["utm_northing"].diff())) % 360)
.rolling(rolling_window)
.apply(lambda x: circ_median(x), raw=True)
)
df["course_elevation"] = (
np.degrees(np.arctan2(df["vert_distance"], df["hor_distance"]))
.rolling(rolling_window)
.apply(lambda x: circ_median(x), raw=True)
)
return df
def manual_filtering(dict_dfs: dict, split_times: dict, mask_spans: dict) -> dict:
filtered_dfs = {}
for name, df in dict_dfs.items():
if name in mask_spans:
for i in range(len(mask_spans[name])):
df = df.drop(
df.between_time(mask_spans[name][i].split(" - ")[0], mask_spans[name][i].split(" - ")[1]).index,
).copy()
filtered_dfs[name] = df.copy()
if name in split_times:
split_times[name].append("23:59:59")
split_times[name].insert(0, "00:00:00")
for i in range(len(split_times[name]) - 1):
df2 = df.between_time(split_times[name][i], split_times[name][i + 1]).copy()
filtered_dfs[name + "_" + str(i)] = df2.copy()
elif name not in split_times:
filtered_dfs[name] = df.copy()
return filtered_dfs
def remove_outliers(df: pd.DataFrame, column: str, name: str):
q1 = df[column].quantile(0.25)
q3 = df[column].quantile(0.75)
iqr = q3 - q1
fence_low = q1 - 3 * iqr
fence_high = q3 + 3 * iqr
2026-04-20 09:43:21 +08:00
try:
from . import plotting
fig = plotting._outliers_wrapper(df, column, name)
except Exception:
fig = None
2026-04-20 09:43:21 +08:00
outliers = df.loc[(df[column] < fence_low) | (df[column] > fence_high)]
if len(outliers) > 0:
print(f"{len(outliers)} outliers removed from {name} {column} data")
# nan for outliers, not row removal
df.loc[(df[column] < fence_low) | (df[column] > fence_high), column] = float("nan")
elif len(outliers) == 0:
print(f"No outliers found in {name} {column} data")
return df, fig