restructured repo, fixed git ignore
This commit is contained in:
@@ -0,0 +1,111 @@
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
from pathlib import Path
|
||||
|
||||
def process_parquet_files(input_dir, output_file, window_size=1250, step_size=125):
|
||||
"""
|
||||
Verarbeitet Parquet-Dateien mit Sliding Window Aggregation.
|
||||
|
||||
Parameters:
|
||||
-----------
|
||||
input_dir : str
|
||||
Verzeichnis mit Parquet-Dateien
|
||||
output_file : str
|
||||
Pfad für die Ausgabe-Parquet-Datei
|
||||
window_size : int
|
||||
Größe des Sliding Windows (default: 3000)
|
||||
step_size : int
|
||||
Schrittweite in Einträgen (default: 250 = 10 Sekunden bei 25 Hz)
|
||||
"""
|
||||
|
||||
input_path = Path(input_dir)
|
||||
parquet_files = sorted(input_path.glob("*.parquet"))
|
||||
|
||||
if not parquet_files:
|
||||
print(f"Keine Parquet-Dateien in {input_dir} gefunden!")
|
||||
return
|
||||
|
||||
print(f"Gefundene Dateien: {len(parquet_files)}")
|
||||
|
||||
all_windows = []
|
||||
|
||||
for file_idx, parquet_file in enumerate(parquet_files):
|
||||
print(f"\nVerarbeite Datei {file_idx + 1}/{len(parquet_files)}: {parquet_file.name}")
|
||||
|
||||
# Lade Parquet-Datei
|
||||
df = pd.read_parquet(parquet_file)
|
||||
print(f" Einträge: {len(df)}")
|
||||
|
||||
# Identifiziere AU-Spalten
|
||||
au_columns = [col for col in df.columns if col.startswith('AU')]
|
||||
print(f" AU-Spalten: {len(au_columns)}")
|
||||
|
||||
# Gruppiere nach STUDY, LEVEL, PHASE (um Übergänge zu vermeiden)
|
||||
for (study_val, level_val, phase_val), level_df in df.groupby(['STUDY', 'LEVEL', 'PHASE'], sort=False):
|
||||
print(f" STUDY {study_val}, LEVEL {level_val}, PHASE {phase_val}: {len(level_df)} Einträge")
|
||||
|
||||
# Reset index für korrekte Position-Berechnung
|
||||
level_df = level_df.reset_index(drop=True)
|
||||
|
||||
# Sliding Window über dieses Level
|
||||
num_windows = (len(level_df) - window_size) // step_size + 1
|
||||
|
||||
if num_windows <= 0:
|
||||
print(f" Zu wenige Einträge für Window (benötigt {window_size})")
|
||||
continue
|
||||
|
||||
for i in range(num_windows):
|
||||
start_idx = i * step_size
|
||||
end_idx = start_idx + window_size
|
||||
|
||||
window_df = level_df.iloc[start_idx:end_idx]
|
||||
|
||||
# Erstelle aggregiertes Ergebnis
|
||||
result = {
|
||||
'subjectID': window_df['subjectID'].iloc[0],
|
||||
'start_time': window_df['rowID'].iloc[0], # rowID als start_time
|
||||
'STUDY': window_df['STUDY'].iloc[0],
|
||||
'LEVEL': window_df['LEVEL'].iloc[0],
|
||||
'PHASE': window_df['PHASE'].iloc[0]
|
||||
}
|
||||
|
||||
# Summiere alle AU-Spalten
|
||||
for au_col in au_columns:
|
||||
result[f'{au_col}_sum'] = window_df[au_col].sum()
|
||||
|
||||
all_windows.append(result)
|
||||
|
||||
print(f" Windows erstellt: {num_windows}")
|
||||
|
||||
# Erstelle finalen DataFrame
|
||||
result_df = pd.DataFrame(all_windows)
|
||||
|
||||
print(f"\n{'='*60}")
|
||||
print(f"Gesamt Windows erstellt: {len(result_df)}")
|
||||
print(f"Spalten: {list(result_df.columns)}")
|
||||
|
||||
# Speichere Ergebnis
|
||||
result_df.to_parquet(output_file, index=False)
|
||||
print(f"\nErgebnis gespeichert in: {output_file}")
|
||||
|
||||
return result_df
|
||||
|
||||
|
||||
# Beispiel-Verwendung
|
||||
if __name__ == "__main__":
|
||||
# Anpassen an deine Pfade
|
||||
input_directory = ""
|
||||
output_file = "./output/output_windowed.parquet"
|
||||
|
||||
|
||||
result = process_parquet_files(
|
||||
input_dir=input_directory,
|
||||
output_file=output_file,
|
||||
window_size=1250,
|
||||
step_size=125
|
||||
)
|
||||
|
||||
# Zeige erste Zeilen
|
||||
if result is not None:
|
||||
print("\nErste 5 Zeilen des Ergebnisses:")
|
||||
print(result.head())
|
||||
@@ -0,0 +1,64 @@
|
||||
# %pip install pyocclient
|
||||
import yaml
|
||||
import owncloud
|
||||
import pandas as pd
|
||||
import h5py
|
||||
|
||||
num_files = 30 # number of files to process (min: 1, max: 30)
|
||||
# Load credentials
|
||||
with open("login.yaml") as f:
|
||||
cfg = yaml.safe_load(f)
|
||||
print("ahahahah")
|
||||
url, password = cfg[0]["url"], cfg[1]["password"]
|
||||
|
||||
# Connect once
|
||||
oc = owncloud.Client.from_public_link(url, folder_password=password)
|
||||
print("connection aufgebaut")
|
||||
# File pattern
|
||||
base = "adabase-public-{num:04d}-v_0_0_2.h5py"
|
||||
|
||||
for i in range(num_files):
|
||||
file_name = base.format(num=i)
|
||||
local_tmp = f"tmp_{i:04d}.h5"
|
||||
|
||||
# Download file from ownCloud
|
||||
oc.get_file(file_name, local_tmp)
|
||||
print(f"{file_name} geoeffnet")
|
||||
# Load into memory and extract needed columns
|
||||
# with h5py.File(local_tmp, "r") as f:
|
||||
# # Adjust this path depending on actual dataset layout inside .h5py file
|
||||
# df = pd.DataFrame({k: f[k][()] for k in f.keys() if k in ["STUDY", "LEVEL", "PHASE"] or k.startswith("AU")})
|
||||
|
||||
with pd.HDFStore(local_tmp, mode="r") as store:
|
||||
cols = store.select("SIGNALS", start=0, stop=1).columns # get column names
|
||||
|
||||
# Step 2: Filter columns that start with "AU"
|
||||
au_cols = [c for c in cols if c.startswith("AU")]
|
||||
print(au_cols)
|
||||
|
||||
# Step 3: Read only those columns (plus any others you want)
|
||||
df = pd.read_hdf(local_tmp, key="SIGNALS", columns=["STUDY", "LEVEL", "PHASE"] + au_cols)
|
||||
|
||||
|
||||
print("load done")
|
||||
|
||||
# Add metadata columns
|
||||
df["subjectID"] = i
|
||||
df["rowID"] = range(len(df))
|
||||
|
||||
print("extra columns done")
|
||||
# Clean data
|
||||
# drop level = 0
|
||||
print(df.columns)
|
||||
df = df[df["LEVEL"] != 0]
|
||||
|
||||
df = df.dropna()
|
||||
|
||||
print("data cleaning done")
|
||||
|
||||
|
||||
# Save to parquet
|
||||
out_name = f"cleaned_{i:04d}.parquet"
|
||||
df.to_parquet(out_name, index=False)
|
||||
|
||||
print(f"Processed {file_name} -> {out_name}")
|
||||
@@ -0,0 +1,99 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "2b3fface",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import pandas as pd"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "74f1f5ec",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df= pd.read_parquet(\"cleaned_0000.parquet\")\n",
|
||||
"print(df.shape)\n",
|
||||
"\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "05775454",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "99e17328",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df.tail()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "0238d802",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"step2 = pd.read_parquet(\"output_windowed.parquet\")\n",
|
||||
"step2.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "1257c535",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"step2.shape"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "3754c664",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Zeigt alle Kombinationen mit Häufigkeit\n",
|
||||
"step2[['STUDY', 'LEVEL', 'PHASE']].value_counts()"
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"kernelspec": {
|
||||
"display_name": "base",
|
||||
"language": "python",
|
||||
"name": "python3"
|
||||
},
|
||||
"language_info": {
|
||||
"codemirror_mode": {
|
||||
"name": "ipython",
|
||||
"version": 3
|
||||
},
|
||||
"file_extension": ".py",
|
||||
"mimetype": "text/x-python",
|
||||
"name": "python",
|
||||
"nbconvert_exporter": "python",
|
||||
"pygments_lexer": "ipython3",
|
||||
"version": "3.11.5"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 5
|
||||
}
|
||||
Reference in New Issue
Block a user