- changed MAD Outlier Removal - Function

- added regulation to xgboost
This commit is contained in:
TimoKurz
2025-12-13 13:45:31 +01:00
parent 15b32a9792
commit 87c5e21daf
3 changed files with 668 additions and 38 deletions
@@ -8,6 +8,64 @@
"Im folgenden wird auf die Daten das MAD Outlier removal angewendet."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "46bd036d",
"metadata": {},
"outputs": [],
"source": [
"import numpy as np\n",
"import pandas as pd\n",
"\n",
"def calculate_mad_params(df, columns):\n",
" \"\"\"\n",
" Calculate median and MAD parameters for each column.\n",
" This should be run ONLY on the training data.\n",
" \n",
" Returns a dictionary: {col: (median, mad)}\n",
" \"\"\"\n",
" params = {}\n",
" for col in columns:\n",
" median = df[col].median()\n",
" mad = np.median(np.abs(df[col] - median))\n",
" params[col] = (median, mad)\n",
" return params"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e0691732",
"metadata": {},
"outputs": [],
"source": [
"def apply_mad_filter(df, params, threshold=3.5):\n",
" \"\"\"\n",
" Apply MAD-based outlier removal using precomputed parameters.\n",
" Works on training, validation, and test data.\n",
" \n",
" df: DataFrame to filter\n",
" params: dictionary {col: (median, mad)} from training data\n",
" threshold: cutoff for robust Z-score\n",
" \"\"\"\n",
" df_clean = df.copy()\n",
"\n",
" for col, (median, mad) in params.items():\n",
" if mad == 0:\n",
" continue # no spread; nothing to remove for this column\n",
"\n",
" robust_z = 0.6745 * (df_clean[col] - median) / mad\n",
" outlier_mask = np.abs(robust_z) > threshold\n",
"\n",
" # Remove values only in this specific column\n",
" df_clean.loc[outlier_mask, col] = np.nan\n",
" print(df_clean.shape)\n",
" \n",
" print(df_clean.shape)\n",
" return df_clean"
]
},
{
"cell_type": "code",
"execution_count": null,
@@ -15,6 +73,9 @@
"metadata": {},
"outputs": [],
"source": [
"# old removal - with this we were able to get 85% accuracy\n",
"# the values of the old validation & test data set is stored privately on the Cluster\n",
"\n",
"import numpy as np\n",
"import pandas as pd\n",
"from sklearn.preprocessing import StandardScaler, MinMaxScaler\n",
@@ -32,8 +93,9 @@
" continue # keine Streuung, keine Ausreißer\n",
" robust_z = 0.6745 * (df_clean[col] - median) / mad\n",
" mask = np.abs(robust_z) <= threshold\n",
" df_clean = df_clean[mask]\n",
" return df_clean"
" output = df_clean[mask]\n",
"\n",
" return output"
]
}
],