- changed MAD Outlier Removal - Function
- added regulation to xgboost
This commit is contained in:
@@ -8,6 +8,64 @@
|
||||
"Im folgenden wird auf die Daten das MAD Outlier removal angewendet."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "46bd036d",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import numpy as np\n",
|
||||
"import pandas as pd\n",
|
||||
"\n",
|
||||
"def calculate_mad_params(df, columns):\n",
|
||||
" \"\"\"\n",
|
||||
" Calculate median and MAD parameters for each column.\n",
|
||||
" This should be run ONLY on the training data.\n",
|
||||
" \n",
|
||||
" Returns a dictionary: {col: (median, mad)}\n",
|
||||
" \"\"\"\n",
|
||||
" params = {}\n",
|
||||
" for col in columns:\n",
|
||||
" median = df[col].median()\n",
|
||||
" mad = np.median(np.abs(df[col] - median))\n",
|
||||
" params[col] = (median, mad)\n",
|
||||
" return params"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "e0691732",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"def apply_mad_filter(df, params, threshold=3.5):\n",
|
||||
" \"\"\"\n",
|
||||
" Apply MAD-based outlier removal using precomputed parameters.\n",
|
||||
" Works on training, validation, and test data.\n",
|
||||
" \n",
|
||||
" df: DataFrame to filter\n",
|
||||
" params: dictionary {col: (median, mad)} from training data\n",
|
||||
" threshold: cutoff for robust Z-score\n",
|
||||
" \"\"\"\n",
|
||||
" df_clean = df.copy()\n",
|
||||
"\n",
|
||||
" for col, (median, mad) in params.items():\n",
|
||||
" if mad == 0:\n",
|
||||
" continue # no spread; nothing to remove for this column\n",
|
||||
"\n",
|
||||
" robust_z = 0.6745 * (df_clean[col] - median) / mad\n",
|
||||
" outlier_mask = np.abs(robust_z) > threshold\n",
|
||||
"\n",
|
||||
" # Remove values only in this specific column\n",
|
||||
" df_clean.loc[outlier_mask, col] = np.nan\n",
|
||||
" print(df_clean.shape)\n",
|
||||
" \n",
|
||||
" print(df_clean.shape)\n",
|
||||
" return df_clean"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -15,6 +73,9 @@
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# old removal - with this we were able to get 85% accuracy\n",
|
||||
"# the values of the old validation & test data set is stored privately on the Cluster\n",
|
||||
"\n",
|
||||
"import numpy as np\n",
|
||||
"import pandas as pd\n",
|
||||
"from sklearn.preprocessing import StandardScaler, MinMaxScaler\n",
|
||||
@@ -32,8 +93,9 @@
|
||||
" continue # keine Streuung, keine Ausreißer\n",
|
||||
" robust_z = 0.6745 * (df_clean[col] - median) / mad\n",
|
||||
" mask = np.abs(robust_z) <= threshold\n",
|
||||
" df_clean = df_clean[mask]\n",
|
||||
" return df_clean"
|
||||
" output = df_clean[mask]\n",
|
||||
"\n",
|
||||
" return output"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
Reference in New Issue
Block a user