diff --git a/datasets/Titanic.ipynb b/datasets/Titanic.ipynb index 0bd2531..15ba729 100644 --- a/datasets/Titanic.ipynb +++ b/datasets/Titanic.ipynb @@ -10,7 +10,7 @@ }, { "cell_type": "code", - "execution_count": 6, + "execution_count": 7, "id": "ecd595e6-4de0-45f1-8022-92ad3c47e540", "metadata": {}, "outputs": [ @@ -49,7 +49,13 @@ "\n", "# Keep only the 10 core features used by TensorFlow tutorials\n", "final_features = ['survived', 'sex', 'age', 'n_siblings_spouses', 'parch', 'fare', 'class', 'deck', 'embark_town', 'alone']\n", - "processed_df = df[final_features].dropna(subset=['age', 'fare']).reset_index(drop=True)" + "processed_df = df[final_features].dropna(subset=['age', 'fare']).reset_index(drop=True)\n", + "\n", + "# Generate identical train (627 rows) and eval (264 rows) splits\n", + "np.random.seed(42) \n", + "mask = np.random.rand(len(processed_df)) < 0.704\n", + "train_df = processed_df[mask]\n", + "eval_df = processed_df[~mask]" ] }, {