diff --git a/datasets/Titanic.ipynb b/datasets/Titanic.ipynb index f79912e..a70c61e 100644 --- a/datasets/Titanic.ipynb +++ b/datasets/Titanic.ipynb @@ -5,12 +5,12 @@ "id": "bbadefe5-d75c-4e15-8f24-689e7e30dcf2", "metadata": {}, "source": [ - "# Titanic" + "# Create Titanic's training and evaluation data files" ] }, { "cell_type": "code", - "execution_count": 4, + "execution_count": 5, "id": "ecd595e6-4de0-45f1-8022-92ad3c47e540", "metadata": {}, "outputs": [ @@ -39,7 +39,13 @@ " 'Survived': 'survived', 'Sex': 'sex', 'Age': 'age',\n", " 'SibSp': 'n_siblings_spouses', 'Parch': 'parch', 'Fare': 'fare',\n", " 'Pclass': 'class', 'Embarked': 'embark_town'\n", - "})" + "})\n", + "\n", + "# Feature mappings matching the tf-datasets distribution\n", + "df['class'] = df['class'].map({1: 'First', 2: 'Second', 3: 'Third'})\n", + "df['deck'] = df['Cabin'].dropna().astype(str).str[0].reindex(df.index, fill_value='unknown')\n", + "df['embark_town'] = df['embark_town'].map({'S': 'Southampton', 'C': 'Cherbourg', 'Q': 'Queenstown'}).fillna('unknown')\n", + "df['alone'] = np.where((df['n_siblings_spouses'] == 0) & (df['parch'] == 0), 'y', 'n')" ] }, {