diff --git a/lab-hyper-tuning.ipynb b/lab-hyper-tuning.ipynb
index 847d487..d4101c5 100644
--- a/lab-hyper-tuning.ipynb
+++ b/lab-hyper-tuning.ipynb
@@ -36,7 +36,14 @@
{
"cell_type": "code",
"execution_count": 1,
- "metadata": {},
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-08-13T18:33:41.912130Z",
+ "iopub.status.busy": "2026-08-13T18:33:41.911762Z",
+ "iopub.status.idle": "2026-08-13T18:33:43.922017Z",
+ "shell.execute_reply": "2026-08-13T18:33:43.921611Z"
+ }
+ },
"outputs": [],
"source": [
"#Libraries\n",
@@ -48,7 +55,14 @@
{
"cell_type": "code",
"execution_count": 2,
- "metadata": {},
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-08-13T18:33:43.924611Z",
+ "iopub.status.busy": "2026-08-13T18:33:43.924367Z",
+ "iopub.status.idle": "2026-08-13T18:33:44.165586Z",
+ "shell.execute_reply": "2026-08-13T18:33:44.165257Z"
+ }
+ },
"outputs": [
{
"data": {
@@ -221,11 +235,34 @@
},
{
"cell_type": "code",
- "execution_count": 9,
- "metadata": {},
+ "execution_count": 3,
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-08-13T18:33:44.202056Z",
+ "iopub.status.busy": "2026-08-13T18:33:44.201827Z",
+ "iopub.status.idle": "2026-08-13T18:33:44.222111Z",
+ "shell.execute_reply": "2026-08-13T18:33:44.221728Z"
+ }
+ },
"outputs": [],
"source": [
- "#your code here"
+ "spaceship = spaceship.dropna()\n",
+ "spaceship[\"Cabin\"] = spaceship[\"Cabin\"].str.split(\"/\").str[0]\n",
+ "spaceship = spaceship.drop(columns=[\"PassengerId\", \"Name\"])\n",
+ "spaceship = pd.get_dummies(spaceship, dtype=int)\n",
+ "\n",
+ "X = spaceship.drop(columns=\"Transported\")\n",
+ "y = spaceship[\"Transported\"]\n",
+ "\n",
+ "X_train, X_test, y_train, y_test = train_test_split(\n",
+ " X, y, test_size=0.2, random_state=42\n",
+ ")\n",
+ "\n",
+ "from sklearn.preprocessing import StandardScaler\n",
+ "\n",
+ "scaler = StandardScaler()\n",
+ "X_train_scaled = scaler.fit_transform(X_train)\n",
+ "X_test_scaled = scaler.transform(X_test)"
]
},
{
@@ -237,11 +274,754 @@
},
{
"cell_type": "code",
- "execution_count": null,
- "metadata": {},
- "outputs": [],
+ "execution_count": 4,
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-08-13T18:33:44.224790Z",
+ "iopub.status.busy": "2026-08-13T18:33:44.224610Z",
+ "iopub.status.idle": "2026-08-13T18:33:44.998853Z",
+ "shell.execute_reply": "2026-08-13T18:33:44.998472Z"
+ }
+ },
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "
RandomForestClassifier(random_state=42) In a Jupyter environment, please rerun this cell to show the HTML representation or trust the notebook. On GitHub, the HTML representation is unable to render, please try loading this page with nbviewer.org. \n",
+ "
\n",
+ "
\n",
+ " Parameters \n",
+ " \n",
+ " \n",
+ " \n",
+ " \n",
+ " \n",
+ " n_estimators \n",
+ " 100 \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " criterion \n",
+ " 'gini' \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " max_depth \n",
+ " None \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " min_samples_split \n",
+ " 2 \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " min_samples_leaf \n",
+ " 1 \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " min_weight_fraction_leaf \n",
+ " 0.0 \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " max_features \n",
+ " 'sqrt' \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " max_leaf_nodes \n",
+ " None \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " min_impurity_decrease \n",
+ " 0.0 \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " bootstrap \n",
+ " True \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " oob_score \n",
+ " False \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " n_jobs \n",
+ " None \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " random_state \n",
+ " 42 \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " verbose \n",
+ " 0 \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " warm_start \n",
+ " False \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " class_weight \n",
+ " None \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " ccp_alpha \n",
+ " 0.0 \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " max_samples \n",
+ " None \n",
+ " \n",
+ " \n",
+ "\n",
+ " \n",
+ " \n",
+ " monotonic_cst \n",
+ " None \n",
+ " \n",
+ " \n",
+ " \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
"
+ ],
+ "text/plain": [
+ "RandomForestClassifier(random_state=42)"
+ ]
+ },
+ "execution_count": 4,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
- "#your code here"
+ "from sklearn.ensemble import RandomForestClassifier\n",
+ "\n",
+ "random_forest = RandomForestClassifier(random_state=42)\n",
+ "random_forest.fit(X_train_scaled, y_train)"
]
},
{
@@ -253,11 +1033,30 @@
},
{
"cell_type": "code",
- "execution_count": 1,
- "metadata": {},
- "outputs": [],
+ "execution_count": 5,
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-08-13T18:33:45.000911Z",
+ "iopub.status.busy": "2026-08-13T18:33:45.000754Z",
+ "iopub.status.idle": "2026-08-13T18:33:45.024274Z",
+ "shell.execute_reply": "2026-08-13T18:33:45.023945Z"
+ }
+ },
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "Baseline accuracy: 0.810\n"
+ ]
+ }
+ ],
"source": [
- "#your code here"
+ "from sklearn.metrics import accuracy_score\n",
+ "\n",
+ "baseline_predictions = random_forest.predict(X_test_scaled)\n",
+ "baseline_accuracy = accuracy_score(y_test, baseline_predictions)\n",
+ "print(f\"Baseline accuracy: {baseline_accuracy:.3f}\")"
]
},
{
@@ -283,11 +1082,22 @@
},
{
"cell_type": "code",
- "execution_count": null,
- "metadata": {},
+ "execution_count": 6,
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-08-13T18:33:45.026313Z",
+ "iopub.status.busy": "2026-08-13T18:33:45.026156Z",
+ "iopub.status.idle": "2026-08-13T18:33:45.028157Z",
+ "shell.execute_reply": "2026-08-13T18:33:45.027851Z"
+ }
+ },
"outputs": [],
"source": [
- "#your code here"
+ "param_grid = {\n",
+ " \"n_estimators\": [50, 100],\n",
+ " \"max_depth\": [None, 10],\n",
+ " \"min_samples_split\": [2, 5]\n",
+ "}"
]
},
{
@@ -299,10 +1109,39 @@
},
{
"cell_type": "code",
- "execution_count": null,
- "metadata": {},
- "outputs": [],
- "source": []
+ "execution_count": 7,
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-08-13T18:33:45.030399Z",
+ "iopub.status.busy": "2026-08-13T18:33:45.030267Z",
+ "iopub.status.idle": "2026-08-13T18:33:50.603192Z",
+ "shell.execute_reply": "2026-08-13T18:33:50.602787Z"
+ }
+ },
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "{'max_depth': 10, 'min_samples_split': 5, 'n_estimators': 100}"
+ ]
+ },
+ "execution_count": 7,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
+ "source": [
+ "from sklearn.model_selection import GridSearchCV\n",
+ "\n",
+ "grid_search = GridSearchCV(\n",
+ " RandomForestClassifier(random_state=42),\n",
+ " param_grid,\n",
+ " cv=3,\n",
+ " scoring=\"accuracy\"\n",
+ ")\n",
+ "grid_search.fit(X_train_scaled, y_train)\n",
+ "grid_search.best_params_"
+ ]
},
{
"cell_type": "markdown",
@@ -313,10 +1152,32 @@
},
{
"cell_type": "code",
- "execution_count": null,
- "metadata": {},
- "outputs": [],
- "source": []
+ "execution_count": 8,
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-08-13T18:33:50.605418Z",
+ "iopub.status.busy": "2026-08-13T18:33:50.605271Z",
+ "iopub.status.idle": "2026-08-13T18:33:50.624917Z",
+ "shell.execute_reply": "2026-08-13T18:33:50.624536Z"
+ }
+ },
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "Tuned model accuracy: 0.808\n",
+ "Accuracy change: -0.23 percentage points\n"
+ ]
+ }
+ ],
+ "source": [
+ "tuned_predictions = grid_search.best_estimator_.predict(X_test_scaled)\n",
+ "tuned_accuracy = accuracy_score(y_test, tuned_predictions)\n",
+ "\n",
+ "print(f\"Tuned model accuracy: {tuned_accuracy:.3f}\")\n",
+ "print(f\"Accuracy change: {(tuned_accuracy - baseline_accuracy) * 100:+.2f} percentage points\")"
+ ]
}
],
"metadata": {
@@ -335,7 +1196,7 @@
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
- "version": "3.10.9"
+ "version": "3.13.9"
}
},
"nbformat": 4,