diff --git a/your-code/main.ipynb b/your-code/main.ipynb
index f6f69ed..1589863 100755
--- a/your-code/main.ipynb
+++ b/your-code/main.ipynb
@@ -12,12 +12,14 @@
},
{
"cell_type": "code",
- "execution_count": 174,
+ "execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"# Import your libraries:\n",
- "\n"
+ "import numpy as np\n",
+ "import scipy\n",
+ "from sklearn import datasets"
]
},
{
@@ -38,12 +40,92 @@
},
{
"cell_type": "code",
- "execution_count": 175,
- "metadata": {},
- "outputs": [],
+ "execution_count": 2,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "{'data': array([[ 0.03807591, 0.05068012, 0.06169621, ..., -0.00259226,\n",
+ " 0.01990842, -0.01764613],\n",
+ " [-0.00188202, -0.04464164, -0.05147406, ..., -0.03949338,\n",
+ " -0.06832974, -0.09220405],\n",
+ " [ 0.08529891, 0.05068012, 0.04445121, ..., -0.00259226,\n",
+ " 0.00286377, -0.02593034],\n",
+ " ...,\n",
+ " [ 0.04170844, 0.05068012, -0.01590626, ..., -0.01107952,\n",
+ " -0.04687948, 0.01549073],\n",
+ " [-0.04547248, -0.04464164, 0.03906215, ..., 0.02655962,\n",
+ " 0.04452837, -0.02593034],\n",
+ " [-0.04547248, -0.04464164, -0.0730303 , ..., -0.03949338,\n",
+ " -0.00421986, 0.00306441]]),\n",
+ " 'target': array([151., 75., 141., 206., 135., 97., 138., 63., 110., 310., 101.,\n",
+ " 69., 179., 185., 118., 171., 166., 144., 97., 168., 68., 49.,\n",
+ " 68., 245., 184., 202., 137., 85., 131., 283., 129., 59., 341.,\n",
+ " 87., 65., 102., 265., 276., 252., 90., 100., 55., 61., 92.,\n",
+ " 259., 53., 190., 142., 75., 142., 155., 225., 59., 104., 182.,\n",
+ " 128., 52., 37., 170., 170., 61., 144., 52., 128., 71., 163.,\n",
+ " 150., 97., 160., 178., 48., 270., 202., 111., 85., 42., 170.,\n",
+ " 200., 252., 113., 143., 51., 52., 210., 65., 141., 55., 134.,\n",
+ " 42., 111., 98., 164., 48., 96., 90., 162., 150., 279., 92.,\n",
+ " 83., 128., 102., 302., 198., 95., 53., 134., 144., 232., 81.,\n",
+ " 104., 59., 246., 297., 258., 229., 275., 281., 179., 200., 200.,\n",
+ " 173., 180., 84., 121., 161., 99., 109., 115., 268., 274., 158.,\n",
+ " 107., 83., 103., 272., 85., 280., 336., 281., 118., 317., 235.,\n",
+ " 60., 174., 259., 178., 128., 96., 126., 288., 88., 292., 71.,\n",
+ " 197., 186., 25., 84., 96., 195., 53., 217., 172., 131., 214.,\n",
+ " 59., 70., 220., 268., 152., 47., 74., 295., 101., 151., 127.,\n",
+ " 237., 225., 81., 151., 107., 64., 138., 185., 265., 101., 137.,\n",
+ " 143., 141., 79., 292., 178., 91., 116., 86., 122., 72., 129.,\n",
+ " 142., 90., 158., 39., 196., 222., 277., 99., 196., 202., 155.,\n",
+ " 77., 191., 70., 73., 49., 65., 263., 248., 296., 214., 185.,\n",
+ " 78., 93., 252., 150., 77., 208., 77., 108., 160., 53., 220.,\n",
+ " 154., 259., 90., 246., 124., 67., 72., 257., 262., 275., 177.,\n",
+ " 71., 47., 187., 125., 78., 51., 258., 215., 303., 243., 91.,\n",
+ " 150., 310., 153., 346., 63., 89., 50., 39., 103., 308., 116.,\n",
+ " 145., 74., 45., 115., 264., 87., 202., 127., 182., 241., 66.,\n",
+ " 94., 283., 64., 102., 200., 265., 94., 230., 181., 156., 233.,\n",
+ " 60., 219., 80., 68., 332., 248., 84., 200., 55., 85., 89.,\n",
+ " 31., 129., 83., 275., 65., 198., 236., 253., 124., 44., 172.,\n",
+ " 114., 142., 109., 180., 144., 163., 147., 97., 220., 190., 109.,\n",
+ " 191., 122., 230., 242., 248., 249., 192., 131., 237., 78., 135.,\n",
+ " 244., 199., 270., 164., 72., 96., 306., 91., 214., 95., 216.,\n",
+ " 263., 178., 113., 200., 139., 139., 88., 148., 88., 243., 71.,\n",
+ " 77., 109., 272., 60., 54., 221., 90., 311., 281., 182., 321.,\n",
+ " 58., 262., 206., 233., 242., 123., 167., 63., 197., 71., 168.,\n",
+ " 140., 217., 121., 235., 245., 40., 52., 104., 132., 88., 69.,\n",
+ " 219., 72., 201., 110., 51., 277., 63., 118., 69., 273., 258.,\n",
+ " 43., 198., 242., 232., 175., 93., 168., 275., 293., 281., 72.,\n",
+ " 140., 189., 181., 209., 136., 261., 113., 131., 174., 257., 55.,\n",
+ " 84., 42., 146., 212., 233., 91., 111., 152., 120., 67., 310.,\n",
+ " 94., 183., 66., 173., 72., 49., 64., 48., 178., 104., 132.,\n",
+ " 220., 57.]),\n",
+ " 'frame': None,\n",
+ " 'DESCR': '.. _diabetes_dataset:\\n\\nDiabetes dataset\\n----------------\\n\\nTen baseline variables, age, sex, body mass index, average blood\\npressure, and six blood serum measurements were obtained for each of n =\\n442 diabetes patients, as well as the response of interest, a\\nquantitative measure of disease progression one year after baseline.\\n\\n**Data Set Characteristics:**\\n\\n :Number of Instances: 442\\n\\n :Number of Attributes: First 10 columns are numeric predictive values\\n\\n :Target: Column 11 is a quantitative measure of disease progression one year after baseline\\n\\n :Attribute Information:\\n - age age in years\\n - sex\\n - bmi body mass index\\n - bp average blood pressure\\n - s1 tc, T-Cells (a type of white blood cells)\\n - s2 ldl, low-density lipoproteins\\n - s3 hdl, high-density lipoproteins\\n - s4 tch, thyroid stimulating hormone\\n - s5 ltg, lamotrigine\\n - s6 glu, blood sugar level\\n\\nNote: Each of these 10 feature variables have been mean centered and scaled by the standard deviation times `n_samples` (i.e. the sum of squares of each column totals 1).\\n\\nSource URL:\\nhttps://www4.stat.ncsu.edu/~boos/var.select/diabetes.html\\n\\nFor more information see:\\nBradley Efron, Trevor Hastie, Iain Johnstone and Robert Tibshirani (2004) \"Least Angle Regression,\" Annals of Statistics (with discussion), 407-499.\\n(https://web.stanford.edu/~hastie/Papers/LARS/LeastAngle_2002.pdf)',\n",
+ " 'feature_names': ['age',\n",
+ " 'sex',\n",
+ " 'bmi',\n",
+ " 'bp',\n",
+ " 's1',\n",
+ " 's2',\n",
+ " 's3',\n",
+ " 's4',\n",
+ " 's5',\n",
+ " 's6'],\n",
+ " 'data_filename': '/home/vdiazpliego/anaconda3/lib/python3.8/site-packages/sklearn/datasets/data/diabetes_data.csv.gz',\n",
+ " 'target_filename': '/home/vdiazpliego/anaconda3/lib/python3.8/site-packages/sklearn/datasets/data/diabetes_target.csv.gz'}"
+ ]
+ },
+ "execution_count": 2,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
"# Your code here:\n",
- "\n"
+ "\n",
+ "diabetes = datasets.load_diabetes()\n",
+ "diabetes"
]
},
{
@@ -55,12 +137,23 @@
},
{
"cell_type": "code",
- "execution_count": 176,
- "metadata": {},
- "outputs": [],
+ "execution_count": 3,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "dict_keys(['data', 'target', 'frame', 'DESCR', 'feature_names', 'data_filename', 'target_filename'])"
+ ]
+ },
+ "execution_count": 3,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
"# Your code here:\n",
- "\n"
+ "diabetes.keys()\n"
]
},
{
@@ -72,11 +165,23 @@
},
{
"cell_type": "code",
- "execution_count": 177,
- "metadata": {},
- "outputs": [],
+ "execution_count": 4,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "'.. _diabetes_dataset:\\n\\nDiabetes dataset\\n----------------\\n\\nTen baseline variables, age, sex, body mass index, average blood\\npressure, and six blood serum measurements were obtained for each of n =\\n442 diabetes patients, as well as the response of interest, a\\nquantitative measure of disease progression one year after baseline.\\n\\n**Data Set Characteristics:**\\n\\n :Number of Instances: 442\\n\\n :Number of Attributes: First 10 columns are numeric predictive values\\n\\n :Target: Column 11 is a quantitative measure of disease progression one year after baseline\\n\\n :Attribute Information:\\n - age age in years\\n - sex\\n - bmi body mass index\\n - bp average blood pressure\\n - s1 tc, T-Cells (a type of white blood cells)\\n - s2 ldl, low-density lipoproteins\\n - s3 hdl, high-density lipoproteins\\n - s4 tch, thyroid stimulating hormone\\n - s5 ltg, lamotrigine\\n - s6 glu, blood sugar level\\n\\nNote: Each of these 10 feature variables have been mean centered and scaled by the standard deviation times `n_samples` (i.e. the sum of squares of each column totals 1).\\n\\nSource URL:\\nhttps://www4.stat.ncsu.edu/~boos/var.select/diabetes.html\\n\\nFor more information see:\\nBradley Efron, Trevor Hastie, Iain Johnstone and Robert Tibshirani (2004) \"Least Angle Regression,\" Annals of Statistics (with discussion), 407-499.\\n(https://web.stanford.edu/~hastie/Papers/LARS/LeastAngle_2002.pdf)'"
+ ]
+ },
+ "execution_count": 4,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
- "# Your code here:\n"
+ "# Your code here:\n",
+ "diabetes.DESCR"
]
},
{
@@ -104,12 +209,232 @@
},
{
"cell_type": "code",
- "execution_count": 178,
- "metadata": {},
- "outputs": [],
+ "execution_count": 7,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "
\n",
+ "\n",
+ "
\n",
+ " \n",
+ " \n",
+ " | \n",
+ " 0 | \n",
+ " 1 | \n",
+ " 2 | \n",
+ " 3 | \n",
+ " 4 | \n",
+ " 5 | \n",
+ " 6 | \n",
+ " 7 | \n",
+ " 8 | \n",
+ " 9 | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 0 | \n",
+ " 0.038076 | \n",
+ " 0.050680 | \n",
+ " 0.061696 | \n",
+ " 0.021872 | \n",
+ " -0.044223 | \n",
+ " -0.034821 | \n",
+ " -0.043401 | \n",
+ " -0.002592 | \n",
+ " 0.019908 | \n",
+ " -0.017646 | \n",
+ "
\n",
+ " \n",
+ " | 1 | \n",
+ " -0.001882 | \n",
+ " -0.044642 | \n",
+ " -0.051474 | \n",
+ " -0.026328 | \n",
+ " -0.008449 | \n",
+ " -0.019163 | \n",
+ " 0.074412 | \n",
+ " -0.039493 | \n",
+ " -0.068330 | \n",
+ " -0.092204 | \n",
+ "
\n",
+ " \n",
+ " | 2 | \n",
+ " 0.085299 | \n",
+ " 0.050680 | \n",
+ " 0.044451 | \n",
+ " -0.005671 | \n",
+ " -0.045599 | \n",
+ " -0.034194 | \n",
+ " -0.032356 | \n",
+ " -0.002592 | \n",
+ " 0.002864 | \n",
+ " -0.025930 | \n",
+ "
\n",
+ " \n",
+ " | 3 | \n",
+ " -0.089063 | \n",
+ " -0.044642 | \n",
+ " -0.011595 | \n",
+ " -0.036656 | \n",
+ " 0.012191 | \n",
+ " 0.024991 | \n",
+ " -0.036038 | \n",
+ " 0.034309 | \n",
+ " 0.022692 | \n",
+ " -0.009362 | \n",
+ "
\n",
+ " \n",
+ " | 4 | \n",
+ " 0.005383 | \n",
+ " -0.044642 | \n",
+ " -0.036385 | \n",
+ " 0.021872 | \n",
+ " 0.003935 | \n",
+ " 0.015596 | \n",
+ " 0.008142 | \n",
+ " -0.002592 | \n",
+ " -0.031991 | \n",
+ " -0.046641 | \n",
+ "
\n",
+ " \n",
+ " | ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ " ... | \n",
+ "
\n",
+ " \n",
+ " | 437 | \n",
+ " 0.041708 | \n",
+ " 0.050680 | \n",
+ " 0.019662 | \n",
+ " 0.059744 | \n",
+ " -0.005697 | \n",
+ " -0.002566 | \n",
+ " -0.028674 | \n",
+ " -0.002592 | \n",
+ " 0.031193 | \n",
+ " 0.007207 | \n",
+ "
\n",
+ " \n",
+ " | 438 | \n",
+ " -0.005515 | \n",
+ " 0.050680 | \n",
+ " -0.015906 | \n",
+ " -0.067642 | \n",
+ " 0.049341 | \n",
+ " 0.079165 | \n",
+ " -0.028674 | \n",
+ " 0.034309 | \n",
+ " -0.018118 | \n",
+ " 0.044485 | \n",
+ "
\n",
+ " \n",
+ " | 439 | \n",
+ " 0.041708 | \n",
+ " 0.050680 | \n",
+ " -0.015906 | \n",
+ " 0.017282 | \n",
+ " -0.037344 | \n",
+ " -0.013840 | \n",
+ " -0.024993 | \n",
+ " -0.011080 | \n",
+ " -0.046879 | \n",
+ " 0.015491 | \n",
+ "
\n",
+ " \n",
+ " | 440 | \n",
+ " -0.045472 | \n",
+ " -0.044642 | \n",
+ " 0.039062 | \n",
+ " 0.001215 | \n",
+ " 0.016318 | \n",
+ " 0.015283 | \n",
+ " -0.028674 | \n",
+ " 0.026560 | \n",
+ " 0.044528 | \n",
+ " -0.025930 | \n",
+ "
\n",
+ " \n",
+ " | 441 | \n",
+ " -0.045472 | \n",
+ " -0.044642 | \n",
+ " -0.073030 | \n",
+ " -0.081414 | \n",
+ " 0.083740 | \n",
+ " 0.027809 | \n",
+ " 0.173816 | \n",
+ " -0.039493 | \n",
+ " -0.004220 | \n",
+ " 0.003064 | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
442 rows × 10 columns
\n",
+ "
"
+ ],
+ "text/plain": [
+ " 0 1 2 3 4 5 6 \\\n",
+ "0 0.038076 0.050680 0.061696 0.021872 -0.044223 -0.034821 -0.043401 \n",
+ "1 -0.001882 -0.044642 -0.051474 -0.026328 -0.008449 -0.019163 0.074412 \n",
+ "2 0.085299 0.050680 0.044451 -0.005671 -0.045599 -0.034194 -0.032356 \n",
+ "3 -0.089063 -0.044642 -0.011595 -0.036656 0.012191 0.024991 -0.036038 \n",
+ "4 0.005383 -0.044642 -0.036385 0.021872 0.003935 0.015596 0.008142 \n",
+ ".. ... ... ... ... ... ... ... \n",
+ "437 0.041708 0.050680 0.019662 0.059744 -0.005697 -0.002566 -0.028674 \n",
+ "438 -0.005515 0.050680 -0.015906 -0.067642 0.049341 0.079165 -0.028674 \n",
+ "439 0.041708 0.050680 -0.015906 0.017282 -0.037344 -0.013840 -0.024993 \n",
+ "440 -0.045472 -0.044642 0.039062 0.001215 0.016318 0.015283 -0.028674 \n",
+ "441 -0.045472 -0.044642 -0.073030 -0.081414 0.083740 0.027809 0.173816 \n",
+ "\n",
+ " 7 8 9 \n",
+ "0 -0.002592 0.019908 -0.017646 \n",
+ "1 -0.039493 -0.068330 -0.092204 \n",
+ "2 -0.002592 0.002864 -0.025930 \n",
+ "3 0.034309 0.022692 -0.009362 \n",
+ "4 -0.002592 -0.031991 -0.046641 \n",
+ ".. ... ... ... \n",
+ "437 -0.002592 0.031193 0.007207 \n",
+ "438 0.034309 -0.018118 0.044485 \n",
+ "439 -0.011080 -0.046879 0.015491 \n",
+ "440 0.026560 0.044528 -0.025930 \n",
+ "441 -0.039493 -0.004220 0.003064 \n",
+ "\n",
+ "[442 rows x 10 columns]"
+ ]
+ },
+ "execution_count": 7,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
"# Your code here:\n",
- "\n"
+ "import pandas as pd\n",
+ "diabetes.data.shape\n",
+ "df = pd.DataFrame(diabetes.data)\n",
+ "df"
]
},
{
@@ -130,12 +455,15 @@
},
{
"cell_type": "code",
- "execution_count": 179,
+ "execution_count": 23,
"metadata": {},
"outputs": [],
"source": [
"# Your code here:\n",
- "\n"
+ "from sklearn.feature_selection import RFE\n",
+ "\n",
+ "from sklearn.model_selection import train_test_split as tts\n",
+ "from sklearn.linear_model import LinearRegression as LinReg"
]
},
{
@@ -147,12 +475,12 @@
},
{
"cell_type": "code",
- "execution_count": 180,
+ "execution_count": 24,
"metadata": {},
"outputs": [],
"source": [
"# Your code here:\n",
- "\n"
+ "diabetes_model=LinReg()\n"
]
},
{
@@ -164,12 +492,47 @@
},
{
"cell_type": "code",
- "execution_count": 181,
- "metadata": {},
- "outputs": [],
+ "execution_count": 25,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "array([ -10.01219782, -239.81908937, 519.83978679, 324.39042769,\n",
+ " -792.18416163, 476.74583782, 101.04457032, 177.06417623,\n",
+ " 751.27932109, 67.62538639])"
+ ]
+ },
+ "execution_count": 25,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
"# Your code here:\n",
- "\n"
+ "\n",
+ "model = diabetes_model.fit(diabetes.data , diabetes.target)\n",
+ "model.coef_"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 26,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "152.1334841628965"
+ ]
+ },
+ "execution_count": 26,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
+ "source": [
+ "model.intercept_\n"
]
},
{
@@ -192,7 +555,7 @@
},
{
"cell_type": "code",
- "execution_count": 182,
+ "execution_count": 11,
"metadata": {},
"outputs": [],
"source": [
@@ -218,12 +581,12 @@
},
{
"cell_type": "code",
- "execution_count": 183,
+ "execution_count": 14,
"metadata": {},
"outputs": [],
"source": [
"# Your code here:\n",
- "\n",
+ "import pandas as pd\n",
"auto = pd.read_csv('../auto-mpg.csv')"
]
},
@@ -236,12 +599,124 @@
},
{
"cell_type": "code",
- "execution_count": 184,
- "metadata": {},
- "outputs": [],
+ "execution_count": 15,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/html": [
+ "\n",
+ "\n",
+ "
\n",
+ " \n",
+ " \n",
+ " | \n",
+ " mpg | \n",
+ " cylinders | \n",
+ " displacement | \n",
+ " horse_power | \n",
+ " weight | \n",
+ " acceleration | \n",
+ " model_year | \n",
+ " car_name | \n",
+ "
\n",
+ " \n",
+ " \n",
+ " \n",
+ " | 0 | \n",
+ " 18.0 | \n",
+ " 8 | \n",
+ " 307.0 | \n",
+ " 130.0 | \n",
+ " 3504 | \n",
+ " 12.0 | \n",
+ " 70 | \n",
+ " \\t\"chevrolet chevelle malibu\" | \n",
+ "
\n",
+ " \n",
+ " | 1 | \n",
+ " 15.0 | \n",
+ " 8 | \n",
+ " 350.0 | \n",
+ " 165.0 | \n",
+ " 3693 | \n",
+ " 11.5 | \n",
+ " 70 | \n",
+ " \\t\"buick skylark 320\" | \n",
+ "
\n",
+ " \n",
+ " | 2 | \n",
+ " 18.0 | \n",
+ " 8 | \n",
+ " 318.0 | \n",
+ " 150.0 | \n",
+ " 3436 | \n",
+ " 11.0 | \n",
+ " 70 | \n",
+ " \\t\"plymouth satellite\" | \n",
+ "
\n",
+ " \n",
+ " | 3 | \n",
+ " 16.0 | \n",
+ " 8 | \n",
+ " 304.0 | \n",
+ " 150.0 | \n",
+ " 3433 | \n",
+ " 12.0 | \n",
+ " 70 | \n",
+ " \\t\"amc rebel sst\" | \n",
+ "
\n",
+ " \n",
+ " | 4 | \n",
+ " 17.0 | \n",
+ " 8 | \n",
+ " 302.0 | \n",
+ " 140.0 | \n",
+ " 3449 | \n",
+ " 10.5 | \n",
+ " 70 | \n",
+ " \\t\"ford torino\" | \n",
+ "
\n",
+ " \n",
+ "
\n",
+ "
"
+ ],
+ "text/plain": [
+ " mpg cylinders displacement horse_power weight acceleration \\\n",
+ "0 18.0 8 307.0 130.0 3504 12.0 \n",
+ "1 15.0 8 350.0 165.0 3693 11.5 \n",
+ "2 18.0 8 318.0 150.0 3436 11.0 \n",
+ "3 16.0 8 304.0 150.0 3433 12.0 \n",
+ "4 17.0 8 302.0 140.0 3449 10.5 \n",
+ "\n",
+ " model_year car_name \n",
+ "0 70 \\t\"chevrolet chevelle malibu\" \n",
+ "1 70 \\t\"buick skylark 320\" \n",
+ "2 70 \\t\"plymouth satellite\" \n",
+ "3 70 \\t\"amc rebel sst\" \n",
+ "4 70 \\t\"ford torino\" "
+ ]
+ },
+ "execution_count": 15,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
"# Your code here:\n",
- "\n"
+ "auto.head()\n"
]
},
{
@@ -253,12 +728,35 @@
},
{
"cell_type": "code",
- "execution_count": 185,
- "metadata": {},
- "outputs": [],
+ "execution_count": 16,
+ "metadata": {},
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "\n",
+ "RangeIndex: 398 entries, 0 to 397\n",
+ "Data columns (total 8 columns):\n",
+ " # Column Non-Null Count Dtype \n",
+ "--- ------ -------------- ----- \n",
+ " 0 mpg 398 non-null float64\n",
+ " 1 cylinders 398 non-null int64 \n",
+ " 2 displacement 398 non-null float64\n",
+ " 3 horse_power 392 non-null float64\n",
+ " 4 weight 398 non-null int64 \n",
+ " 5 acceleration 398 non-null float64\n",
+ " 6 model_year 398 non-null int64 \n",
+ " 7 car_name 398 non-null object \n",
+ "dtypes: float64(4), int64(3), object(1)\n",
+ "memory usage: 25.0+ KB\n"
+ ]
+ }
+ ],
"source": [
"# Your code here:\n",
- "\n"
+ "\n",
+ "auto.info()"
]
},
{
@@ -270,12 +768,32 @@
},
{
"cell_type": "code",
- "execution_count": 186,
- "metadata": {},
- "outputs": [],
+ "execution_count": 17,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "count 398.000000\n",
+ "mean 76.010050\n",
+ "std 3.697627\n",
+ "min 70.000000\n",
+ "25% 73.000000\n",
+ "50% 76.000000\n",
+ "75% 79.000000\n",
+ "max 82.000000\n",
+ "Name: model_year, dtype: float64"
+ ]
+ },
+ "execution_count": 17,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
"# Your code here:\n",
- "\n"
+ "auto.model_year.describe()\n",
+ "#el mas moderno del año 82, y el más antiguo del año 70"
]
},
{
@@ -287,12 +805,38 @@
},
{
"cell_type": "code",
- "execution_count": 187,
- "metadata": {},
- "outputs": [],
+ "execution_count": 18,
+ "metadata": {},
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "\n",
+ "Int64Index: 392 entries, 0 to 397\n",
+ "Data columns (total 8 columns):\n",
+ " # Column Non-Null Count Dtype \n",
+ "--- ------ -------------- ----- \n",
+ " 0 mpg 392 non-null float64\n",
+ " 1 cylinders 392 non-null int64 \n",
+ " 2 displacement 392 non-null float64\n",
+ " 3 horse_power 392 non-null float64\n",
+ " 4 weight 392 non-null int64 \n",
+ " 5 acceleration 392 non-null float64\n",
+ " 6 model_year 392 non-null int64 \n",
+ " 7 car_name 392 non-null object \n",
+ "dtypes: float64(4), int64(3), object(1)\n",
+ "memory usage: 27.6+ KB\n"
+ ]
+ }
+ ],
"source": [
"# Your code here:\n",
- "\n"
+ "nan_cols = auto.isna().sum()\n",
+ "nan_cols[nan_cols>0]\n",
+ "\n",
+ "auto.dropna(inplace=True)\n",
+ "auto.info()\n"
]
},
{
@@ -304,12 +848,30 @@
},
{
"cell_type": "code",
- "execution_count": 188,
- "metadata": {},
- "outputs": [],
+ "execution_count": 20,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "4 199\n",
+ "8 103\n",
+ "6 83\n",
+ "3 4\n",
+ "5 3\n",
+ "Name: cylinders, dtype: int64"
+ ]
+ },
+ "execution_count": 20,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
"# Your code here:\n",
- "\n"
+ "\n",
+ "auto.cylinders.value_counts()\n",
+ "#exiten 5 posibles valores de cilindros. (4,8,6,3,5)"
]
},
{
@@ -323,31 +885,36 @@
},
{
"cell_type": "code",
- "execution_count": 189,
+ "execution_count": 30,
"metadata": {},
"outputs": [],
"source": [
- "# Import the necessary function\n",
- "\n"
+ "X=auto.drop(columns=['mpg'])._get_numeric_data()\n",
+ "\n",
+ "y=auto.mpg"
]
},
{
"cell_type": "code",
- "execution_count": 190,
+ "execution_count": 32,
"metadata": {},
- "outputs": [],
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "((313, 6), (313,), (79, 6), (79,))"
+ ]
+ },
+ "execution_count": 32,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
- "# Your code here:\n",
- "\n"
+ "X_train, X_test, y_train, y_test = tts(X, y, test_size=0.2, train_size=0.8, random_state=42)\n",
+ "X_train.shape, y_train.shape, X_test.shape, y_test.shape\n"
]
},
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {},
- "outputs": [],
- "source": []
- },
{
"cell_type": "markdown",
"metadata": {},
@@ -357,12 +924,33 @@
},
{
"cell_type": "code",
- "execution_count": 191,
- "metadata": {},
- "outputs": [],
+ "execution_count": 34,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "mpg True\n",
+ "cylinders True\n",
+ "displacement True\n",
+ "horse_power True\n",
+ "weight True\n",
+ "acceleration True\n",
+ "model_year True\n",
+ "dtype: bool"
+ ]
+ },
+ "execution_count": 34,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
"# Your code here:\n",
- "\n"
+ "from sklearn.linear_model import LinearRegression as LinReg\n",
+ "\n",
+ "from sklearn.metrics import r2_score\n",
+ "np.isfinite(auto.all())\n"
]
},
{
@@ -374,14 +962,66 @@
},
{
"cell_type": "code",
- "execution_count": 192,
+ "execution_count": 36,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "0"
+ ]
+ },
+ "execution_count": 36,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
+ "source": [
+ "# Your code here:\n",
+ "X_train.isnull().count(),\n",
+ "y_train.isnull().sum()\n",
+ "\n"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 38,
"metadata": {},
"outputs": [],
"source": [
- "# Your code here:\n",
+ "X_train.horse_power.fillna(X_train.horse_power.mean(), inplace= True)\n",
+ "X_test.horse_power.fillna(X_test.horse_power.mean(), inplace= True)\n",
"\n"
]
},
+ {
+ "cell_type": "code",
+ "execution_count": 40,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "LinearRegression()"
+ ]
+ },
+ "execution_count": 40,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
+ "source": [
+ "linreg=LinReg()\n",
+ "linreg.fit(X_train, y_train)\n"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "metadata": {},
+ "outputs": [],
+ "source": []
+ },
{
"cell_type": "markdown",
"metadata": {},
@@ -395,21 +1035,35 @@
},
{
"cell_type": "code",
- "execution_count": 193,
+ "execution_count": 27,
"metadata": {},
"outputs": [],
"source": [
"# Import the necessary function:\n",
- "\n"
+ "\n",
+ "from sklearn.metrics import r2_score\n"
]
},
{
"cell_type": "code",
- "execution_count": 194,
- "metadata": {},
- "outputs": [],
+ "execution_count": 42,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "0.8037468831261813"
+ ]
+ },
+ "execution_count": 42,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
"# Your code here:\n",
+ "y_pred=linreg.predict(X_test)\n",
+ "r2_score(y_pred, y_test)\n",
"\n"
]
},
@@ -424,11 +1078,25 @@
},
{
"cell_type": "code",
- "execution_count": 195,
- "metadata": {},
- "outputs": [],
+ "execution_count": 43,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "0.8037468831261813"
+ ]
+ },
+ "execution_count": 43,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
"source": [
"# Your code here:\n",
+ "y_test_pred=linreg.predict(X_test)\n",
+ "\n",
+ "r2_score(y_test_pred, y_test)\n",
"\n"
]
},
@@ -445,12 +1113,13 @@
},
{
"cell_type": "code",
- "execution_count": 196,
+ "execution_count": 44,
"metadata": {},
"outputs": [],
"source": [
"# Your code here:\n",
- "\n"
+ "\n",
+ "X_train09, X_test09, y_train09, y_test09 = tts(X, y, test_size=0.1, train_size=0.9, random_state=42)\n"
]
},
{
@@ -462,14 +1131,38 @@
},
{
"cell_type": "code",
- "execution_count": 197,
+ "execution_count": 46,
"metadata": {},
"outputs": [],
"source": [
"# Your code here:\n",
+ "X_train09.horse_power.fillna(X_train09.horse_power.mean(), inplace= True)\n",
"\n"
]
},
+ {
+ "cell_type": "code",
+ "execution_count": 47,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "LinearRegression()"
+ ]
+ },
+ "execution_count": 47,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
+ "source": [
+ "auto_model09=LinReg()\n",
+ "\n",
+ "\n",
+ "linreg.fit(X_train09, y_train09)"
+ ]
+ },
{
"cell_type": "markdown",
"metadata": {},
@@ -479,12 +1172,13 @@
},
{
"cell_type": "code",
- "execution_count": 198,
+ "execution_count": 48,
"metadata": {},
"outputs": [],
"source": [
"# Your code here:\n",
- "\n"
+ "\n",
+ "y_pred09=linreg.predict(X_test09)\n"
]
},
{
@@ -496,12 +1190,33 @@
},
{
"cell_type": "code",
- "execution_count": 199,
+ "execution_count": 49,
+ "metadata": {},
+ "outputs": [
+ {
+ "data": {
+ "text/plain": [
+ "0.8363926533372651"
+ ]
+ },
+ "execution_count": 49,
+ "metadata": {},
+ "output_type": "execute_result"
+ }
+ ],
+ "source": [
+ "# Your code here:\n",
+ "\n",
+ "r2_score(y_pred09, y_test09)\n"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 50,
"metadata": {},
"outputs": [],
"source": [
- "# Your code here:\n",
- "\n"
+ "#Ha mejorado r2 con este modelo un 3% aprox"
]
},
{
@@ -531,7 +1246,7 @@
},
{
"cell_type": "code",
- "execution_count": 200,
+ "execution_count": null,
"metadata": {},
"outputs": [],
"source": [
@@ -548,7 +1263,7 @@
},
{
"cell_type": "code",
- "execution_count": 201,
+ "execution_count": null,
"metadata": {},
"outputs": [],
"source": [
@@ -567,7 +1282,7 @@
},
{
"cell_type": "code",
- "execution_count": 202,
+ "execution_count": null,
"metadata": {},
"outputs": [],
"source": [
@@ -584,7 +1299,7 @@
},
{
"cell_type": "code",
- "execution_count": 203,
+ "execution_count": null,
"metadata": {},
"outputs": [],
"source": [
@@ -616,7 +1331,7 @@
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
- "version": "3.6.6"
+ "version": "3.8.5"
}
},
"nbformat": 4,