In [ ]:
# This Python 3 environment comes with many helpful analytics libraries installed
# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python
# For example, here's several helpful packages to load

import numpy as np # linear algebra
import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)

# Input data files are available in the read-only "../input/" directory
# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory

import warnings
warnings.filterwarnings("ignore")
pd.set_option('display.max_columns', None)
pd.set_option('display.max_rows', None)
import matplotlib.pyplot as plt
import seaborn as sns
sns.set()

import os
for dirname, _, filenames in os.walk('/kaggle/input'):
    for filename in filenames:
        print(os.path.join(dirname, filename))

# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using "Save & Run All" 
# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session
In [ ]:
train = pd.read_csv('/kaggle/input/home-data-for-ml-course/train.csv', index_col = 'Id')
test = pd.read_csv('/kaggle/input/home-data-for-ml-course/test.csv', index_col = 'Id')
In [ ]:
train.shape
In [ ]:
train.info()
In [ ]:
num_data = train.select_dtypes(exclude = ['object']).drop('SalePrice', axis = 1).copy()
num_features = num_data.columns
cat_data = train.select_dtypes(include = ['object']).copy()
cat_features = cat_data.columns
In [ ]:
#checking for features with missing values 
missing_values = train.isnull().sum()
missing_values = missing_values[missing_values > 0]
missing_values.sort_values(inplace = True)
(missing_values/train.shape[0])*100
missing_values.plot.bar()
In [ ]:
target = train['SalePrice']
num_features
In [ ]:
# distribution plot 
fig = plt.figure(figsize = (12,18))
for i in range (0, len(num_features)):
    fig.add_subplot(9,4,i+1)
    sns.distplot(num_data.iloc[:,i].dropna(), kde = False)
    plt.xlabel(num_features[i])
plt.tight_layout()
plt.show()
In [ ]:
#boxplots
fig = plt.figure(figsize = (12,18))
for i in range (0, len(num_features)):
    fig.add_subplot(9,4,i+1)
    sns.boxplot(y = num_data.iloc[:,i].dropna())

plt.tight_layout()
plt.show()
In [ ]:
#scatterplots btw target variable vs num_feature
fig = plt.figure(figsize = (12,18))
for i in range (0,len(num_features)):
    fig.add_subplot(9,4,i+1)
    sns.scatterplot(x= num_data.iloc[:,i], y= target)

plt.tight_layout()
plt.show()
In [ ]:
#Features with Outliers:
#LotFrontage
#LotArea
#MasVnrArea
#BsmtFinSF1
#TotalBsmtSF
#1stFlrSF
#GrLivArea
#EnclosedPorch
#LowQualFinSF
In [ ]:
#Regression Plot for Outliers
outliers = train[['LotFrontage','LotArea','MasVnrArea','BsmtFinSF1','TotalBsmtSF','1stFlrSF','GrLivArea','EnclosedPorch','LowQualFinSF']].copy()
fig = plt.figure(figsize = (12,9))
for i in range(0,len(outliers.columns)):
    fig.add_subplot(3,3,i+1)
    sns.regplot(x = outliers.iloc[:,i],y= target,scatter_kws={'s':10})

    
plt.tight_layout()
plt.show()
In [ ]:
train.shape
In [ ]:
#dropping the above outliers
train = train.drop(train[train['LotFrontage'] > 200].index)
train = train.drop(train[train['LotArea'] > 100000].index)
train = train.drop(train[train['MasVnrArea'] >1500].index)
train = train.drop(train[train['BsmtFinSF1'] > 4000].index)
train = train.drop(train[train['TotalBsmtSF'] > 4000].index)
train = train.drop(train[train['1stFlrSF'] > 4000].index)
train = train.drop(train[(train['GrLivArea'] > 4000) & (train['SalePrice'] < 400000)].index)
train = train.drop(train[train['EnclosedPorch'] > 500].index)
train = train.drop(train[(train['LowQualFinSF'] > 500) & (train['SalePrice'] > 400000)].index)
In [ ]:
train.shape
In [ ]:
#correlation plots to select features to reduce redundantcy 
num_corr = train.select_dtypes(exclude = 'object').corr()
plt.figure(figsize = (12,12))
plt.title('Correlation between Numerical Features')
sns.heatmap(num_corr>0.8, annot=True, square=True)
plt.show()
In [ ]:
#highly correlated features 
#GarageYrBlt & YearBuilt
#1stFloorSF & TotalBsmtSF
#TotRmsAbvGrd & GrLivArea
#GarageArea & GarageCars
In [ ]:
#To select which of the correlated features to incorporate into the model
num_corr.SalePrice.sort_values(ascending = False)
In [ ]:
#Dropping columns GarageYrBlt, 1stFlrSF, TotRmsAbvGrd, GarageArea from the train and test set 
train = train.drop(['GarageYrBlt', '1stFlrSF', 'TotRmsAbvGrd', 'GarageArea'], axis=1)
In [ ]:
test = test.drop(['GarageYrBlt', '1stFlrSF', 'TotRmsAbvGrd', 'GarageArea'], axis=1)
In [ ]:
test.shape
In [ ]:
train.shape
In [ ]:
#imputing missing values 
#feature selection 
#feature engineering 
In [ ]:
train.isnull().sum().sort_values(ascending=False).head(18)/train.shape[0]
In [ ]:
#dropping columns with over 90% missing values 
# Pool area has 99.5% values as 0
train = train.drop(['PoolQC', 'MiscFeature', 'Alley','PoolArea'], axis=1)
test = test.drop(['PoolQC', 'MiscFeature', 'Alley','PoolArea'], axis=1)
In [ ]:
train.isnull().sum().sort_values(ascending = False).head(15)
In [ ]:
test.isnull().sum().sort_values(ascending = False).head(28)
In [ ]:
#LotFrontage highly correlated with LotArea
lot = train[['LotFrontage','LotArea']]
lot['LotSide']=np.sqrt(lot['LotArea'])


sns.regplot(y=lot.LotFrontage, x= lot.LotSide,scatter_kws={'s':10} )
plt.show()
In [ ]:
#filling missing values of LotFrontage 
train['LotFrontage'] = train.LotFrontage.fillna(np.sqrt(train.LotArea))
test['LotFrontage'] = test.LotFrontage.fillna(np.sqrt(test.LotArea))
In [ ]:
#filling missing values for MasVnrArea with mean value
train['MasVnrArea'] = train.MasVnrArea.fillna(0.5*(test.MasVnrArea.mean()+ train.MasVnrArea.mean()))
test['MasVnrArea'] = test.MasVnrArea.fillna(0.5*(test.MasVnrArea.mean()+ train.MasVnrArea.mean()))
In [ ]:
train.select_dtypes(exclude = ['object']).isnull().sum().sort_values(ascending = False).head()
#all numerical missing values are handled in training set
In [ ]:
test.select_dtypes(exclude = ['object']).isnull().sum().sort_values(ascending = False).head(7)
In [ ]:
#categorical features
In [ ]:
## Count of categories within Neighborhood attribute
fig = plt.figure(figsize=(12.5,4))
sns.countplot(x='Neighborhood', data=train)
plt.xticks(rotation=90)
plt.ylabel('Frequency')
plt.show()
In [ ]:
#missing categorical features
cat_data.isnull().sum().sort_values(ascending = False).head(16)
#the houses don't have a pool, alley, fence, filling these "NA" values with "None"
In [ ]:
cat_missing_values = ['Fence','FireplaceQu','GarageCond','GarageFinish','GarageQual',
                     'GarageType','BsmtFinType2','BsmtExposure','BsmtFinType1','BsmtQual','BsmtCond','MasVnrType']

for column in cat_missing_values:
    train[column] = train[column].fillna("None")
In [ ]:
for column in cat_missing_values:
    test[column] = test[column].fillna("None")
In [ ]:
train.isnull().sum().sort_values(ascending = False).head()
In [ ]:
train.columns
In [ ]:
train.shape
In [ ]:
y = train.SalePrice
y = np.log(y)
#y

X = train.drop(['SalePrice'],axis=1)
print(y.shape)
print(X.shape)
In [ ]:
from sklearn.preprocessing import StandardScaler
from sklearn.impute import SimpleImputer
from sklearn.model_selection import train_test_split


# One-hot-encoding to transform all categorical data
X = pd.get_dummies(X)

# Split into validation and training data
train_X, val_X, train_y, val_y = train_test_split(X, y, random_state=1)

my_imputer = SimpleImputer()
train_X = my_imputer.fit_transform(train_X)
val_X = my_imputer.transform(val_X)

print(train_X.shape)
print(val_X.shape)
print(train_y.shape)
print(val_y.shape)
In [ ]:
from sklearn.metrics import mean_absolute_error
from sklearn.linear_model import LinearRegression
from sklearn.ensemble import RandomForestRegressor
In [ ]:
#target is now log(SalePrice). After prediction call, need to inverse-transform to obtain SalePrice
def inv_y(transformed_y):
    return np.exp(transformed_y)
In [ ]:
#Linear Regression model
linear_model = LinearRegression()
linear_model.fit(train_X, train_y)
linear_val_predictions = linear_model.predict(val_X)
linear_val_mae = mean_absolute_error(inv_y(linear_val_predictions), inv_y(val_y))


print("Validation MAE for Linear Regression Model: {:,.0f}".format(linear_val_mae))
In [ ]:
#random forest classifier
rf_model = RandomForestRegressor(random_state=5)
rf_model.fit(train_X, train_y)
rf_val_predictions = rf_model.predict(val_X)
rf_val_mae = mean_absolute_error(inv_y(rf_val_predictions), inv_y(val_y))


print("Validation MAE for Random Forest Model: {:,.0f}".format(rf_val_mae))
In [ ]:
#Linear Regression gives lower mean absolute error
In [ ]:
sns.distplot(train.SalePrice)
In [ ]:
#processing of test data
test.shape
In [ ]:
test.isnull().sum().sort_values(ascending = False).head(14)
In [ ]:
test.select_dtypes(exclude = ['object']).isnull().sum().sort_values(ascending = False).head(7)
In [ ]:
test['BsmtFullBath'] = test['BsmtFullBath'].fillna(0)
test['BsmtHalfBath'] = test['BsmtHalfBath'].fillna(0)
test['GarageCars'] = test['GarageCars'].fillna(0)
test['BsmtFinSF1'] = test['BsmtFinSF1'].fillna(0)
test['BsmtFinSF2'] = test['BsmtFinSF2'].fillna(0)
test['BsmtUnfSF'] = test['BsmtUnfSF'].fillna(0)
test['TotalBsmtSF'] = test['TotalBsmtSF'].fillna(0)
In [ ]:
test.select_dtypes(exclude = ['object']).isnull().sum().sort_values(ascending = False).head()
In [ ]:
test.isnull().sum().sort_values(ascending = False).head(7)
In [ ]:
print(X.shape)
print(test.shape)
In [ ]:
# One-hot encoding for categorical data 
test_X = pd.get_dummies(test)

# Ensure test data is encoded in the same manner as training data with align command
final_train, final_test = X.align(test_X, join='left', axis=1)

# Imputer for all other missing values in test data. 
final_test_imputed = my_imputer.transform(final_test)
In [ ]:
print(final_train.shape)
print(final_test.shape)
In [ ]:
# make predictions for submission
test_preds = rf_model.predict(final_test_imputed)

output = pd.DataFrame({'Id': test.index,'SalePrice': inv_y(test_preds)})
output.to_csv('submission.csv', index=False)
In [ ]: