{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-18T14:50:45.465741Z","iopub.execute_input":"2022-07-18T14:50:45.466189Z","iopub.status.idle":"2022-07-18T14:50:45.498232Z","shell.execute_reply.started":"2022-07-18T14:50:45.466098Z","shell.execute_reply":"2022-07-18T14:50:45.496953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:57:38.275295Z","iopub.execute_input":"2022-07-18T14:57:38.276516Z","iopub.status.idle":"2022-07-18T14:57:38.334021Z","shell.execute_reply.started":"2022-07-18T14:57:38.276461Z","shell.execute_reply":"2022-07-18T14:57:38.332920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing and EDA","metadata":{}},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:57:45.489852Z","iopub.execute_input":"2022-07-18T14:57:45.490245Z","iopub.status.idle":"2022-07-18T14:57:45.498840Z","shell.execute_reply.started":"2022-07-18T14:57:45.490214Z","shell.execute_reply":"2022-07-18T14:57:45.497828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:57:47.454332Z","iopub.execute_input":"2022-07-18T14:57:47.455038Z","iopub.status.idle":"2022-07-18T14:57:47.462857Z","shell.execute_reply.started":"2022-07-18T14:57:47.454996Z","shell.execute_reply":"2022-07-18T14:57:47.461603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:57:48.194588Z","iopub.execute_input":"2022-07-18T14:57:48.195267Z","iopub.status.idle":"2022-07-18T14:57:48.212261Z","shell.execute_reply.started":"2022-07-18T14:57:48.195221Z","shell.execute_reply":"2022-07-18T14:57:48.211342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Missing Values","metadata":{}},{"cell_type":"code","source":"features_with_na=[features for features in df.columns if df[features].isnull().sum()>1]\n## 2- step print the feature name and the percentage of missing values\n\nfor feature in features_with_na:\n    print(feature, np.round(df[feature].isnull().mean(), 4),  ' % missing values')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T14:58:01.926429Z","iopub.execute_input":"2022-07-18T14:58:01.927359Z","iopub.status.idle":"2022-07-18T14:58:01.976345Z","shell.execute_reply.started":"2022-07-18T14:58:01.927313Z","shell.execute_reply":"2022-07-18T14:58:01.975030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\n\nfor feature in features_with_na:\n    data = df.copy()\n    \n    # let's make a variable that indicates 1 if the observation was missing or zero otherwise\n    data[feature] = np.where(data[feature].isnull(), 1, 0)\n    \n    # let's calculate the mean SalePrice where the information is missing or present\n    data.groupby(feature)['SalePrice'].median().plot.bar()\n    plt.title(feature)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:00:21.729855Z","iopub.execute_input":"2022-07-18T15:00:21.730920Z","iopub.status.idle":"2022-07-18T15:00:25.438101Z","shell.execute_reply.started":"2022-07-18T15:00:21.730875Z","shell.execute_reply":"2022-07-18T15:00:25.435907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Id of Houses {}\".format(len(df.Id)))","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:01:08.621537Z","iopub.execute_input":"2022-07-18T15:01:08.621931Z","iopub.status.idle":"2022-07-18T15:01:08.629616Z","shell.execute_reply.started":"2022-07-18T15:01:08.621900Z","shell.execute_reply":"2022-07-18T15:01:08.628134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list of numerical variables\nnumerical_features = [feature for feature in df.columns if df[feature].dtypes != 'O'] # O means object, not object or string.\n\nprint('Number of numerical variables: ', len(numerical_features))\n\n# visualise the numerical variables\ndf[numerical_features].head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T15:01:25.626034Z","iopub.execute_input":"2022-07-18T15:01:25.626451Z","iopub.status.idle":"2022-07-18T15:01:25.660794Z","shell.execute_reply.started":"2022-07-18T15:01:25.626418Z","shell.execute_reply":"2022-07-18T15:01:25.659339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}