{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-29T12:56:54.715763Z","iopub.execute_input":"2022-07-29T12:56:54.716273Z","iopub.status.idle":"2022-07-29T12:56:54.760263Z","shell.execute_reply.started":"2022-07-29T12:56:54.716150Z","shell.execute_reply":"2022-07-29T12:56:54.759196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_master = pd.read_csv(\"/kaggle/input/housing-affordability-in-canada/housing-supply-price-rental.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:11.765579Z","iopub.execute_input":"2022-07-29T12:57:11.765988Z","iopub.status.idle":"2022-07-29T12:57:11.794556Z","shell.execute_reply.started":"2022-07-29T12:57:11.765954Z","shell.execute_reply":"2022-07-29T12:57:11.793644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_master.year.unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:13.718997Z","iopub.execute_input":"2022-07-29T12:57:13.719398Z","iopub.status.idle":"2022-07-29T12:57:13.741140Z","shell.execute_reply.started":"2022-07-29T12:57:13.719365Z","shell.execute_reply":"2022-07-29T12:57:13.739787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#converting years into integer\n\nfrom datetime import datetime\n\n\ndf_master = df_master[df_master['year'] != 1993.1].astype({'year': int})\n\n#converting year into datetime\ndf_master['year'] = df_master['year'].transform(lambda x : datetime.strptime(str(x), '%Y'))\n\ndf_master.year.unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:16.527265Z","iopub.execute_input":"2022-07-29T12:57:16.527680Z","iopub.status.idle":"2022-07-29T12:57:16.561926Z","shell.execute_reply.started":"2022-07-29T12:57:16.527647Z","shell.execute_reply":"2022-07-29T12:57:16.561117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pulling out statistics for the Toronto region\nvancouver_df = df_master[df_master.region == 'vancouver']\nvancouver_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:19.956304Z","iopub.execute_input":"2022-07-29T12:57:19.956672Z","iopub.status.idle":"2022-07-29T12:57:19.991656Z","shell.execute_reply.started":"2022-07-29T12:57:19.956643Z","shell.execute_reply":"2022-07-29T12:57:19.990876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\n#read the data in pandas format\nvancouver_df =pd.read_csv('/kaggle/input/housing-affordability-in-canada/housing-supply-and-rental/housing-supply-and-rental/vancouver_section1_.csv')\nvancouver_df.head()\n'''","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking the columns\nvancouver_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:23.578521Z","iopub.execute_input":"2022-07-29T12:57:23.579082Z","iopub.status.idle":"2022-07-29T12:57:23.586784Z","shell.execute_reply.started":"2022-07-29T12:57:23.579033Z","shell.execute_reply":"2022-07-29T12:57:23.585850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check whether the head and tail of the pandas dataframe are consistent\nvancouver_df.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:26.268687Z","iopub.execute_input":"2022-07-29T12:57:26.269102Z","iopub.status.idle":"2022-07-29T12:57:26.295129Z","shell.execute_reply.started":"2022-07-29T12:57:26.269068Z","shell.execute_reply":"2022-07-29T12:57:26.294051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(vancouver_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:29.040035Z","iopub.execute_input":"2022-07-29T12:57:29.040404Z","iopub.status.idle":"2022-07-29T12:57:29.047295Z","shell.execute_reply.started":"2022-07-29T12:57:29.040373Z","shell.execute_reply":"2022-07-29T12:57:29.046195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(vancouver_df.drop_duplicates())","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:31.517136Z","iopub.execute_input":"2022-07-29T12:57:31.517549Z","iopub.status.idle":"2022-07-29T12:57:31.533026Z","shell.execute_reply.started":"2022-07-29T12:57:31.517514Z","shell.execute_reply":"2022-07-29T12:57:31.531841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#review the data types of each column\nvancouver_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:34.308280Z","iopub.execute_input":"2022-07-29T12:57:34.308809Z","iopub.status.idle":"2022-07-29T12:57:34.321182Z","shell.execute_reply.started":"2022-07-29T12:57:34.308761Z","shell.execute_reply":"2022-07-29T12:57:34.319993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Confirm the rename columns\nvancouver_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:39.592087Z","iopub.execute_input":"2022-07-29T12:57:39.592618Z","iopub.status.idle":"2022-07-29T12:57:39.621091Z","shell.execute_reply.started":"2022-07-29T12:57:39.592578Z","shell.execute_reply":"2022-07-29T12:57:39.620149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#determining the shape of the column\nvancouver_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:42.423909Z","iopub.execute_input":"2022-07-29T12:57:42.424446Z","iopub.status.idle":"2022-07-29T12:57:42.432756Z","shell.execute_reply.started":"2022-07-29T12:57:42.424398Z","shell.execute_reply":"2022-07-29T12:57:42.431887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#want to remove the last 2 rows of the dataframe\nvancouver_df.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:45.149933Z","iopub.execute_input":"2022-07-29T12:57:45.150343Z","iopub.status.idle":"2022-07-29T12:57:45.178317Z","shell.execute_reply.started":"2022-07-29T12:57:45.150310Z","shell.execute_reply":"2022-07-29T12:57:45.177418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\n#REMOVING THE LAST TWO ROWS OF THE DATAFRAME\nvancouver_df_col_modified.drop([27,28], axis=0, inplace=True)\n'''","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vancouver_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:48.797575Z","iopub.execute_input":"2022-07-29T12:57:48.798018Z","iopub.status.idle":"2022-07-29T12:57:48.819553Z","shell.execute_reply.started":"2022-07-29T12:57:48.797982Z","shell.execute_reply":"2022-07-29T12:57:48.818391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking the null values\nvancouver_df.isnull().values.any()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:55.544110Z","iopub.execute_input":"2022-07-29T12:57:55.544534Z","iopub.status.idle":"2022-07-29T12:57:55.552860Z","shell.execute_reply.started":"2022-07-29T12:57:55.544500Z","shell.execute_reply":"2022-07-29T12:57:55.551674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking the null values\nvancouver_df.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:57:57.838498Z","iopub.execute_input":"2022-07-29T12:57:57.840161Z","iopub.status.idle":"2022-07-29T12:57:57.850798Z","shell.execute_reply.started":"2022-07-29T12:57:57.840107Z","shell.execute_reply":"2022-07-29T12:57:57.849287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vancouver_df_clean = vancouver_df.dropna(how='all', axis=1)\nvancouver_df_clean.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:58:00.888055Z","iopub.execute_input":"2022-07-29T12:58:00.888609Z","iopub.status.idle":"2022-07-29T12:58:00.917756Z","shell.execute_reply.started":"2022-07-29T12:58:00.888577Z","shell.execute_reply":"2022-07-29T12:58:00.916519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#correlation matrix as ususal\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nplt.figure(figsize=(20,20))\nsns.heatmap(vancouver_df_clean.corr(),annot=True,cmap='coolwarm')\nplt.savefig('heatmap.png')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:58:03.871774Z","iopub.execute_input":"2022-07-29T12:58:03.872169Z","iopub.status.idle":"2022-07-29T12:58:12.000177Z","shell.execute_reply.started":"2022-07-29T12:58:03.872138Z","shell.execute_reply":"2022-07-29T12:58:11.998978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from plotly.offline import download_plotlyjs, init_notebook_mode,plot, iplot","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:58:16.061295Z","iopub.execute_input":"2022-07-29T12:58:16.061674Z","iopub.status.idle":"2022-07-29T12:58:16.083509Z","shell.execute_reply.started":"2022-07-29T12:58:16.061644Z","shell.execute_reply":"2022-07-29T12:58:16.082651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cufflinks as cf\ncf.go_offline()\ncf.set_config_file(offline=False, world_readable=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:58:18.489505Z","iopub.execute_input":"2022-07-29T12:58:18.490481Z","iopub.status.idle":"2022-07-29T12:58:20.774462Z","shell.execute_reply.started":"2022-07-29T12:58:18.490442Z","shell.execute_reply":"2022-07-29T12:58:20.773460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#trying to plot year built histogram\nvancouver_df_clean['total_dwelling'].iplot(kind='hist')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:58:24.466615Z","iopub.execute_input":"2022-07-29T12:58:24.467145Z","iopub.status.idle":"2022-07-29T12:58:25.457991Z","shell.execute_reply.started":"2022-07-29T12:58:24.467095Z","shell.execute_reply":"2022-07-29T12:58:25.456753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Based on total dwelling histogram, the most common cost for total dwelling is between 15k and 20k.","metadata":{}},{"cell_type":"code","source":"vancouver_df_clean['HPI_change'].iplot(kind='hist')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:58:29.533395Z","iopub.execute_input":"2022-07-29T12:58:29.534197Z","iopub.status.idle":"2022-07-29T12:58:29.604843Z","shell.execute_reply.started":"2022-07-29T12:58:29.534149Z","shell.execute_reply":"2022-07-29T12:58:29.603864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vancouver_df_clean['one_bedroom'].iplot(kind='hist')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:58:33.477429Z","iopub.execute_input":"2022-07-29T12:58:33.477798Z","iopub.status.idle":"2022-07-29T12:58:33.539009Z","shell.execute_reply.started":"2022-07-29T12:58:33.477767Z","shell.execute_reply":"2022-07-29T12:58:33.538162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### One Bedroom will cost mostly between 600 and 700 canadian dollars.","metadata":{}},{"cell_type":"code","source":"vancouver_df_clean['two_bedroom'].iplot(kind='hist')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:58:37.314945Z","iopub.execute_input":"2022-07-29T12:58:37.315324Z","iopub.status.idle":"2022-07-29T12:58:37.374263Z","shell.execute_reply.started":"2022-07-29T12:58:37.315293Z","shell.execute_reply":"2022-07-29T12:58:37.373220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Two bedrooms will cost between 800 and 1000 canadian dollars.","metadata":{}},{"cell_type":"code","source":"vancouver_df_clean['three_bedroom'].iplot(kind='hist')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:58:41.174508Z","iopub.execute_input":"2022-07-29T12:58:41.174919Z","iopub.status.idle":"2022-07-29T12:58:41.237356Z","shell.execute_reply.started":"2022-07-29T12:58:41.174883Z","shell.execute_reply":"2022-07-29T12:58:41.236237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Three bedrooms will range mostly from 800 to 1200 canadian dollars.","metadata":{}},{"cell_type":"code","source":"#correlation of a column\ndef corr_of_a_column(df,column_name,how_many=None,ax=None):\n    how_many = len(df)if how_many == None else how_many\n    cor = df.corr()[[column_name]].sort_values(by=column_name,ascending=False)[:how_many]\n    print(cor)\n    sns.heatmap(cor,annot=True,cmap=sns.cubehelix_palette(start=.5,rot=-.5,as_cmap=True),ax=ax)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:58:45.230748Z","iopub.execute_input":"2022-07-29T12:58:45.231139Z","iopub.status.idle":"2022-07-29T12:58:45.238764Z","shell.execute_reply.started":"2022-07-29T12:58:45.231108Z","shell.execute_reply":"2022-07-29T12:58:45.237197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### HPI is highly correlated to the new total dwellings?","metadata":{}},{"cell_type":"code","source":"corr_of_a_column(vancouver_df_clean, 'HPI_change')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:58:48.295333Z","iopub.execute_input":"2022-07-29T12:58:48.296129Z","iopub.status.idle":"2022-07-29T12:58:48.606074Z","shell.execute_reply.started":"2022-07-29T12:58:48.296079Z","shell.execute_reply":"2022-07-29T12:58:48.604864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### HPL is correlated to total dwelling but owned_accomodation_costs_chage is highly correlated to HPL.","metadata":{}},{"cell_type":"code","source":"vancouver_df_clean['owned_accommodation_costs_change'].iplot(kind='hist')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T12:58:52.153246Z","iopub.execute_input":"2022-07-29T12:58:52.153883Z","iopub.status.idle":"2022-07-29T12:58:52.214148Z","shell.execute_reply.started":"2022-07-29T12:58:52.153834Z","shell.execute_reply":"2022-07-29T12:58:52.212952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Most owned accomodation cost will change from -1 to +1 range.","metadata":{}},{"cell_type":"code","source":"vancouver_df_clean.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-29T13:05:02.314711Z","iopub.execute_input":"2022-07-29T13:05:02.315123Z","iopub.status.idle":"2022-07-29T13:05:02.322966Z","shell.execute_reply.started":"2022-07-29T13:05:02.315088Z","shell.execute_reply":"2022-07-29T13:05:02.321857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vancouver_df_clean.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T13:06:34.728617Z","iopub.execute_input":"2022-07-29T13:06:34.729040Z","iopub.status.idle":"2022-07-29T13:06:34.745447Z","shell.execute_reply.started":"2022-07-29T13:06:34.729003Z","shell.execute_reply":"2022-07-29T13:06:34.744482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_continuous_df(df):\n    return df.select_dtypes(include=[np.int64, np.float64])\n\n# Get df with columns having categorical variables\ndef get_categorical_df(df):\n    return df.select_dtypes(include=['object'])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T13:08:55.395459Z","iopub.execute_input":"2022-07-29T13:08:55.395874Z","iopub.status.idle":"2022-07-29T13:08:55.401674Z","shell.execute_reply.started":"2022-07-29T13:08:55.395840Z","shell.execute_reply":"2022-07-29T13:08:55.400792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"continuous_df = get_continuous_df(vancouver_df_clean)\ncontinuous_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T13:09:47.411391Z","iopub.execute_input":"2022-07-29T13:09:47.411853Z","iopub.status.idle":"2022-07-29T13:09:47.444276Z","shell.execute_reply.started":"2022-07-29T13:09:47.411785Z","shell.execute_reply":"2022-07-29T13:09:47.442895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_df = get_categorical_df(vancouver_df_clean)\ncategorical_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T13:11:15.074073Z","iopub.execute_input":"2022-07-29T13:11:15.074606Z","iopub.status.idle":"2022-07-29T13:11:15.090715Z","shell.execute_reply.started":"2022-07-29T13:11:15.074559Z","shell.execute_reply":"2022-07-29T13:11:15.089510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_base_relation(df, figsize=(20, 200)):\n    columns = df.columns.tolist()\n    _, axs = plt.subplots(len(columns), 4, figsize=figsize)\n    \n    for idx, column in enumerate(columns):\n        # To get distribution of data\n        sns.histplot(\n            x=df[column],\n            kde=False,\n            color='#65b87b', alpha=.7,\n            ax=axs[idx][0]\n        )\n\n        # To get knowledge about outliers\n        sns.boxplot(\n            x=df[column],\n            color='#6fb9bd',\n            ax=axs[idx][1]\n        )\n\n        # To get its realtion with SalePrice\n        sns.scatterplot(\n            x=column, y='HPI_change', data=df,\n            color='#706dbd', alpha=.7, s=80,\n            ax=axs[idx][2]\n        )\n        \n        # To get count plot for `column`\n        sns.countplot(\n            x=column, data=df,\n            color='#42b0f5', alpha=.7,\n            ax=axs[idx][3]\n        )\n        \n        \nplot_base_relation(continuous_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T13:11:48.306551Z","iopub.execute_input":"2022-07-29T13:11:48.306998Z","iopub.status.idle":"2022-07-29T13:12:12.352743Z","shell.execute_reply.started":"2022-07-29T13:11:48.306961Z","shell.execute_reply":"2022-07-29T13:12:12.351534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### From the above visualization, it is clear that HPI change is in regression format.","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.scatter(vancouver_df_clean['year'], vancouver_df_clean['HPI_change'])\nplt.title(\"Scatter Plot of HPI_change with year\")\nplt.xlabel(\"Year\")\nplt.ylabel(\"HPI_change\")\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2022-07-29T13:58:30.588483Z","iopub.execute_input":"2022-07-29T13:58:30.588874Z","iopub.status.idle":"2022-07-29T13:58:30.839246Z","shell.execute_reply.started":"2022-07-29T13:58:30.588843Z","shell.execute_reply":"2022-07-29T13:58:30.837910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### It is clearly seen that HPI change is lower in 2009. In addition, 2009 is the year of great recession so we can conclude that HPI change in Vancouver is low due to recession on 2009.","metadata":{}},{"cell_type":"markdown","source":"#### ","metadata":{}},{"cell_type":"code","source":"from scipy.stats import zscore, pearsonr","metadata":{"execution":{"iopub.status.busy":"2022-07-29T13:22:59.701862Z","iopub.execute_input":"2022-07-29T13:22:59.702225Z","iopub.status.idle":"2022-07-29T13:22:59.707730Z","shell.execute_reply.started":"2022-07-29T13:22:59.702196Z","shell.execute_reply":"2022-07-29T13:22:59.706115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_corr(data_1, data_2):\n    return round(pearsonr(data_1, data_2)[0], 2)\n    \n    \ndef get_corr_info_for_df(df):\n    columns = df.columns.tolist()\n    \n    # To keep record of which corr info is displayed & hence\n    # not to display opposite corr info \n    # eg. if a - b is display then b - a will not be displayed\n    # since they will have same result\n    displayed = []\n    def is_displayed(main_column, secondary_column, displayed_list):\n        return any([\n            True for tup in displayed_list if \n            (tup[0] == main_column and tup[1] == secondary_column) \n            or \n            (tup[1] == main_column and tup[0] == secondary_column)\n        ])\n\n    \n    # main column is the one for which we'll find all the corr info\n    # related to secondary column\n    for main_column in columns:\n        for secondary_column in columns:\n            if secondary_column != main_column and not is_displayed(main_column, secondary_column, displayed):\n                corr = get_corr(df[main_column], df[secondary_column])\n                if corr >= .7 or corr <= -.7:\n                    # print only if pearson correlation is high (.7 to .9) or very high (.9 to 1)\n                    print(f'{secondary_column} - {main_column}: {corr}')\n                    displayed.append((main_column, secondary_column))\n                    \n                    \nget_corr_info_for_df(continuous_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T13:23:03.363358Z","iopub.execute_input":"2022-07-29T13:23:03.363727Z","iopub.status.idle":"2022-07-29T13:23:03.509181Z","shell.execute_reply.started":"2022-07-29T13:23:03.363696Z","shell.execute_reply":"2022-07-29T13:23:03.508004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Total dwelling is highly correlated to multiple, row, apartment and condo. Therefore, it means that most of the houses in Vancouver are in multiple, row , apartment and condo.","metadata":{}},{"cell_type":"markdown","source":"#### From the correlation, we can conclude that most of the people live in two bedroom and three bedroom as population is highly correlated to two-bedroom and three-bedroom.\n","metadata":{}},{"cell_type":"code","source":"vancouver_df_col_modified.sample(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}