{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![images.png](attachment:19b134c4-6b41-468d-9548-fae5669aee00.png)","metadata":{},"attachments":{"19b134c4-6b41-468d-9548-fae5669aee00.png":{"image/png":"iVBORw0KGgoAAAANSUhEUgAAASsAAACoCAMAAACPKThEAAABrVBMVEX///8AAJn9YwAAAJQAAJDp7O9ycrrOzuOoqNJsbLjNzeKMjMXIyOK2ttmWlsXw8Pe/v934+Pj+t5/9WwD9jmC8vNuZmcuamsff3+6oqKj09PTa2tr29vro6PLIyOCwsNPT09Ph4eGzs7OTk8iqqtOfn5+Hh8LW1taTk5Ourq7AwMB+fr/a2uuQkI+BgYDExMRVVa5kZLRLS6t4eLxdXbITE5x6enr9UAD+qYrO0eDU2uBpaWhDQ6hUVFNnZ2cbG50tLaE6OqU8PDs3NzZJSUj+w639g08pKqD+2Mv9o4BAQKj949v6yLn9l2/kbk/k2eL+iln0XxHUtL//7ube6Prgnpf+vqj9cCjcpKTHTkirQ1z+pYSfP2PTiIYSEhAmJiTnZz26ZnaFaYxZPoisl6idg5b/9N4pCYFzW4zx38ujmLWiiJcAAIFmZaNOVp7/2Lr6t4v8kEpEGnf6rl/7fjKuZ2D/tTz9v3/7nT1yUXmmSi3efFfixr/7wG3+qlOeUkByMiz+wFf/nWTAY0RvQUVlEAD8y3v/cwDJeWCdYljxhiOQJwDkYxvIkYC0h396/ALaAAAXOklEQVR4nO1dC4Pi1nW+a0ZIgDSSPCuhN3ryEBKCgTCZx75iZ712dlO729hJmyZp67qp67Z2k9RpHm3jPvJo09/cc/WCAQbEzLAwDt+yDFdXXN3z6Zxzzz2SLgjtsccee+yxxx577LHHHnvsscceXxqwxXbTN9uLAhDzt6Vg64VaE4ICOzVnmqYKNY2cYrttELX8bTkKnHxKQxq/sGJpU6w7+aw18o/CrB6tPqGbRsoV26w34+4IOt8UAySITawiHCM6DKq6NZBXb4oiosUaiRgx1rKG45BIrMOOFFMfavpIJGkdOaLjIp2LG6vhRpEEFfCxi4UnqSaqxw2wNTGRnnUcETrAIKhwKpg4AfZlKw4NH8V6F+9Diw5uzxUdDvZCA9giabgBwcWNUGJ9yCIXOoF7SeFeVjfBFQVwUBnOaRcT12AZCXwDdHkItQyohMPSIrYA2AVpdbw97j4yBHhzYecKcgJsWbCvIKELaFOTBHwCnKQhLF1NiyVEpIMrcAPAIhNbLNuHzQzWLmAs06smYlL1w03FXxBRFT42tSZiuzTsgBuQGAbvcwz/cfeyXooo7eMtc5W8OXH/oMDoqNyAE++6+KhQQhJHk9BTHfeKc1y3goQui1K3g98EKfl2XcNcOXhDwlXZdV1sf1CB93PAQEkDV+AG+q4r0jFXmM6mFnMVU4t0EbSvEdsgXXfwccmyBFzhvgZcA3wn7jBugGT0TIgmCkTK0TGz1VrS+w1xhWUZ4nfcRZLKnAXWsWqVhvMvxt6dS333IPsmPn0U68TFKa4wsc3ciYtarKS4SFaTLw4mHj727TWNirmK9WqIP8ZcBW7mwoa8iBxQTprUGg5yyCBpIGEUf+zy0G4tJtsgN0AUyrWjKSGXQ2DvxzqLNAfsKnbBwhCbGOYKtFuHTfF2NqZIhzotALspx42ADeqYq1rMcAVJTxADJ1nDZOoIbCXAHGG9asYN1CSoI+Eg7BMNZAT29LfwkRE2JxKUFwtsMEjE0rPAt4OPxMNZqtBIegclDbgxV30Szi+cMaGGKriXg8KByFog07eGiD+JlM5K4BzB5OMyE9RBW4A+FGAPKiTbKVHC39JFUQfFd5NGsJ0GsCf8lVgk1TDBtFjHXGF/RotC/B2stkkDDTgOVjdNd3FRd7DSMpiZQCSxJmP3zIg6VmXcuyA2Tijg7TRKGpCS9kncOi1KcS85xIuJeb9SMBsOAAuGVqvQXL3L5kFuQpVvH8K2O7DHHnvssccee+yxxx57bA98tO0e3BhK/G7lZaJ1Sw2/6YXw3s7L58SCndSFifIdRU+OT/d5vuFcWbCXeZ2W0dhqmzZqh7zctpD1tZbdtlEom56NvJDwQj5qEw/C63V7K3gAmuS10Rh5bSUCkfwHRDu0QMgOFHqhAnV223rDXr/lN9sds4PGvolaoQrnoseHwN8D9EB9YPr4qHZPRd6iM7OrsB5GyHzI90zP7EHfkeLZLXSuylZPPQ/BQFoPlYcKULk+4DtmhHq2iqIOAZSMCQ/4GqOHiuJHyGoTMpL9m3Kl8Pj1ynAuW7LS832QR7aBq44FXEUtmQC5kKdCcUxchyvwVC0T3schH7ZNXG63ZYTVFLQtVMeeb4eWeZ2WJ1DMyFTVtNCxsTuJEtcberPu0cSOOLRmtq7DdK9nmmFPaaOeZ0HfkdIBy2upEQL5lHEb6givh7xr2OArQaRaJnClYsLavGfaHc+WFdnjbd8C3yHzdpuX4WSpni/bHlLHltwh2lbHVmWi0yY86/zuj84FQahqFIEZt0CD0RicLnAD6ou8SG63wEI8KwSGPL5jKygEnTatEIU2WJEZIpkHn4C8bYvwysCbUUs1E71qgT32Qqtnm6BDHlibp5o9qw0aNY4sX7aAQfgX+j6mkh9Hsj8mZPVmPuAugVcJy5wfHhb4IP7SBx6/dii0I+Zjz3QLf+nPLKIdkuEyrpZorrDeJEVpW3g8txDBI0ux8czA8hBP8LwCEQP8RSavKHB8XoFtsJvCw7EUhdhdD9JTE4mgowRhx4IlEsFpx3LKBPQfS6GicK1YSBmbflsZ+74c+iYRRn47dhlmKLdk2zQ7PWRbEPXKsF+np3hqD4Z5S5HlaHe5Oo/MkDgHSdqRTfRanbADQ4UdwWxEhujUI0I78kLbR54lr8mVj6xzBJGbDeRD5GYSMEJh9wohqIci30K2GnUgGpIjn/Ai0+qEFq+aZrS7g7iHiIdYAog5CCSb0G0YXG1etnksVAShKAy9HR6LKa/XMsQwPIxKCIJ0mYgUu0VAWCv7MGKpoK04dFcIIAzCI9ts+YRNtCCIw5s3I+gtoANTGmtsI5jkdlST8E2+Axt4FXizkEooPhYWuk90FHV1a3OYFfw6bewW7r4EX2bETskC85vZruIR81oa/AowG8zw1iRKMOF/C6zItniZV21+PRE6HRP5LdUGV2TZKlLkCNxUJzLB3GWYxNmK31ZhFspHMi93bNOysT+L90adG4u1CSg4/RKPTLLigzDmOZxZKMsK+HPbVEAEBP4eD1+WudYE2lRhQubZkSLDCOspPjQBc1cTxpEOfOyEUce2WjBV8+FoHRh1cVQBXPG+bUc7OlPvmJYK8YCptnk/hMG9o/oR4dtRCz7b6A3Eg88H6uAFUqzTMA9T2HMVc4VjAuCu45myFbY82+7ZkWz7SujFXEHchbkCjQLaOshU5PVOyqtDLyLamKuWL7dBEAga1bZHAFdyCwwiinhbVb1O1DFlmK3f0jGXRJsWjuH43Q0broQ59Q5Q7qAI82Cpwba7cDeguccHpQNm293YfWiN/kHpHqC07Z7sOphRQhTmaifuid1VCIOcKIyD7T/ntqPgKpeIwjjedp92EnS5VJplChSL23a/VgM/dSUiVKYQ6o42f3c4PVxEVIyNH/vGAKcaNBFXE/ETaY83eyyyeyVR4N63/wTlcui14xqSmugRDgdrtU06jaVExVa4809mgF5JXSQxXboqVTb2mE/QvDdPFGyZ3lba+egdxmoe/muszm7qkRuhtoiog8digMjHUxUHG3rYb+fw9aOnV9Q0mwfz9nZMpQGVMEXjo1fW262CP3nt6KsLa1wX9Wepemtag6mJrt3Sg3G7jpOjk9cXbSe7YOdzijU96BmVR/lm7VV195ogWYnTdS61CRbpjzk0eoKfAS1JxVt59vTtRZv1Pn6vzbure/HDyxozKNOT6tImnuO+TWiHjSpFkek6EnDC8dP7Bl5SoLYGVxjCfCx7kfyZc1jAyyPaHXRJxA2aEyoPiqwWskWQTLXeIOMlARBZ69djriDS4kaDdf1HeXYkO06Nipt375gYXRzBxKE7pXZPbi7PJmEYdYak3FRM0KsGg0o10pG0rrFuW6PLmljJi8eLuEIXsY+f0qt7pd3O+mmHAlasdEkdUARN43WdxUHX+vcSHU+PcM5kPjzv3oErLclaTevVjmf9qpxRF8h6Y/WeRXAx+chNryfUXJBaiLkihw1nmqudzvqR9CFXBTO8nda0nCypcqlioV4F3YqgNx5f2rrLWT/NMQwww8OsLDWqukDrq30VW64M0zROleqWh10K851ECdDszERcmFOsA9RndOZ4Nu23y1k/2gDFAjPMyo7BHdKGszIsZI9xNObWYPgrM3jVDU1iypUqkhIvdzG7/8UsVyV2nqh7u531C4RDUCwu1yODqQNb3Eqb7IMKcfFiLtOg0jitP2dK0nz0XiDrV2A+L5UZVGtqSNh8Bow9rGJFym2wUa8Ce9wqG2RrqI/XvNIWOuPuAqq7S9NXUxROjQn0cLUAOnrCay7DZGvC0RRiqQAJtzRaTYOrxoqV567qJFOvcittMGBqT/AKN6iyoNLNV5OjK5Vc9IJcTWX96EKTHv4C0Y+0PvU4FYgbStJbiKsXW1lwGTSkD/oi0gZp4KinipTVV10K2DJW6ZXWRW8dkyzLLljij8zXMqtBf6WLVHamqGJlylREq6An70D7ZHOABulxODBGmACIN3d88WJgOli5k0y+9DoJbHG5DdZdXGZWzswGuuZUhngxrFmwpcwCg4TIbGh7UoSoUqmZNskVogoxta4+7GpaOTW6ADR9pCOmwDqNy6HVjmssO8ALfHUTs+OAGmx22R4uCTPpRn3lOMhfYL3sznsF/pjLDHqYjIlceobn3fs8Ud0gs7uCVM3huCy6g0pwXE4dAcsiFjpAZ2fgK0gf9VGzUuCqVROrBKox3XoyNWapRJGy+mQmXS2gwE6lUlmQihC1nKtM7Ezq4TIrBKLK+IupA7wuVbNgLyj0pNoVuWxefoy6OkP3i1goGLWEV0cLUkORXDL2UFl9MpN218zHXELOVeb3cxd9NVelgwp36Uvx2pHlSgVezvKLJtS9r2BcNZWsUmjQp8njUVo+Rnr52NAa11ggkqvGipUP8mmK5iYXKnKunETvJ3GFu5is0sHUddyMYLqfelTEMpVYyZjFYQCVNHqQlXVS02HUyc42TWl91C9rmQ28g3jhGDHONYZJDajBHior45k01SBvMojkXGnJdSx3MlA8XkRUn5nOaGRczZ2tLmcstMtZrgTjsFGtZ1c/kNFAo3JDqGTX1CSk0XgZyWsIdsqQea4PA8+kG2T9JlPYnCvEzCp6cDBL1LE7M4wsCthiDJIlOucwy1WcDGhUbyMVdlDCOMgU6XKuDwQ14oTW7XCFyFHFvVRXnk7AHDxZYOtXclUX6wuXjZ7lCs9CQISrJJBYI9CqZKHkXHJm8wxkmuvL9SqeSTNFxsErwS3xxVp+IeLgrcXiXMkVChYHfXN6lYiQ7axXJYoMXCNTX6FuMIfGgosDCzDDFc71MVO5Pic+K4e34tsXIZasdPDIuWqkvZqrZS1OccXhOZtgZGRIwAwmJ6uOY2+hmICzXBmJh8qqDRrrMHMrMcNCPC6V7tWWTAtuzNVhnFXKyeAPqxyQk5tv4NZP8UylSNMzXPEOaTgGmTfVOAw4p7pyPrgMy7nqOsuHoBtzFSQiZJ0wgBz8yrijGkHdDdxC4+AMV3GegT7MuUmmh1megdUQVYbwqDZYY3q1nKsrcgf886ffiD/cmCuXiq0sk4gFZqYdsoAz5nW30OA1w9VcnqFBkUKeZ+AkxFPxYeYSnVfjGlw9v39ydHJLXFHxRCS3QR14gpGeynw7eYgvxZCF0luzXOGZszHJMyTTw/T3IiBg5lAVzw4GawQRGVf3XzxbUDvP1TMgCvDNP4pLN+aKTEXIijjYqjfycbCeREWFHPKsDSbDQm6DyfTwMG2ZllC9PiAbo/4aU4KcKyDg7Tm6ZriCOOcF6NT958ni43NcfX0h4dOY5SoOrcnGaVrUDsHkBDLvP50MZtfhiqWo6bOAkulh5hlBEphY8RrLrhFETLh67bV5uqa5evb2yXPg6t3nuLCAq2cvTk5O3vvj5Yeb5aqB52iNetZhCZQIkzOxwVixrmODkpsksLLqdHp4GzED5uq1hK6Xk9qcq2dvQw3mKmVnlquXr793cnL01Zd/skKl52wQi+CSmYAkzHBxvJXboGE4XPXwOr6dIy/nGRrx9LB+G7FoyhVg+uaslKvn2EWdPP0Wl2+5zNXzd0Gl7mOVlFbcgzIXXyUTkfxsO1ysSlnRiGNVbkmEN8GsDabmnTeVTA8n46Kmkzq71i+nzHN1NM/V6ycnT9/nUcAs5Or1o5OTD95P5mzrciUkImSKE3BGfLk40yvjMA63r2ODQSNRrKw6PSt5RiNJcKwVmhbi6tuvx3ZJLubq3aM02kLrc5VM/nMbDCCKhxed22CVw2xdZ45jGImHyqqFOIGVnxXNSe4NKdJyhkJcNZO+X8HV/anbT9e2waTHuUPCJmfQ+cQkvqZ3aBSylPk8Azkd1iYjaj5dMtIhtkjLGQpxVdsYVxy+jD4JCgJwUEAOl+kVnhxCQHmtPINQPQR/nmeZHLoK7OUqGsTzUOH2bXBzXNHJ3SzZ6ZWEwyp+ZQkrmMQdkoJzHRucvacomaTnv/OmJZfw1/ptti1zNZ9nEOqkkI+DgXs5oFyGWRtME1ZZdTyT5vI8A40VFjYUaTmBzjTTL6/JFXdLXCXJAC5zSGSVOQRy8viqEc9MGoUsZYar2XuKsB4Zk3uKpPiitFGcq+FBKXtaZE2u6ldwFawZiya+m866rOOUzFRQlMTaxQLImXx7wOAhdeKQ6MahIUzuKdLrVdjAFbZBfENj6TJX3/nT91ZzFTRHqfqkXP1ZujtZGa4Qi0olysp1A4uQOyQWJKCq7sQG43F/xRyHSGAkOEuLZ2fEqUScnqZF4lTXT2FTVpQkQ9KNUz0prU7oxxniGb36zndXcKXXRpPzfImrev/ypY1L0NIuphJlXdYvi6CfnRL4lVUbxKmhnxpKUlop0MYg0Yu4WmaDOG6nxGnNmXAVlC92/OGA60Msxfc0ltbgiu3OjUcJVydvu31q0Q1LrxLZSp78pDy9kBQ/s2WmehlGpeQBSrqob2fro+Z8PiOZO7//5+RVNw0ukGimi/OFFVuuaFju2aHZGfuhH5fNHl7OO4zsdP0dGWqizrnVttNqdTy22r7/cGXLeDjqO93jg9ogHa6XcyWOmgstbCq/VYirVhj1xpYnh+NE/M65GdrQ3zeTlR/Vth+OVa/jP8yq1VD2zwm5fWWLOXivh9dF7KFeUpajKES2la9VFNeEyE9/jcBrAYu+aq1eeheP208qXZesNoro1RUXRnXnopz9QmpBvbJVzybMKIzSpdNCvCx9D3VSv+2DJlh4MfZ099BLqlcv8qR8z+60o+93/HaiV/IDddzz31T5hGb+e7jmDS/6mplWW+Ne9DXVXqlX6WMkpYODR5VRcul9KVeLAEaJb62pNgfJBcRCXNlvWr1e6/tmLz2db7RBggchStdAh873euYbrYc9Ja02w855T7WL/QDCCntdVLPaujFXEOS8E1/9w9GStiZXGjXlvVhqMDQK6tXavY2rt/r7BxARloXJPS9Bd/CD4lzxjUE3mN3k8Dv9dM51obmD5tyV3G989bWjoyJcCYPhwnSSVgu6N0n87yKYwfCKS97fwNcflnPFVa5cGAL/3LXYp1be53tnQJaXL4IRX665iiujPFiSHEn8VVCufCnCd9bpF0iZPnvxF5NExYQrsjtYfvOdlt0WSBU5ym7DGBR2J0KlnGpfypXUHM3e/TiPyQ7GqHaXTXHdk516NcyVNJ1YKAi9PNzl5zCXgG0O1r+TN44Mhqw4cq4ltdat3EG29Er3mr3W3P666z5MgW/eObbY5tZ8h9Yd3mW/9YrBDta6qPQHDvKu2eEfBqy/3HYP7g5e/tW2e3B38GHpr7fdhbuCj0r3/uajbXfijuDjHx787Yfb7sTdgPLhxwcv91wVwt+h6BN1R39rY/fw0Sfqy9V77YGhfPLxtrtwd/D3/7DtHtwN/OOnn/36n378w/e33Y/dx0c/+tFnX/z4ux/85OdPn2+7LzuOl//8+U+/+PTDf/nWBz/7+Xu/2HZvdhuf//Lzf/3p+794+9v/9tt//8nPtt2b3cavfvX5Tz/97IsXH7z44D/+87+23Zvdxqf//cvPf/fr114cHX3zN09/u+3e7DZe/u53n724/+679+/fP/qfL8UPp20Qys//9+jo6P7R0Q9+u6dqJV7+4ve/+f3/yTv6A6h77PFlRrzEeHITQXorAb+/KnkFLvANCPGidUF6MZLZ4aWHbwFnlx8gsc7gLd5EnF2umP/qk67kdPsDetB3y4NgRI0u6uWlN5ndcfCnQIqiEBa8nxEEOrOUM+X0jD/Dj+bAGx//gdfp2exXA3HUcJwhN6KHZXTcHDlBpTy4ycoCOw/+DEhSzs4InjgDOqBk4Qe14AP8sfC7hamD8ux9y6TEuCJXE7Uuw4CGOYLmuvqXWq8wIUAEaBRQZiECqALSsDrxsarhElQWenxtjz322GOPPfbYY489Non/B0e7ZynVVKArAAAAAElFTkSuQmCC"}}},{"cell_type":"markdown","source":"\n\nHi all! This is a beginner's attempt to predict House prices. I believe this dataset is a great starter dataset for as the columns are not overwhelming. I have tried to use as many techniques as possible to expand my understanding. I am sure there are many gaps and mistakes in my attempt but I was still able to learn a lot. Here below you can see a detailed table of contents of the work:\n\n**1. Preproccessing & EDA**\n\n**2. Feature Engineeering**\n\n**3. Spilliting Train and Test**\n\n**4. Model Definitions & Training**\n\n**5. Blending and Optimising Weights**\n\n**6. Submission**\n\nPlease feel free to leave leave comments and suggestions as I will be happy to learn. \n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T18:51:46.903801Z","iopub.execute_input":"2022-07-30T18:51:46.904137Z","iopub.status.idle":"2022-07-30T18:51:46.914564Z","shell.execute_reply.started":"2022-07-30T18:51:46.904108Z","shell.execute_reply":"2022-07-30T18:51:46.913305Z"},"jupyter":{"source_hidden":true}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Essentials\nimport numpy as np\nimport pandas as pd\nimport datetime\nimport random\n\n# Plots\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Models\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor, AdaBoostRegressor, BaggingRegressor\nfrom sklearn.kernel_ridge import KernelRidge\nfrom sklearn.linear_model import Ridge, RidgeCV\nfrom sklearn.linear_model import ElasticNet, ElasticNetCV\nfrom sklearn.svm import SVR\nfrom mlxtend.regressor import StackingCVRegressor\nimport lightgbm as lgb\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom catboost import CatBoostRegressor\n\n# Stats\nfrom scipy.stats import skew, norm\nfrom scipy.special import boxcox1p\nfrom scipy.stats import boxcox_normmax\n\n# Misc\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import KFold, cross_val_score\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import scale\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.decomposition import PCA\n\npd.set_option('display.max_columns', None)\n\n# Ignore useless warnings\nimport warnings\nwarnings.filterwarnings(action=\"ignore\")\npd.options.display.max_seq_items = 8000\npd.options.display.max_rows = 8000\n\nimport os\nprint(os.listdir(\"../input/house-prices-advanced-regression-techniques\"))\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-30T18:51:47.960952Z","iopub.execute_input":"2022-07-30T18:51:47.962025Z","iopub.status.idle":"2022-07-30T18:51:47.974588Z","shell.execute_reply.started":"2022-07-30T18:51:47.961975Z","shell.execute_reply":"2022-07-30T18:51:47.973241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n**Loading Datasets**\n","metadata":{}},{"cell_type":"code","source":"#Load Datasets\ntrain = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ntest = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\ntrain.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:51:48.998543Z","iopub.execute_input":"2022-07-30T18:51:48.999567Z","iopub.status.idle":"2022-07-30T18:51:49.041356Z","shell.execute_reply.started":"2022-07-30T18:51:48.999520Z","shell.execute_reply":"2022-07-30T18:51:49.040478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Preview Datasets\ntrain.head(5).sort_values('SalePrice', ascending = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:51:50.709570Z","iopub.execute_input":"2022-07-30T18:51:50.710192Z","iopub.status.idle":"2022-07-30T18:51:50.761216Z","shell.execute_reply.started":"2022-07-30T18:51:50.710157Z","shell.execute_reply":"2022-07-30T18:51:50.760261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# **1. PREPROCESSING & EDA**\n","metadata":{}},{"cell_type":"markdown","source":"\n> **Distributiuon of Traget**\n","metadata":{}},{"cell_type":"code","source":"sns.set_style(\"white\")\nsns.set_color_codes(palette='deep')\nf, ax = plt.subplots(figsize=(8, 7))\nsns.set(font_scale = 1)\n#Check the new distribution \nsns.distplot(train['SalePrice'], color=\"b\");\nax.xaxis.grid(False)\nax.set(ylabel=\"Frequency\")\nax.set(xlabel=\"SalePrice\")\nax.set(title=\"SalePrice distribution\")\nsns.despine(trim=True, left=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:51:52.157661Z","iopub.execute_input":"2022-07-30T18:51:52.158231Z","iopub.status.idle":"2022-07-30T18:51:52.444572Z","shell.execute_reply.started":"2022-07-30T18:51:52.158195Z","shell.execute_reply":"2022-07-30T18:51:52.443373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*Looks like our Target Label is skewed, we will fix that later*","metadata":{}},{"cell_type":"code","source":"# Skew and kurt\nprint(\"Skewness: %f\" % train['SalePrice'].skew())\nprint(\"Kurtosis: %f\" % train['SalePrice'].kurt())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:51:53.116055Z","iopub.execute_input":"2022-07-30T18:51:53.116703Z","iopub.status.idle":"2022-07-30T18:51:53.122901Z","shell.execute_reply.started":"2022-07-30T18:51:53.116667Z","shell.execute_reply":"2022-07-30T18:51:53.121959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nLets **Visualize** our *Numeric Features*\n","metadata":{}},{"cell_type":"code","source":"# Finding numeric features\nnumeric_dtypes = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\nnumeric = []\nfor i in train.columns:\n    if train[i].dtype in numeric_dtypes:\n        if i in ['Id', 'SalePrice']:\n            pass\n        else:\n            numeric.append(i)     \n# visualising some more outliers in the data values\nfig, axs = plt.subplots(ncols=2, nrows=1, figsize=(12, 120))\nplt.subplots_adjust(right=2)\nplt.subplots_adjust(top=2)\nsns.color_palette(\"husl\", 8)\nfor i, feature in enumerate(list(train[numeric]), 1):\n   # if(feature=='MiscVal'):\n   #    break\n    plt.subplot(len(list(numeric)), 3, i)\n    sns.scatterplot(x=feature, y='SalePrice', hue='SalePrice', palette='Blues', data=train)\n    sns.regplot(x=feature, y='SalePrice', scatter=False, data=train)\n    \n    plt.xlabel('{}'.format(feature), size=15,labelpad=12.5)\n    plt.ylabel('SalePrice', size=15, labelpad=12.5)\n    \n    for j in range(2):\n        plt.tick_params(axis='x', labelsize=12)\n        plt.tick_params(axis='y', labelsize=12)\n    \n    plt.legend(loc='best', prop={'size': 10})\n        \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:51:54.053824Z","iopub.execute_input":"2022-07-30T18:51:54.054163Z","iopub.status.idle":"2022-07-30T18:52:12.209871Z","shell.execute_reply.started":"2022-07-30T18:51:54.054136Z","shell.execute_reply":"2022-07-30T18:52:12.208981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nLooks like some features have direct **linear relationship** with our Traget like **GrLivArea**(other area paramters) and **OverallQual** especially.\n\n\n\n\n\n\nLets looks at the **Heatmap or corr Matrix**","metadata":{}},{"cell_type":"code","source":"# Correlation Matrix\n\nf, ax = plt.subplots(figsize=(30, 25))\nmat = train.corr('pearson')\nmask = np.triu(np.ones_like(mat, dtype=bool))\ncmap = sns.diverging_palette(230, 20, as_cmap=True)\nsns.heatmap(mat, mask=mask, cmap=cmap, vmax=1, center=0, annot = True,\n            square=True, linewidths=.5, cbar_kws={\"shrink\": .5})\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:12.212054Z","iopub.execute_input":"2022-07-30T18:52:12.212698Z","iopub.status.idle":"2022-07-30T18:52:16.046299Z","shell.execute_reply.started":"2022-07-30T18:52:12.212656Z","shell.execute_reply":"2022-07-30T18:52:16.045487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As suspected **OverallQual, GrLivArea, GarargeArea/Carsm Basmt/1stFloor** are highly correlated\n\n\n","metadata":{}},{"cell_type":"markdown","source":"\n\nLets take a look at some **Individual Plots** with our Target on these variable\n","metadata":{}},{"cell_type":"code","source":"# OverallQuall - looks highly co-related Peasrson- 0.79\nPearson_GrLiv = 0.79\nfigure, ax = plt.subplots(1,3, figsize = (20,8))\nsns.stripplot(data=train, x = 'OverallQual', y='SalePrice', ax = ax[0])\nsns.violinplot(data=train, x = 'OverallQual', y='SalePrice', ax = ax[1])\nsns.boxplot(data=train, x = 'OverallQual', y='SalePrice', ax = ax[2])\nplt.legend(['$Pearson=$ {:.2f}'.format(Pearson_GrLiv)], loc = 'best')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:16.047752Z","iopub.execute_input":"2022-07-30T18:52:16.048314Z","iopub.status.idle":"2022-07-30T18:52:16.879388Z","shell.execute_reply.started":"2022-07-30T18:52:16.048278Z","shell.execute_reply":"2022-07-30T18:52:16.878443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GrLivArea vs SalePrice [corr = 0.71]\n\nPearson_GrLiv = 0.71\nplt.figure(figsize = (12,6))\nsns.regplot(data=train, x = 'GrLivArea', y='SalePrice', scatter_kws={'alpha':0.2})\nplt.title('GrLivArea vs SalePrice', fontsize = 12)\nplt.legend(['$Pearson=$ {:.2f}'.format(Pearson_GrLiv)], loc = 'best')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:16.881983Z","iopub.execute_input":"2022-07-30T18:52:16.882920Z","iopub.status.idle":"2022-07-30T18:52:17.333274Z","shell.execute_reply.started":"2022-07-30T18:52:16.882881Z","shell.execute_reply":"2022-07-30T18:52:17.332329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Pearson_TBSF = 0.63\nplt.figure(figsize = (12,6))\nsns.regplot(data=train, x = 'TotalBsmtSF', y='SalePrice', scatter_kws={'alpha':0.2})\nplt.title('TotalBsmtSF vs SalePrice', fontsize = 12)\nplt.legend(['$Pearson=$ {:.2f}'.format(Pearson_TBSF)], loc = 'best')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:17.334645Z","iopub.execute_input":"2022-07-30T18:52:17.335371Z","iopub.status.idle":"2022-07-30T18:52:17.794317Z","shell.execute_reply.started":"2022-07-30T18:52:17.335331Z","shell.execute_reply":"2022-07-30T18:52:17.793460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# YearBuilt vs SalePrice\n\nPearson_YrBlt = 0.56\nplt.figure(figsize = (12,6))\nsns.regplot(data=train, x = 'YearBuilt', y='SalePrice', scatter_kws={'alpha':0.2})\nplt.title('YearBuilt vs SalePrice', fontsize = 12)\nplt.legend(['$Pearson=$ {:.2f}'.format(Pearson_YrBlt)], loc = 'best')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:17.795830Z","iopub.execute_input":"2022-07-30T18:52:17.796162Z","iopub.status.idle":"2022-07-30T18:52:18.251875Z","shell.execute_reply.started":"2022-07-30T18:52:17.796128Z","shell.execute_reply":"2022-07-30T18:52:18.250972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*Lets  see of our most importatnt features are also correlated*","metadata":{}},{"cell_type":"code","source":"cols = ['GrLivArea','LotArea','TotalBsmtSF','YearBuilt','OverallQual', 'LotFrontage']\ndf = train[cols]\ncorr = df.corr()\nsns.heatmap(corr, cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:18.253357Z","iopub.execute_input":"2022-07-30T18:52:18.253737Z","iopub.status.idle":"2022-07-30T18:52:18.504377Z","shell.execute_reply.started":"2022-07-30T18:52:18.253702Z","shell.execute_reply":"2022-07-30T18:52:18.503531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# **2. Feature Engineeering**\n","metadata":{}},{"cell_type":"markdown","source":"*Lets drop [Id] features as it as no signficane to our regression*","metadata":{}},{"cell_type":"code","source":"# Remove the Ids from train and test, as they are unique for each row and hence not useful for the model\ntrain_ID = train['Id']\ntest_ID = test['Id']\ntrain.drop(['Id'], axis=1, inplace=True)\ntest.drop(['Id'], axis=1, inplace=True)\ntrain.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:18.505840Z","iopub.execute_input":"2022-07-30T18:52:18.506184Z","iopub.status.idle":"2022-07-30T18:52:18.519799Z","shell.execute_reply.started":"2022-07-30T18:52:18.506138Z","shell.execute_reply":"2022-07-30T18:52:18.518596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*Lets fix skewness in our target SalesPrice as it important for our models*","metadata":{}},{"cell_type":"code","source":"#Removing Sales Price skewness\ntrain['SalePrice']= np.log1p(train['SalePrice'])\ntrain.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:18.521437Z","iopub.execute_input":"2022-07-30T18:52:18.521867Z","iopub.status.idle":"2022-07-30T18:52:18.533121Z","shell.execute_reply.started":"2022-07-30T18:52:18.521830Z","shell.execute_reply":"2022-07-30T18:52:18.532152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_style(\"white\")\nsns.set_color_codes(palette='deep')\nf, ax = plt.subplots(figsize=(8, 7))\n#Check the new distribution \nsns.distplot(train['SalePrice'], color=\"b\", fit=norm);\n\n# Get the fitted parameters used by the function\n(mu, sigma) = norm.fit(train['SalePrice'])\nsigma2 = np.std(train['SalePrice'])\nprint( '\\n mu = {:.2f} and sigma = {:.2f}\\n'.format(mu, sigma))\n#Now plot the distribution\nplt.legend(['Normal dist. ($\\mu=$ {:.2f} and $\\sigma=$ {:.2f} )'.format(mu, sigma)],\n            loc='best')\nax.xaxis.grid(False)\nax.set(ylabel=\"Frequency\")\nax.set(xlabel=\"SalePrice\")\nax.set(title=\"SalePrice distribution\")\nsns.despine(trim=True, left=True)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:18.538236Z","iopub.execute_input":"2022-07-30T18:52:18.538491Z","iopub.status.idle":"2022-07-30T18:52:18.876336Z","shell.execute_reply.started":"2022-07-30T18:52:18.538468Z","shell.execute_reply":"2022-07-30T18:52:18.875364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Skew and kurt\nprint(\"New Skewness: %f\" % train['SalePrice'].skew())\nprint(\"New Kurtosis: %f\" % train['SalePrice'].kurt())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:18.877699Z","iopub.execute_input":"2022-07-30T18:52:18.878305Z","iopub.status.idle":"2022-07-30T18:52:18.885940Z","shell.execute_reply.started":"2022-07-30T18:52:18.878266Z","shell.execute_reply":"2022-07-30T18:52:18.884798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The SalePrice is now normally distributed","metadata":{}},{"cell_type":"markdown","source":"**Combining train & Test Features**\n> This should make our transformation and addition of new features easier\n","metadata":{}},{"cell_type":"code","source":"# Split features and labels and record our label indices\ntrain_labels = train['SalePrice'].reset_index(drop=True)\ntrain_features = train.drop(['SalePrice'], axis=1)\ntest_features = test\n\n# Combine train and test features in order to apply the feature transformation pipeline to the entire dataset\nall_features = pd.concat([train_features, test_features]).reset_index(drop=True)\nall_features.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:18.887255Z","iopub.execute_input":"2022-07-30T18:52:18.888217Z","iopub.status.idle":"2022-07-30T18:52:18.914879Z","shell.execute_reply.started":"2022-07-30T18:52:18.888179Z","shell.execute_reply":"2022-07-30T18:52:18.913865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n**Handling *NaNs***\n\n","metadata":{}},{"cell_type":"code","source":"#custom function to help us look at missing data\ndef missing (train, n=5):\n    missing_number = train.isnull().sum().sort_values(ascending=False)\n    missing_percent = ((train.isnull().sum()/train.isnull().count())*100).sort_values(ascending=False)\n    missing_values = pd.concat([missing_number,missing_percent], axis=1, keys=['Missing_Number', 'Missing_Percent']).reset_index()\n    return missing_values[missing_values['Missing_Number']>0][:n]\n\nmissing(all_features,5)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:18.916253Z","iopub.execute_input":"2022-07-30T18:52:18.917146Z","iopub.status.idle":"2022-07-30T18:52:18.971299Z","shell.execute_reply.started":"2022-07-30T18:52:18.917120Z","shell.execute_reply":"2022-07-30T18:52:18.970080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting Nans for a quick Visual\nplt.figure(figsize = (15,5))\nx = missing(all_features, 100)['index']\ny = missing(all_features,100)['Missing_Percent']\nsns.barplot(x = x, y = y)\nplt.xticks(rotation=45)\nplt.title('Features containing Nan')\nplt.xlabel('Features')\nplt.ylabel('% of Missing Data')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:18.972664Z","iopub.execute_input":"2022-07-30T18:52:18.973093Z","iopub.status.idle":"2022-07-30T18:52:19.529223Z","shell.execute_reply.started":"2022-07-30T18:52:18.973056Z","shell.execute_reply":"2022-07-30T18:52:19.528312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Some of the non-numeric predictors are stored as numbers; convert them into strings \nall_features['MSSubClass'] = all_features['MSSubClass'].apply(str)\nall_features['YrSold'] = all_features['YrSold'].astype(str)\nall_features['MoSold'] = all_features['MoSold'].astype(str)\n\n# Filling Categorical NaN (with Median)\nall_features['Functional'] = all_features['Functional'].fillna('Typ')\nall_features['Electrical'] = all_features['Electrical'].fillna(\"SBrkr\")\nall_features['KitchenQual'] = all_features['KitchenQual'].fillna(\"TA\")\nall_features['Exterior1st'] = all_features['Exterior1st'].fillna(all_features['Exterior1st'].mode()[0])\nall_features['Exterior2nd'] = all_features['Exterior2nd'].fillna(all_features['Exterior2nd'].mode()[0])\nall_features['SaleType'] = all_features['SaleType'].fillna(all_features['SaleType'].mode()[0])\nall_features[\"PoolQC\"] = all_features[\"PoolQC\"].fillna(\"None\")\nall_features[\"Alley\"] = all_features[\"Alley\"].fillna(\"None\")\nall_features['FireplaceQu'] = all_features['FireplaceQu'].fillna(\"None\")\nall_features['Fence'] = all_features['Fence'].fillna(\"None\")\nall_features['MiscFeature'] = all_features['MiscFeature'].fillna(\"None\")\n\n# Filling rest of the Categorical NaNs as Zero or None\nfor col in ('GarageArea', 'GarageCars'):\n    all_features[col] = all_features[col].fillna(0)\n        \nfor col in ['GarageType', 'GarageFinish', 'GarageQual', 'GarageCond']:\n    all_features[col] = all_features[col].fillna('None')\n    \nfor col in ('BsmtQual', 'BsmtCond', 'BsmtExposure', 'BsmtFinType1', 'BsmtFinType2'):\n    all_features[col] = all_features[col].fillna('None')\n    \n# Checking the features with NaN remained out\nmissing(all_features,5)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:19.530788Z","iopub.execute_input":"2022-07-30T18:52:19.531428Z","iopub.status.idle":"2022-07-30T18:52:19.619455Z","shell.execute_reply.started":"2022-07-30T18:52:19.531390Z","shell.execute_reply":"2022-07-30T18:52:19.618407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n**Impusting other Numeric NaNs**\n","metadata":{}},{"cell_type":"code","source":"# Removing the useless variables, t\nuseless = ['GarageYrBlt','Utilities', 'Street', 'PoolQC'] \nfor col in useless:\n    if col in all_features:\n        all_features = all_features.drop(useless, axis = 1)\n\n\n# Imputing with KnnRegressor (we can also use different Imputers)\n\ndef impute_knn(df):\n    ttn = df.select_dtypes(include=[np.number])\n    ttc = df.select_dtypes(exclude=[np.number])\n\n    cols_nan = ttn.columns[ttn.isna().any()].tolist()         # columns w/ nan \n    cols_no_nan = ttn.columns.difference(cols_nan).values     # columns w/n nan\n\n    for col in cols_nan:\n        imp_test = ttn[ttn[col].isna()]   # indicies which have missing data will become our test set\n        imp_train = ttn.dropna()          # all indicies which which have no missing data \n        print('Updating Col: {}, MissingCount: {}, TrainingWith: {}' .format(col, imp_test.shape[0], imp_train.shape[0]))\n        model = KNeighborsRegressor(n_neighbors=5)  # KNR Unsupervised Approach\n        knr = model.fit(imp_train[cols_no_nan], imp_train[col])\n        ttn.loc[ttn[col].isna(), col] = knr.predict(imp_test[cols_no_nan])\n    \n    return pd.concat([ttn,ttc],axis=1)\n\nall_features = impute_knn(all_features)\n\n\n# setting remainder Categorical to None\nobjects = []\nfor i in all_features.columns:\n    if all_features[i].dtype == object:\n        objects.append(i)\n        if all_features[i].isnull().values.any():\n            print('Updating to \"NONE\" col: {},  Values: {}' .format(i, all_features[i].isnull().sum()))\nall_features.update(all_features[objects].fillna('None'))\n\nmissing(all_features)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:19.621182Z","iopub.execute_input":"2022-07-30T18:52:19.621549Z","iopub.status.idle":"2022-07-30T18:52:19.913368Z","shell.execute_reply.started":"2022-07-30T18:52:19.621514Z","shell.execute_reply":"2022-07-30T18:52:19.912408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n*Adding some **New Features** that could be useful for our models*\n","metadata":{}},{"cell_type":"code","source":"#adding new features\nall_features[\"SqFtPerRoom\"] = all_features[\"GrLivArea\"] / (all_features[\"TotRmsAbvGrd\"] +\n                                                       all_features[\"FullBath\"] +\n                                                       all_features[\"HalfBath\"] +\n                                                       all_features[\"KitchenAbvGr\"])\nall_features['Total_Home_Quality'] = all_features['OverallQual'] + all_features['OverallCond']\nall_features['Total_Bathrooms'] = (all_features['FullBath'] + (0.5 * all_features['HalfBath']) +\n                               all_features['BsmtFullBath'] + (0.5 * all_features['BsmtHalfBath']))\nall_features[\"HighQualSF\"] = all_features[\"1stFlrSF\"] + all_features[\"2ndFlrSF\"]\n\n\n# More helpful categorical features\ndef catFeat(all_features):\n    all_features['BsmtFinType1_Unf'] = 1*(all_features['BsmtFinType1'] == 'Unf')\n    all_features['HasWoodDeck'] = (all_features['WoodDeckSF'] > 0) * 1\n    all_features['HasOpenPorch'] = (all_features['OpenPorchSF'] > 0) * 1\n    all_features['HasEnclosedPorch'] = (all_features['EnclosedPorch'] > 0) * 1\n    all_features['Has3SsnPorch'] = (all_features['3SsnPorch'] > 0) * 1\n    all_features['HasScreenPorch'] = (all_features['ScreenPorch'] > 0) * 1\n    all_features['YearsSinceRemodel'] = all_features['YrSold'].astype(int) - all_features['YearRemodAdd'].astype(int)\n    all_features['Haspool'] = all_features['PoolArea'].apply(lambda x: 1 if x > 0 else 0)\n    all_features['Has2ndfloor'] = all_features['2ndFlrSF'].apply(lambda x: 1 if x > 0 else 0)\n    all_features['Hasgarage'] = all_features['GarageArea'].apply(lambda x: 1 if x > 0 else 0)\n    all_features['Hasbsmt'] = all_features['TotalBsmtSF'].apply(lambda x: 1 if x > 0 else 0)\n    all_features['Hasfireplace'] = all_features['Fireplaces'].apply(lambda x: 1 if x > 0 else 0)\n    return all_features\n\nall_features = catFeat(all_features)\n\n# Fetch all numeric features\nnumeric_features = all_features.dtypes[all_features.dtypes != object].index\nskewed_features = all_features[numeric_features].apply(lambda x: skew(x)).sort_values(ascending=False)\nhigh_skew = skewed_features[skewed_features > 0.5]\nskew_index = high_skew.index\n# Normalize skewed features using log_transformation\nall_feature_nskew = all_features.copy()\n\nfor i in skew_index:\n    all_feature_nskew[i] = np.log1p(all_feature_nskew[i])\n\nall_features.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:19.915178Z","iopub.execute_input":"2022-07-30T18:52:19.915570Z","iopub.status.idle":"2022-07-30T18:52:19.985031Z","shell.execute_reply.started":"2022-07-30T18:52:19.915535Z","shell.execute_reply":"2022-07-30T18:52:19.984099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating dummy variables from categorical features\nall_features_dummy = pd.get_dummies(all_feature_nskew)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:19.986120Z","iopub.execute_input":"2022-07-30T18:52:19.986381Z","iopub.status.idle":"2022-07-30T18:52:20.036183Z","shell.execute_reply.started":"2022-07-30T18:52:19.986356Z","shell.execute_reply":"2022-07-30T18:52:20.035249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_features_dummy.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:20.037482Z","iopub.execute_input":"2022-07-30T18:52:20.037909Z","iopub.status.idle":"2022-07-30T18:52:20.048054Z","shell.execute_reply.started":"2022-07-30T18:52:20.037871Z","shell.execute_reply":"2022-07-30T18:52:20.047000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Re-splitting Training and Test Sets","metadata":{}},{"cell_type":"code","source":"# first lets remove any duplicate cols\nall_features_dummy = all_features_dummy.loc[:,~all_features_dummy.columns.duplicated()].copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:20.049961Z","iopub.execute_input":"2022-07-30T18:52:20.050467Z","iopub.status.idle":"2022-07-30T18:52:20.060078Z","shell.execute_reply.started":"2022-07-30T18:52:20.050334Z","shell.execute_reply":"2022-07-30T18:52:20.059168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = all_features_dummy.iloc[:len(train_labels), :]\nX_test = all_features_dummy.iloc[len(train_labels):, :]\nX.shape, train_labels.shape, X_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:20.061661Z","iopub.execute_input":"2022-07-30T18:52:20.062060Z","iopub.status.idle":"2022-07-30T18:52:20.072211Z","shell.execute_reply.started":"2022-07-30T18:52:20.062022Z","shell.execute_reply":"2022-07-30T18:52:20.071276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Exporting our X and y for determining model metrics..explained later..","metadata":{}},{"cell_type":"code","source":"X.to_csv('processed_X.csv',index=False)\ntrain_labels.to_csv('processed_y.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:20.073527Z","iopub.execute_input":"2022-07-30T18:52:20.073785Z","iopub.status.idle":"2022-07-30T18:52:20.213068Z","shell.execute_reply.started":"2022-07-30T18:52:20.073761Z","shell.execute_reply":"2022-07-30T18:52:20.212075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **4. Training Models**\nKey features of the model training process:\nCross Validation: Using 12-fold cross-validation\nModels: On each run of cross-validation I fit 9 models (ridge, svr, gradient boosting, random forest, xgboost, lightgbm regressors, catboost)\nStacking: In addition, I trained a meta StackingCVRegressor optimized using xgboost\nBlending: All models trained will overfit the training data to varying degrees. Therefore, to make final predictions, I blended their predictions together to get more robust predictions.\n\n**Setup cross validation and define error metrics**","metadata":{}},{"cell_type":"code","source":"# Setup cross validation folds\nkf = KFold(n_splits=12, random_state=42, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:20.214523Z","iopub.execute_input":"2022-07-30T18:52:20.214879Z","iopub.status.idle":"2022-07-30T18:52:20.220217Z","shell.execute_reply.started":"2022-07-30T18:52:20.214845Z","shell.execute_reply":"2022-07-30T18:52:20.219150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define error metrics\ndef rmsle(y, y_pred):\n    return np.sqrt(mean_squared_error(y, y_pred))\n\ndef cv_rmse(model, X=X):\n    rmse = np.sqrt(-cross_val_score(model, X, train_labels, scoring=\"neg_mean_squared_error\", cv=kf))\n    return (rmse)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:20.221916Z","iopub.execute_input":"2022-07-30T18:52:20.223157Z","iopub.status.idle":"2022-07-30T18:52:20.230207Z","shell.execute_reply.started":"2022-07-30T18:52:20.223104Z","shell.execute_reply":"2022-07-30T18:52:20.229246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model Defs**\n\n> Model paramters were optimised in a seperate notebook using processed features and labels","metadata":{}},{"cell_type":"code","source":"# Light Gradient Boosting Regressor\nlightgbm = LGBMRegressor(objective='regression',\n                         boosting_type='gbdt',\n                       learning_rate=0.03,\n                         num_leaves =106,\n                       n_estimators=6000,\n                         max_depth = 4,\n                         min_child_samples = 1,\n                         min_split_gain=0, \n                       subsample=0.4,\n                       subsample_freq=3, \n                       colsample_bytree=0.6,\n                         reg_alpha=0.008,\n                         reg_lambda=0.03,\n                       verbose=-1,\n                       random_state=42)\n\n# XGBoost Regressor\nxgboost = XGBRegressor(learning_rate=0.018,\n                       n_estimators=10000,\n                       max_depth=10,\n                       min_child_weight=3,\n                       subsample=0.6,\n                       colsample_bytree=0.7,\n                       seed=42,\n                       alpha=0.124,\n                       reg_lambda= 0.197,\n                       random_state=2020\n                       ,\n                       tree_method='gpu_hist'\n                      )\n\n# Ridge Regressor\nridge_alphas = [1e-15, 1e-10, 1e-8, 9e-4, 7e-4, 5e-4, 3e-4, 1e-4, 1e-3, 5e-2, 1e-2, 0.1, 0.3, 1, 3, 5, 10, 15, 18, 20, 30, 50, 75, 100]\nridge = make_pipeline(RobustScaler(), RidgeCV(alphas=ridge_alphas, cv=kf))\n\n# Support Vector Regressor\nsvr = make_pipeline(RobustScaler(), SVR(C= 20, epsilon= 0.008, gamma=0.0003))\n\n# Gradient Boosting Regressor\ngbr = GradientBoostingRegressor(n_estimators=6000,\n                                learning_rate=0.01,\n                                max_depth=4,\n                                max_features='sqrt',\n                                min_samples_leaf=15,\n                                min_samples_split=10,\n                                loss='huber',\n                                random_state=42)  \n\n# Random Forest Regressor\nrf = RandomForestRegressor(n_estimators=1200,\n                          max_depth=15,\n                          min_samples_split=5,\n                          min_samples_leaf=5,\n                          max_features=None,\n                          oob_score=True,\n                          random_state=42)\n\n# Cat Bosst Regressor\ncatb = CatBoostRegressor(task_type= 'GPU', \n                            verbose = False)\n\n# Stack up all the models above, optimized using xgboost\nstack_gen = StackingCVRegressor(regressors=(xgboost, lightgbm, svr, ridge, gbr, rf, catb),\n                                meta_regressor=xgboost,\n                                use_features_in_secondary=True)\n\n# Initiating Scores Dict\nscores = {}","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:20.231667Z","iopub.execute_input":"2022-07-30T18:52:20.232234Z","iopub.status.idle":"2022-07-30T18:52:20.247063Z","shell.execute_reply.started":"2022-07-30T18:52:20.232200Z","shell.execute_reply":"2022-07-30T18:52:20.246048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Training Models***\n\nGet cross validation scores for each model","metadata":{"execution":{"iopub.status.busy":"2022-07-26T22:28:21.564621Z","iopub.execute_input":"2022-07-26T22:28:21.565511Z","iopub.status.idle":"2022-07-26T22:28:21.573179Z","shell.execute_reply.started":"2022-07-26T22:28:21.565467Z","shell.execute_reply":"2022-07-26T22:28:21.571581Z"}}},{"cell_type":"code","source":"score = cv_rmse(lightgbm)\nprint(\"lightgbm: {:.4f} ({:.4f})\".format(score.mean(), score.std()))\nscores['lgb'] = (score.mean(), score.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:52:20.248678Z","iopub.execute_input":"2022-07-30T18:52:20.249089Z","iopub.status.idle":"2022-07-30T18:53:33.968814Z","shell.execute_reply.started":"2022-07-30T18:52:20.249053Z","shell.execute_reply":"2022-07-30T18:53:33.967994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = cv_rmse(xgboost)\nprint(\"xgboost: {:.4f} ({:.4f})\".format(score.mean(), score.std()))\nscores['xgb'] = (score.mean(), score.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:53:33.971496Z","iopub.execute_input":"2022-07-30T18:53:33.973832Z","iopub.status.idle":"2022-07-30T19:03:20.938723Z","shell.execute_reply.started":"2022-07-30T18:53:33.973794Z","shell.execute_reply":"2022-07-30T19:03:20.937632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = cv_rmse(svr)\nprint(\"SVR: {:.4f} ({:.4f})\".format(score.mean(), score.std()))\nscores['svr'] = (score.mean(), score.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:03:20.946886Z","iopub.execute_input":"2022-07-30T19:03:20.947462Z","iopub.status.idle":"2022-07-30T19:03:26.708585Z","shell.execute_reply.started":"2022-07-30T19:03:20.947403Z","shell.execute_reply":"2022-07-30T19:03:26.707456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = cv_rmse(ridge)\nprint(\"ridge: {:.4f} ({:.4f})\".format(score.mean(), score.std()))\nscores['ridge'] = (score.mean(), score.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:03:26.713084Z","iopub.execute_input":"2022-07-30T19:03:26.715686Z","iopub.status.idle":"2022-07-30T19:04:54.816088Z","shell.execute_reply.started":"2022-07-30T19:03:26.715642Z","shell.execute_reply":"2022-07-30T19:04:54.814861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = cv_rmse(rf)\nprint(\"rf: {:.4f} ({:.4f})\".format(score.mean(), score.std()))\nscores['rf'] = (score.mean(), score.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:04:54.817717Z","iopub.execute_input":"2022-07-30T19:04:54.818066Z","iopub.status.idle":"2022-07-30T19:09:22.437366Z","shell.execute_reply.started":"2022-07-30T19:04:54.818031Z","shell.execute_reply":"2022-07-30T19:09:22.436387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = cv_rmse(gbr)\nprint(\"gbr: {:.4f} ({:.4f})\".format(score.mean(), score.std()))\nscores['gbr'] = (score.mean(), score.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:09:22.438892Z","iopub.execute_input":"2022-07-30T19:09:22.439500Z","iopub.status.idle":"2022-07-30T19:14:59.996881Z","shell.execute_reply.started":"2022-07-30T19:09:22.439461Z","shell.execute_reply":"2022-07-30T19:14:59.994849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = cv_rmse(catb)\nprint(\"cat: {:.4f} ({:.4f})\".format(score.mean(), score.std()))\nscores['cat'] = (score.mean(), score.std())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:15:00.001670Z","iopub.execute_input":"2022-07-30T19:15:00.004865Z","iopub.status.idle":"2022-07-30T19:17:34.970670Z","shell.execute_reply.started":"2022-07-30T19:15:00.004822Z","shell.execute_reply":"2022-07-30T19:17:34.969821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*Stacking takes a very long time to run so inputting scores from my previous run for refernce*","metadata":{}},{"cell_type":"code","source":"# score = cv_rmse(stack_gen)\n# print(\"stack_gen: {:.4f} ({:.4f})\".format(score.mean(), score.std()))\n# scores['stack_gen'] = (score.mean(), score.std())\n# Manually inputting stack score as this take a long time to run so pulled scores from a previous run\nprint(\"stack_gen: {:.4f} ({:.4f})\".format(0.1190 , 0.0185))\nscores['stack_gen'] = (0.1190, 0.0185)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:17:34.974371Z","iopub.execute_input":"2022-07-30T19:17:34.975296Z","iopub.status.idle":"2022-07-30T19:17:34.981762Z","shell.execute_reply.started":"2022-07-30T19:17:34.975260Z","shell.execute_reply":"2022-07-30T19:17:34.980501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('stack_gen')\nstack_gen_model = stack_gen.fit(np.array(X), np.array(train_labels))\nprint('lightgbm')\nlgb_model_full_data = lightgbm.fit(X, train_labels)\nprint('xgboost')\nxgb_model_full_data = xgboost.fit(X, train_labels)\nprint('Svr')\nsvr_model_full_data = svr.fit(X, train_labels)\nprint('Ridge')\nridge_model_full_data = ridge.fit(X, train_labels)\nprint('RandomForest')\nrf_model_full_data = rf.fit(X, train_labels)\nprint('GradientBoosting')\ngbr_model_full_data = gbr.fit(X, train_labels)\nprint('CatBoosting')\ncat_model_full_data = catb.fit(X, train_labels)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T19:17:34.983211Z","iopub.execute_input":"2022-07-30T19:17:34.983636Z","iopub.status.idle":"2022-07-30T19:31:32.277580Z","shell.execute_reply.started":"2022-07-30T19:17:34.983594Z","shell.execute_reply":"2022-07-30T19:31:32.276619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_cv_score = pd.DataFrame(scores).T.reset_index()\nfinal_cv_score.columns =['Regressors', 'RMSE_mean', 'std_mean']\nfinal_cv_score","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:04:09.797393Z","iopub.execute_input":"2022-07-30T20:04:09.798370Z","iopub.status.idle":"2022-07-30T20:04:09.811937Z","shell.execute_reply.started":"2022-07-30T20:04:09.798324Z","shell.execute_reply":"2022-07-30T20:04:09.810862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10,6))\nax = sns.barplot(final_cv_score.sort_values('RMSE_mean')['Regressors'],final_cv_score.sort_values('RMSE_mean')['RMSE_mean'])\nplt.ylim(0.1, 0.15)\nplt.xlabel('Regressors', fontsize = 12)\nplt.ylabel('CV_Mean_RMSE', fontsize = 12)\nax.bar_label(ax.containers[0], padding=3)\nax.margins(y=0.3)\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:16:06.708305Z","iopub.execute_input":"2022-07-30T20:16:06.708979Z","iopub.status.idle":"2022-07-30T20:16:06.957538Z","shell.execute_reply.started":"2022-07-30T20:16:06.708940Z","shell.execute_reply":"2022-07-30T20:16:06.956379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **5. Blending models and getting predictions**\n> Guessing Blending weights purely based on Cross-Validation performance. We will try to optimise these using a simple minimize optimiser.","metadata":{}},{"cell_type":"code","source":"# Blend models in order to make the final predictions more robust to overfitting\ndef blended_predictions(X):\n    return ((0.05 * ridge_model_full_data.predict(X)) + \\\n            (0.1 * svr_model_full_data.predict(X)) + \\\n            (0.1 * cat_model_full_data.predict(X)) + \\\n            (0.1 * gbr_model_full_data.predict(X)) + \\\n            (0.15 * xgb_model_full_data.predict(X)) + \\\n            (0.1 * lgb_model_full_data.predict(X)) + \\\n            (0.05 * rf_model_full_data.predict(X)) + \\\n            (0.35 * stack_gen_model.predict(np.array(X))))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:19:20.342857Z","iopub.execute_input":"2022-07-30T20:19:20.343231Z","iopub.status.idle":"2022-07-30T20:19:20.351508Z","shell.execute_reply.started":"2022-07-30T20:19:20.343191Z","shell.execute_reply":"2022-07-30T20:19:20.348490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get final precitions from the blended model\nblended_score = rmsle(train_labels, blended_predictions(X))\nscores['blended'] = (blended_score, 0)\nprint('RMSLE score on train data:')\nprint(blended_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:19:26.653576Z","iopub.execute_input":"2022-07-30T20:19:26.653938Z","iopub.status.idle":"2022-07-30T20:19:30.969741Z","shell.execute_reply.started":"2022-07-30T20:19:26.653904Z","shell.execute_reply":"2022-07-30T20:19:30.968864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n**Optimizing Blending Weights**\n","metadata":{}},{"cell_type":"code","source":"models=[]\nmodels.append(ridge_model_full_data)\nmodels.append(svr_model_full_data)\nmodels.append(cat_model_full_data)\nmodels.append(gbr_model_full_data)\nmodels.append(xgb_model_full_data)\nmodels.append(lgb_model_full_data)\nmodels.append(rf_model_full_data)\nmodels.append(stack_gen_model)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:19:43.541509Z","iopub.execute_input":"2022-07-30T20:19:43.541970Z","iopub.status.idle":"2022-07-30T20:19:43.549749Z","shell.execute_reply.started":"2022-07-30T20:19:43.541928Z","shell.execute_reply":"2022-07-30T20:19:43.548802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will be setting the minimum weight to 0.05 and optimising using SLSQP","metadata":{}},{"cell_type":"code","source":"## Optimization\n\nfrom scipy.optimize import minimize\n\npredictions = []\nfor model in models:\n    if model in ['stack_gen_model']:\n        predictions.append(model.predict(np.array(X)))\n    else:\n        predictions.append(model.predict(X))\n    \n\ndef mse_func(weights):\n    #scipy minimize will pass the weights as a numpy array\n    final_prediction = 0\n    for weight, prediction in zip(weights, predictions):\n            final_prediction += weight*prediction\n    #return np.mean((y_test-final_prediction)**2)\n    return np.sqrt(mean_squared_error(train_labels, final_prediction))\n    \nstarting_values = [0]*len(predictions)\n\ncons = ({'type':'ineq','fun':lambda w: 1-sum(w)})\n#our weights are bound between 0 and 1\nbounds = [(0.05,1)]*len(predictions)\n\nres = minimize(mse_func, starting_values, method='SLSQP', bounds=bounds, constraints=cons)\n\nprint('Ensamble Score: {best_score}'.format(best_score=res['fun']))\nprint('Best Weights: {weights}'.format(weights=res['x']))\n\nmlist = ['ridge', 'svr', 'cat', 'gbr' , 'xgb', 'lgb', 'rf', 'stack_gen']\nblend_wts = pd.DataFrame({'model': mlist, 'optimised_wts': list(res['x'])}, columns=['model', 'optimised_wts'])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:34:33.529224Z","iopub.execute_input":"2022-07-30T20:34:33.529685Z","iopub.status.idle":"2022-07-30T20:34:38.636909Z","shell.execute_reply.started":"2022-07-30T20:34:33.529643Z","shell.execute_reply":"2022-07-30T20:34:38.635881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blend_wts","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:35:04.595613Z","iopub.execute_input":"2022-07-30T20:35:04.596033Z","iopub.status.idle":"2022-07-30T20:35:04.610450Z","shell.execute_reply.started":"2022-07-30T20:35:04.595995Z","shell.execute_reply":"2022-07-30T20:35:04.609347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n*Plugging new weights*\n","metadata":{}},{"cell_type":"code","source":"# Blend models in order to make the final predictions more robust to overfitting\ndef blended_predictions(X):\n    return ((0.05 * ridge_model_full_data.predict(X)) + \\\n            (0.05 * svr_model_full_data.predict(X)) + \\\n            (0.05 * cat_model_full_data.predict(X)) + \\\n            (0.05 * gbr_model_full_data.predict(X)) + \\\n            (0.22 * xgb_model_full_data.predict(X)) + \\\n            (0.33 * lgb_model_full_data.predict(X)) + \\\n            (0.05 * rf_model_full_data.predict(X)) + \\\n            (0.20 * stack_gen_model.predict(np.array(X))))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:37:20.849062Z","iopub.execute_input":"2022-07-30T20:37:20.849548Z","iopub.status.idle":"2022-07-30T20:37:20.857334Z","shell.execute_reply.started":"2022-07-30T20:37:20.849509Z","shell.execute_reply":"2022-07-30T20:37:20.856347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get final precitions from the blended model\nblended_score = rmsle(train_labels, blended_predictions(X))\nscores['blended'] = (blended_score, 0)\nprint('RMSLE score on train data:')\nprint(blended_score)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:37:45.336259Z","iopub.execute_input":"2022-07-30T20:37:45.336596Z","iopub.status.idle":"2022-07-30T20:37:49.827828Z","shell.execute_reply.started":"2022-07-30T20:37:45.336569Z","shell.execute_reply":"2022-07-30T20:37:49.827083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nWe can see there is a unignoreable change CV RMSLE score upon changing weights.\n\n**RMLSE:    *0.03734* vs *0.24***\n","metadata":{}},{"cell_type":"markdown","source":"**Lets look at our Blended model performance compared to individual prediction models**","metadata":{}},{"cell_type":"code","source":"# Plot the predictions for each model\nsns.set_style(\"white\")\nfig = plt.figure(figsize=(18, 12))\n\nax = sns.pointplot(x=list(scores.keys()), y=[score for score, _ in scores.values()], markers=['o'], linestyles=['-'])\nfor i, score in enumerate(scores.values()):\n    ax.text(i, score[0] + 0.002, '{:.6f}'.format(score[0]), horizontalalignment='left', size='large', color='black', weight='semibold')\n\nplt.ylabel('Score (RMSE)', size=20, labelpad=12.5)\nplt.xlabel('Model', size=20, labelpad=12.5)\nplt.tick_params(axis='x', labelsize=13.5)\nplt.tick_params(axis='y', labelsize=12.5)\n\nplt.title('Scores of Models', size=20)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:46:42.316862Z","iopub.execute_input":"2022-07-30T20:46:42.317215Z","iopub.status.idle":"2022-07-30T20:46:42.601277Z","shell.execute_reply.started":"2022-07-30T20:46:42.317175Z","shell.execute_reply":"2022-07-30T20:46:42.600363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nThe Blended model outperformed individual models by a SIGNFICANT margin which is jsy amazing to observe !!\n","metadata":{}},{"cell_type":"markdown","source":"# **6. Submission**","metadata":{}},{"cell_type":"code","source":"# Read in sample_submission dataframe\nsubmission = pd.read_csv(\"../input/house-prices-advanced-regression-techniques/sample_submission.csv\")\nsubmission.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:49:11.308625Z","iopub.execute_input":"2022-07-30T20:49:11.309265Z","iopub.status.idle":"2022-07-30T20:49:11.326517Z","shell.execute_reply.started":"2022-07-30T20:49:11.309231Z","shell.execute_reply":"2022-07-30T20:49:11.325651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Append predictions from blended models\nsubmission.iloc[:,1] = np.floor(np.expm1(blended_predictions(X_test)))\nsubmission.to_csv(\"submission_regression1.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T20:50:38.210925Z","iopub.execute_input":"2022-07-30T20:50:38.211589Z","iopub.status.idle":"2022-07-30T20:50:42.639673Z","shell.execute_reply.started":"2022-07-30T20:50:38.211552Z","shell.execute_reply":"2022-07-30T20:50:42.638876Z"},"trusted":true},"execution_count":null,"outputs":[]}]}