{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Importing Necessary Modules and Loading Datasets","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:38.962671Z","iopub.execute_input":"2022-07-19T17:41:38.963061Z","iopub.status.idle":"2022-07-19T17:41:38.967722Z","shell.execute_reply.started":"2022-07-19T17:41:38.963027Z","shell.execute_reply":"2022-07-19T17:41:38.966584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/Covid19-Death-Predictions/train.csv\")\ntest = pd.read_csv(\"../input/Covid19-Death-Predictions/test.csv\")\nsample_submission = pd.read_csv(\"../input/Covid19-Death-Predictions/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.087382Z","iopub.execute_input":"2022-07-19T17:41:39.088055Z","iopub.status.idle":"2022-07-19T17:41:39.395085Z","shell.execute_reply.started":"2022-07-19T17:41:39.088021Z","shell.execute_reply":"2022-07-19T17:41:39.393942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Shape of train.csv is \" + str(train.shape))\nprint(\"Shape of test.csv is \" + str(test.shape))\nprint(\"Shape of sample_submission.csv is \" + str(sample_submission.shape))","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.397232Z","iopub.execute_input":"2022-07-19T17:41:39.397939Z","iopub.status.idle":"2022-07-19T17:41:39.404080Z","shell.execute_reply.started":"2022-07-19T17:41:39.397892Z","shell.execute_reply":"2022-07-19T17:41:39.403310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Starter EDA","metadata":{}},{"cell_type":"markdown","source":"Looking at the number of n/a values in the dataset.","metadata":{}},{"cell_type":"code","source":"for i in train.columns:\n    print(\"n/a values in \" + i + \"is \" + str(train[i].isna().sum())) ","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.405372Z","iopub.execute_input":"2022-07-19T17:41:39.405712Z","iopub.status.idle":"2022-07-19T17:41:39.436107Z","shell.execute_reply.started":"2022-07-19T17:41:39.405682Z","shell.execute_reply":"2022-07-19T17:41:39.435012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in test.columns:\n    print(\"n/a values in \" + i + \" is \" + str(test[i].isna().sum())) ","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.438000Z","iopub.execute_input":"2022-07-19T17:41:39.438717Z","iopub.status.idle":"2022-07-19T17:41:39.457852Z","shell.execute_reply.started":"2022-07-19T17:41:39.438603Z","shell.execute_reply":"2022-07-19T17:41:39.456331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As most n/a values are in vaccinations, and vaccinations weren't avaliable for a significant period of time, I will be assuming that n/a values equal to 0. You may choose to fill those values in a different manner.","metadata":{}},{"cell_type":"code","source":"train = train.fillna(0)\ntest = test.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.459474Z","iopub.execute_input":"2022-07-19T17:41:39.460437Z","iopub.status.idle":"2022-07-19T17:41:39.489228Z","shell.execute_reply.started":"2022-07-19T17:41:39.460390Z","shell.execute_reply":"2022-07-19T17:41:39.488297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.491139Z","iopub.execute_input":"2022-07-19T17:41:39.492188Z","iopub.status.idle":"2022-07-19T17:41:39.667073Z","shell.execute_reply.started":"2022-07-19T17:41:39.492138Z","shell.execute_reply":"2022-07-19T17:41:39.665996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Linear Regression Model with an Added Column","metadata":{}},{"cell_type":"markdown","source":"As the Weekly Deaths and Next Week's Deeaths are heavily correlated, I will be introducing function to establish a new column for the difference between those columns, and delete these 2 columns. This will allow me to focus on the difference, and the effect of other columns.","metadata":{}},{"cell_type":"code","source":"#Importing the necesary model\nfrom sklearn.linear_model import LinearRegression","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.668897Z","iopub.execute_input":"2022-07-19T17:41:39.669853Z","iopub.status.idle":"2022-07-19T17:41:39.674720Z","shell.execute_reply.started":"2022-07-19T17:41:39.669811Z","shell.execute_reply":"2022-07-19T17:41:39.673495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preparing(used_df):\n    df = used_df\n    \n    if \"Next Week's Deaths\" in df.columns:\n        df[\"Difference\"] = df[\"Next Week's Deaths\"] - df[\"Weekly Deaths\"]\n        df.pop(\"Next Week's Deaths\")\n    \n    weekly_deaths = df[\"Weekly Deaths\"]\n    \n    df.pop(\"Weekly Deaths\")\n    \n    return df, weekly_deaths","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.675778Z","iopub.execute_input":"2022-07-19T17:41:39.676074Z","iopub.status.idle":"2022-07-19T17:41:39.686981Z","shell.execute_reply.started":"2022-07-19T17:41:39.676047Z","shell.execute_reply":"2022-07-19T17:41:39.686103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df, deaths_train = preparing(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.688572Z","iopub.execute_input":"2022-07-19T17:41:39.689199Z","iopub.status.idle":"2022-07-19T17:41:39.701448Z","shell.execute_reply.started":"2022-07-19T17:41:39.689163Z","shell.execute_reply":"2022-07-19T17:41:39.700120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You may convert \"Location d\"","metadata":{}},{"cell_type":"code","source":"df.pop(\"Location\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.715837Z","iopub.execute_input":"2022-07-19T17:41:39.716306Z","iopub.status.idle":"2022-07-19T17:41:39.726558Z","shell.execute_reply.started":"2022-07-19T17:41:39.716236Z","shell.execute_reply":"2022-07-19T17:41:39.725364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df.loc[:, df.columns != \"Difference\"]\ny_true = df[\"Difference\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.756832Z","iopub.execute_input":"2022-07-19T17:41:39.757462Z","iopub.status.idle":"2022-07-19T17:41:39.775541Z","shell.execute_reply.started":"2022-07-19T17:41:39.757423Z","shell.execute_reply":"2022-07-19T17:41:39.774735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LinearRegression()\nmodel.fit(X, y_true)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.838325Z","iopub.execute_input":"2022-07-19T17:41:39.838955Z","iopub.status.idle":"2022-07-19T17:41:39.914605Z","shell.execute_reply.started":"2022-07-19T17:41:39.838903Z","shell.execute_reply":"2022-07-19T17:41:39.913434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.916732Z","iopub.execute_input":"2022-07-19T17:41:39.917386Z","iopub.status.idle":"2022-07-19T17:41:39.946248Z","shell.execute_reply.started":"2022-07-19T17:41:39.917339Z","shell.execute_reply":"2022-07-19T17:41:39.944115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nimport math\n\nMSE = mean_squared_error(y_true, y_pred)\n \nRMSE = math.sqrt(MSE)\nprint(\"Root Mean Square Error:\\n\")\nprint(RMSE)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:39.981434Z","iopub.execute_input":"2022-07-19T17:41:39.981975Z","iopub.status.idle":"2022-07-19T17:41:39.992447Z","shell.execute_reply.started":"2022-07-19T17:41:39.981931Z","shell.execute_reply":"2022-07-19T17:41:39.991320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Predicting","metadata":{}},{"cell_type":"code","source":"# Preparing the test dataset for prediction\ndf_test, deaths = preparing(test)\ndf_test.pop(\"Location\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:40.055678Z","iopub.execute_input":"2022-07-19T17:41:40.056278Z","iopub.status.idle":"2022-07-19T17:41:40.068350Z","shell.execute_reply.started":"2022-07-19T17:41:40.056196Z","shell.execute_reply":"2022-07-19T17:41:40.067332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting the difference and adding it to this week's deaths\ndf_test[\"Next Week's Deaths\"] = model.predict(df_test) + deaths","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:40.090212Z","iopub.execute_input":"2022-07-19T17:41:40.091278Z","iopub.status.idle":"2022-07-19T17:41:40.108172Z","shell.execute_reply.started":"2022-07-19T17:41:40.091213Z","shell.execute_reply":"2022-07-19T17:41:40.106864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Changing the negative predictions to \"0\" as there can't be a negative number of deaths\ndf_test[\"Next Week's Deaths\"].values[df_test[\"Next Week's Deaths\"].values < 0] = 0","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:41:40.110209Z","iopub.execute_input":"2022-07-19T17:41:40.110915Z","iopub.status.idle":"2022-07-19T17:41:40.118200Z","shell.execute_reply.started":"2022-07-19T17:41:40.110868Z","shell.execute_reply":"2022-07-19T17:41:40.116797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Submission","metadata":{}},{"cell_type":"code","source":"# Creating a new dataframe to be submitted\nsubmission = df_test[[\"Id\", \"Next Week's Deaths\"]]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting the submission dataframe to csv to submit\nsubmission.to_csv(\"./submission.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]}]}