{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T07:58:26.871713Z","iopub.execute_input":"2022-07-13T07:58:26.873043Z","iopub.status.idle":"2022-07-13T07:58:26.905595Z","shell.execute_reply.started":"2022-07-13T07:58:26.872909Z","shell.execute_reply":"2022-07-13T07:58:26.904259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.read_csv('../input/tabular-playground-series-mar-2021/train.csv')\ndf_test=pd.read_csv('../input/tabular-playground-series-mar-2021/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:26.907545Z","iopub.execute_input":"2022-07-13T07:58:26.907857Z","iopub.status.idle":"2022-07-13T07:58:31.082166Z","shell.execute_reply.started":"2022-07-13T07:58:26.907831Z","shell.execute_reply":"2022-07-13T07:58:31.080491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()\ndf_test.info()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:31.084400Z","iopub.execute_input":"2022-07-13T07:58:31.084948Z","iopub.status.idle":"2022-07-13T07:58:31.551043Z","shell.execute_reply.started":"2022-07-13T07:58:31.084902Z","shell.execute_reply":"2022-07-13T07:58:31.549606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_id=df_test['id']\ndf_test=df_test.drop(['id'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:31.553963Z","iopub.execute_input":"2022-07-13T07:58:31.555134Z","iopub.status.idle":"2022-07-13T07:58:31.609164Z","shell.execute_reply.started":"2022-07-13T07:58:31.555080Z","shell.execute_reply":"2022-07-13T07:58:31.607809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"df_new=df.drop(['target','id'],axis=1)\ny=df['target']\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:31.610784Z","iopub.execute_input":"2022-07-13T07:58:31.611232Z","iopub.status.idle":"2022-07-13T07:58:31.688910Z","shell.execute_reply.started":"2022-07-13T07:58:31.611188Z","shell.execute_reply":"2022-07-13T07:58:31.687812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_new.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:31.690644Z","iopub.execute_input":"2022-07-13T07:58:31.691396Z","iopub.status.idle":"2022-07-13T07:58:31.970070Z","shell.execute_reply.started":"2022-07-13T07:58:31.691347Z","shell.execute_reply":"2022-07-13T07:58:31.969258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder=LabelEncoder()\nfor c in df_new.columns:\n    if df_new.dtypes[c]=='object':\n        label_encoder.fit(df_new[c])\n        df_new[c]=label_encoder.transform(df_new[c])\n    else:\n        continue\ndf_new.head()\ndf['target']","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:31.971829Z","iopub.execute_input":"2022-07-13T07:58:31.972566Z","iopub.status.idle":"2022-07-13T07:58:34.314370Z","shell.execute_reply.started":"2022-07-13T07:58:31.972519Z","shell.execute_reply":"2022-07-13T07:58:34.313137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder=LabelEncoder()\nfor c in df_test.columns:\n    if df_test.dtypes[c]=='object':\n        label_encoder.fit(df_test[c])\n        df_test[c]=label_encoder.transform(df_test[c])\n    else:\n        continue","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:34.315916Z","iopub.execute_input":"2022-07-13T07:58:34.316258Z","iopub.status.idle":"2022-07-13T07:58:35.227753Z","shell.execute_reply.started":"2022-07-13T07:58:34.316228Z","shell.execute_reply":"2022-07-13T07:58:35.225915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col=df_test.columns\ndata_t=df_test.loc[0:200000]\nfrom sklearn.preprocessing import StandardScaler\nscaler=StandardScaler()\nscaler.fit(data_t)\ndata_scaled=scaler.transform(data_t)\nx_test=pd.DataFrame(data=data_scaled, columns=col)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:35.229039Z","iopub.execute_input":"2022-07-13T07:58:35.229385Z","iopub.status.idle":"2022-07-13T07:58:35.393632Z","shell.execute_reply.started":"2022-07-13T07:58:35.229353Z","shell.execute_reply":"2022-07-13T07:58:35.392462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col=df_new.columns\ndata=df_new.loc[0:300000]\nfrom sklearn.preprocessing import StandardScaler\nscaler=StandardScaler()\nscaler.fit(data)\ndata_scaled=scaler.transform(data)\nx_train=pd.DataFrame(data=data_scaled, columns=col)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:35.396521Z","iopub.execute_input":"2022-07-13T07:58:35.396865Z","iopub.status.idle":"2022-07-13T07:58:35.632249Z","shell.execute_reply.started":"2022-07-13T07:58:35.396835Z","shell.execute_reply":"2022-07-13T07:58:35.630883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:35.633691Z","iopub.execute_input":"2022-07-13T07:58:35.634023Z","iopub.status.idle":"2022-07-13T07:58:35.690206Z","shell.execute_reply.started":"2022-07-13T07:58:35.633991Z","shell.execute_reply":"2022-07-13T07:58:35.688982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nmodel=LogisticRegression(max_iter=1000,C=0.1)\nmodel.fit(x_train, y)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:35.691752Z","iopub.execute_input":"2022-07-13T07:58:35.692180Z","iopub.status.idle":"2022-07-13T07:58:36.885987Z","shell.execute_reply.started":"2022-07-13T07:58:35.692137Z","shell.execute_reply":"2022-07-13T07:58:36.884692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_predicted=model.predict(x_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:36.888005Z","iopub.execute_input":"2022-07-13T07:58:36.888868Z","iopub.status.idle":"2022-07-13T07:58:36.915503Z","shell.execute_reply.started":"2022-07-13T07:58:36.888816Z","shell.execute_reply":"2022-07-13T07:58:36.914219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nprint(accuracy_score(y, train_predicted))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:36.917826Z","iopub.execute_input":"2022-07-13T07:58:36.918882Z","iopub.status.idle":"2022-07-13T07:58:36.980496Z","shell.execute_reply.started":"2022-07-13T07:58:36.918824Z","shell.execute_reply":"2022-07-13T07:58:36.979205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nconfusion_matrix(y, train_predicted)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:03:13.496547Z","iopub.execute_input":"2022-07-13T08:03:13.496986Z","iopub.status.idle":"2022-07-13T08:03:13.569956Z","shell.execute_reply.started":"2022-07-13T08:03:13.496952Z","shell.execute_reply":"2022-07-13T08:03:13.568612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nclf=RandomForestClassifier()\nclf.fit(x_train,y)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:23:24.357332Z","iopub.execute_input":"2022-07-13T08:23:24.359374Z","iopub.status.idle":"2022-07-13T08:26:05.974799Z","shell.execute_reply.started":"2022-07-13T08:23:24.359311Z","shell.execute_reply":"2022-07-13T08:26:05.973524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_rf=clf.predict(x_train)\nprint(accuracy_score(y, prediction_rf))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:29:43.126069Z","iopub.execute_input":"2022-07-13T08:29:43.126547Z","iopub.status.idle":"2022-07-13T08:29:56.202953Z","shell.execute_reply.started":"2022-07-13T08:29:43.126513Z","shell.execute_reply":"2022-07-13T08:29:56.201687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nprediction=model.predict(x_test)\nd={'Id': test_id, 'Predicted': prediction}\nsubmission=pd.DataFrame(d)\nsubmission.head()\nsubmission.to_csv('./submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T07:58:36.982636Z","iopub.execute_input":"2022-07-13T07:58:36.983620Z","iopub.status.idle":"2022-07-13T07:58:37.354461Z","shell.execute_reply.started":"2022-07-13T07:58:36.983565Z","shell.execute_reply":"2022-07-13T07:58:37.353535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_rf=clf.predict(x_test)\nd={'Id': test_id, 'Predicted': prediction_rf}\nsubmission=pd.DataFrame(d)\nsubmission.head()\nsubmission.to_csv('./submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T08:31:00.595075Z","iopub.execute_input":"2022-07-13T08:31:00.595513Z","iopub.status.idle":"2022-07-13T08:31:09.302625Z","shell.execute_reply.started":"2022-07-13T08:31:00.595474Z","shell.execute_reply":"2022-07-13T08:31:09.301280Z"},"trusted":true},"execution_count":null,"outputs":[]}]}