{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-04T02:53:42.858315Z","iopub.execute_input":"2022-08-04T02:53:42.859033Z","iopub.status.idle":"2022-08-04T02:53:42.870167Z","shell.execute_reply.started":"2022-08-04T02:53:42.858982Z","shell.execute_reply":"2022-08-04T02:53:42.868597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport missingno as msn\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T02:53:42.872606Z","iopub.execute_input":"2022-08-04T02:53:42.874045Z","iopub.status.idle":"2022-08-04T02:53:42.884940Z","shell.execute_reply.started":"2022-08-04T02:53:42.873962Z","shell.execute_reply":"2022-08-04T02:53:42.883604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_input_folder = '/kaggle/input/tabular-playground-series-aug-2022/'\n\ntrain_data = pd.read_csv(data_input_folder + \"train.csv\")\ntrain_data.head()\n\ntest_data = pd.read_csv(data_input_folder + \"test.csv\")\ntest_data.head()\n\nqualitative = []\nfeatures = test_data.columns\nfor col in features:\n    if test_data[col].dtype=='object':\n        print(\"==========\\n\",col,test_data[col].unique())\n        only_in_test_data = set(test_data[col].unique()) - set(train_data[col].unique())\n        if len(only_in_test_data)>0:\n            print(\"%s in only test data.\"%only_in_test_data)\n            print(\"train\",train_data[col].unique() )\n            print(\"test\",test_data[col].unique() )\n        qualitative.append(col)\n\n# product code  train ['A' 'B' 'C' 'D' 'E']\n#               test ['F' 'G' 'H' 'I']\n#  drop producto code","metadata":{"execution":{"iopub.status.busy":"2022-08-04T02:53:42.895111Z","iopub.execute_input":"2022-08-04T02:53:42.895896Z","iopub.status.idle":"2022-08-04T02:53:43.149505Z","shell.execute_reply.started":"2022-08-04T02:53:42.895841Z","shell.execute_reply":"2022-08-04T02:53:43.147932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"qualitative.remove('product_code')\ntrain_data.drop('product_code',axis=1,inplace=True)\ntest_data.drop('product_code',axis=1,inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T02:53:43.151872Z","iopub.execute_input":"2022-08-04T02:53:43.152623Z","iopub.status.idle":"2022-08-04T02:53:43.164212Z","shell.execute_reply.started":"2022-08-04T02:53:43.152575Z","shell.execute_reply":"2022-08-04T02:53:43.162903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode(train_df, test_df , feature):\n    ordering = pd.DataFrame()\n    ordering['val'] = train_df[feature].unique()\n    ordering.index = ordering.val\n    ordering['failuremean'] = train_df[[feature, 'failure']].groupby(feature).mean()['failure']\n    ordering = ordering.sort_values('failuremean')\n    ordering['ordering'] = range(1, ordering.shape[0]+1)\n    ordering = ordering['ordering'].to_dict()\n\n    print(feature,ordering)\n    for cat, o in ordering.items():\n        train_df.loc[train_df[feature] == cat, feature+'_E'] = o    \n        test_df.loc[test_df[feature] == cat, feature+'_E'] = o    \n\nprint(\"encoding=\")\nqual_encoded = []\nfor q in qualitative:  \n    encode(train_data, test_data,q)\n    qual_encoded.append(q+'_E')\nprint(\"attribute_1 material_7 in test data shoud replaced by material_5(median)\")\ntest_data['attribute_1_E'].fillna(2,inplace=True)\ntrain_data.drop(['attribute_0','attribute_1'],axis=1,inplace=True)\ntest_data.drop(['attribute_0','attribute_1'],axis=1,inplace=True)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T02:53:43.165774Z","iopub.execute_input":"2022-08-04T02:53:43.166145Z","iopub.status.idle":"2022-08-04T02:53:43.243084Z","shell.execute_reply.started":"2022-08-04T02:53:43.166111Z","shell.execute_reply":"2022-08-04T02:53:43.241690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corrmat = train_data.corr()\nf, ax = plt.subplots(figsize=(12, 9))\nsns.heatmap(corrmat, vmax=.8, square=True,center=0)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T02:53:43.246982Z","iopub.execute_input":"2022-08-04T02:53:43.247976Z","iopub.status.idle":"2022-08-04T02:53:44.029786Z","shell.execute_reply.started":"2022-08-04T02:53:43.247932Z","shell.execute_reply":"2022-08-04T02:53:44.028616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nmy_imputer = SimpleImputer()\ndata_with_imputed_values = my_imputer.fit_transform(train_data.drop(['id','failure'],axis=1))\ntest_with_imputed_values = my_imputer.transform(test_data.drop(['id'],axis=1))\n#data_with_imputed_values = my_imputer.fit_transform(train_data[['loading']])\n#test_with_imputed_values = my_imputer.transform(test_data[['loading']])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T02:53:44.031847Z","iopub.execute_input":"2022-08-04T02:53:44.032344Z","iopub.status.idle":"2022-08-04T02:53:44.071422Z","shell.execute_reply.started":"2022-08-04T02:53:44.032299Z","shell.execute_reply":"2022-08-04T02:53:44.070196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()\ntrain_scaled = scaler.fit_transform(data_with_imputed_values)\ntest_scaled = scaler.transform(test_with_imputed_values)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T02:53:44.073356Z","iopub.execute_input":"2022-08-04T02:53:44.073840Z","iopub.status.idle":"2022-08-04T02:53:44.086520Z","shell.execute_reply.started":"2022-08-04T02:53:44.073793Z","shell.execute_reply":"2022-08-04T02:53:44.085471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n#train_x,val_x, train_y, val_y = train_test_split(train_data.drop('failure',axis=1), train_data['failure'], random_state = 0)  \ntrain_x,val_x, train_y, val_y = train_test_split(train_scaled, train_data['failure'], random_state = 0)  \n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T02:53:44.087959Z","iopub.execute_input":"2022-08-04T02:53:44.088660Z","iopub.status.idle":"2022-08-04T02:53:44.104299Z","shell.execute_reply.started":"2022-08-04T02:53:44.088617Z","shell.execute_reply":"2022-08-04T02:53:44.103083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn import metrics\nfrom sklearn.linear_model import LogisticRegression\n\nprint('Running LogisticRegression\\n')\nlogreg = LogisticRegression(max_iter = 600)\n#logreg = LogisticRegression(penalty='elasticnet', l1_ratio=0.8, C=0.007, tol = 1e-2, solver='saga', max_iter=3000, random_state=5)\nscores = cross_val_score(logreg,train_x,train_y,scoring='neg_mean_squared_error',cv=5)\nlogreg_mse = round(abs(scores.mean()), 4)\nlogreg.fit(train_x, train_y)\n#y_pred = logreg.predict(val_x)\n#logreg_acc = round(metrics.accuracy_score(val_y, y_pred), 4)\ny_pred = logreg.predict_proba(val_x)[:, 1]\nlogreg_acc = round(metrics.roc_auc_score(val_y, y_pred), 4)\nprint(\"accuracy \",logreg_acc)\n#y_test = logreg.predict(test_scaled)\ny_test = logreg.predict_proba(test_scaled)[:, 1]\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T02:53:44.105824Z","iopub.execute_input":"2022-08-04T02:53:44.106827Z","iopub.status.idle":"2022-08-04T02:53:45.179034Z","shell.execute_reply.started":"2022-08-04T02:53:44.106786Z","shell.execute_reply":"2022-08-04T02:53:45.177777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Try liner regession","metadata":{}},{"cell_type":"code","source":"output = pd.DataFrame({'id': test_data['id'], 'failure': y_test})\n#output = pd.DataFrame({'PassengerId': pass_ID, 'Transported': pred_test})\noutput.to_csv('submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T02:53:45.180475Z","iopub.execute_input":"2022-08-04T02:53:45.181220Z","iopub.status.idle":"2022-08-04T02:53:45.283233Z","shell.execute_reply.started":"2022-08-04T02:53:45.181177Z","shell.execute_reply":"2022-08-04T02:53:45.281976Z"},"trusted":true},"execution_count":null,"outputs":[]}]}