{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T11:17:48.388169Z","iopub.execute_input":"2022-08-01T11:17:48.388547Z","iopub.status.idle":"2022-08-01T11:17:48.408434Z","shell.execute_reply.started":"2022-08-01T11:17:48.388464Z","shell.execute_reply":"2022-08-01T11:17:48.407286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/tabular-playground-series-aug-2022/train.csv\", index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:48.409945Z","iopub.execute_input":"2022-08-01T11:17:48.415141Z","iopub.status.idle":"2022-08-01T11:17:48.491840Z","shell.execute_reply.started":"2022-08-01T11:17:48.415086Z","shell.execute_reply":"2022-08-01T11:17:48.490706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:48.493354Z","iopub.execute_input":"2022-08-01T11:17:48.493672Z","iopub.status.idle":"2022-08-01T11:17:48.532120Z","shell.execute_reply.started":"2022-08-01T11:17:48.493639Z","shell.execute_reply":"2022-08-01T11:17:48.531357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:48.534591Z","iopub.execute_input":"2022-08-01T11:17:48.534864Z","iopub.status.idle":"2022-08-01T11:17:48.631250Z","shell.execute_reply.started":"2022-08-01T11:17:48.534836Z","shell.execute_reply":"2022-08-01T11:17:48.630468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df_train.failure\nX = df_train.drop(['failure'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:48.632584Z","iopub.execute_input":"2022-08-01T11:17:48.633036Z","iopub.status.idle":"2022-08-01T11:17:48.639967Z","shell.execute_reply.started":"2022-08-01T11:17:48.633004Z","shell.execute_reply":"2022-08-01T11:17:48.638955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_predict","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:48.641200Z","iopub.execute_input":"2022-08-01T11:17:48.641831Z","iopub.status.idle":"2022-08-01T11:17:49.144600Z","shell.execute_reply.started":"2022-08-01T11:17:48.641797Z","shell.execute_reply":"2022-08-01T11:17:49.143705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Manual inspection of the datasets indicates a lazy partition like this is accurate","metadata":{}},{"cell_type":"code","source":"categorical_cols = [cname for cname in X.columns if X[cname].dtype in [ \"object\", \"int\" ]]\nnumerical_cols = [cname for cname in X.columns if cname not in categorical_cols]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.145773Z","iopub.execute_input":"2022-08-01T11:17:49.146485Z","iopub.status.idle":"2022-08-01T11:17:49.153252Z","shell.execute_reply.started":"2022-08-01T11:17:49.146454Z","shell.execute_reply":"2022-08-01T11:17:49.152501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.154094Z","iopub.execute_input":"2022-08-01T11:17:49.154900Z","iopub.status.idle":"2022-08-01T11:17:49.164135Z","shell.execute_reply.started":"2022-08-01T11:17:49.154872Z","shell.execute_reply":"2022-08-01T11:17:49.163260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# numerical_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.165906Z","iopub.execute_input":"2022-08-01T11:17:49.166490Z","iopub.status.idle":"2022-08-01T11:17:49.174339Z","shell.execute_reply.started":"2022-08-01T11:17:49.166313Z","shell.execute_reply":"2022-08-01T11:17:49.173402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder, RobustScaler","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.176164Z","iopub.execute_input":"2022-08-01T11:17:49.176910Z","iopub.status.idle":"2022-08-01T11:17:49.208757Z","shell.execute_reply.started":"2022-08-01T11:17:49.176874Z","shell.execute_reply":"2022-08-01T11:17:49.207740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocessing for categorical data\nnumerical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')),\n    ('scaler', RobustScaler())\n])\n\n# Preprocessing for categorical data\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Bundle preprocessing for numerical and categorical data\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('cat', categorical_transformer, categorical_cols)\n    ])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.210002Z","iopub.execute_input":"2022-08-01T11:17:49.210268Z","iopub.status.idle":"2022-08-01T11:17:49.215967Z","shell.execute_reply.started":"2022-08-01T11:17:49.210239Z","shell.execute_reply":"2022-08-01T11:17:49.215280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegressionCV\nmodel = LogisticRegressionCV(random_state=0, solver='saga', penalty='elasticnet', l1_ratios=np.arange(0, 1, 11), max_iter=10000)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.217012Z","iopub.execute_input":"2022-08-01T11:17:49.217887Z","iopub.status.idle":"2022-08-01T11:17:49.232445Z","shell.execute_reply.started":"2022-08-01T11:17:49.217862Z","shell.execute_reply":"2022-08-01T11:17:49.231287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from xgboost import XGBClassifier\n\n# model = XGBClassifier(n_estimators=100, max_depth=1000)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.235652Z","iopub.execute_input":"2022-08-01T11:17:49.235892Z","iopub.status.idle":"2022-08-01T11:17:49.243304Z","shell.execute_reply.started":"2022-08-01T11:17:49.235868Z","shell.execute_reply":"2022-08-01T11:17:49.242692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_pipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('model', model)\n])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.244245Z","iopub.execute_input":"2022-08-01T11:17:49.244756Z","iopub.status.idle":"2022-08-01T11:17:49.254641Z","shell.execute_reply.started":"2022-08-01T11:17:49.244732Z","shell.execute_reply":"2022-08-01T11:17:49.253770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scores = -1 * cross_val_predict(my_pipeline, X, y, cv=5)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.256318Z","iopub.execute_input":"2022-08-01T11:17:49.256931Z","iopub.status.idle":"2022-08-01T11:17:49.264394Z","shell.execute_reply.started":"2022-08-01T11:17:49.256892Z","shell.execute_reply":"2022-08-01T11:17:49.263784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scores","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.265393Z","iopub.execute_input":"2022-08-01T11:17:49.266141Z","iopub.status.idle":"2022-08-01T11:17:49.274814Z","shell.execute_reply.started":"2022-08-01T11:17:49.266116Z","shell.execute_reply":"2022-08-01T11:17:49.274175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.275774Z","iopub.execute_input":"2022-08-01T11:17:49.276549Z","iopub.status.idle":"2022-08-01T11:17:49.286500Z","shell.execute_reply.started":"2022-08-01T11:17:49.276525Z","shell.execute_reply":"2022-08-01T11:17:49.285593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# accuracy_score(y, scores)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.287687Z","iopub.execute_input":"2022-08-01T11:17:49.288637Z","iopub.status.idle":"2022-08-01T11:17:49.296915Z","shell.execute_reply.started":"2022-08-01T11:17:49.288603Z","shell.execute_reply":"2022-08-01T11:17:49.296259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_pipeline.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:17:49.298353Z","iopub.execute_input":"2022-08-01T11:17:49.299155Z","iopub.status.idle":"2022-08-01T11:22:18.717096Z","shell.execute_reply.started":"2022-08-01T11:17:49.299118Z","shell.execute_reply":"2022-08-01T11:22:18.715909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"/kaggle/input/tabular-playground-series-aug-2022/test.csv\", index_col='id')\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:22:18.718384Z","iopub.execute_input":"2022-08-01T11:22:18.718654Z","iopub.status.idle":"2022-08-01T11:22:18.788414Z","shell.execute_reply.started":"2022-08-01T11:22:18.718629Z","shell.execute_reply":"2022-08-01T11:22:18.786993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_test = my_pipeline.predict_proba(df_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:22:18.789833Z","iopub.execute_input":"2022-08-01T11:22:18.790118Z","iopub.status.idle":"2022-08-01T11:22:18.901239Z","shell.execute_reply.started":"2022-08-01T11:22:18.790094Z","shell.execute_reply":"2022-08-01T11:22:18.900237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'id': df_test.index,\n                       'failure': preds_test})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:22:18.902508Z","iopub.execute_input":"2022-08-01T11:22:18.902802Z","iopub.status.idle":"2022-08-01T11:22:18.951033Z","shell.execute_reply.started":"2022-08-01T11:22:18.902771Z","shell.execute_reply":"2022-08-01T11:22:18.949959Z"},"trusted":true},"execution_count":null,"outputs":[]}]}