{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-07T02:08:53.740119Z","iopub.execute_input":"2022-08-07T02:08:53.740733Z","iopub.status.idle":"2022-08-07T02:08:53.776470Z","shell.execute_reply.started":"2022-08-07T02:08:53.740596Z","shell.execute_reply":"2022-08-07T02:08:53.775200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Optuna Example\n\nI have created a small sample using Optuna to find the best parameters.\n\nIf you have sufficient computer resources, I have made it possible to run the part of the parameter search on multiple Notebooks in parallel.\n\nIf anyone knows of a better approach, I'd be happy to get comments.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport warnings\nfrom colorama import Fore, Back, Style\nimport scipy.stats\n\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler, PolynomialFeatures\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import SimpleImputer, IterativeImputer, KNNImputer\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.linear_model import LogisticRegression, LinearRegression\nfrom sklearn.discriminant_analysis import QuadraticDiscriminantAnalysis, LinearDiscriminantAnalysis\nfrom sklearn.metrics import roc_auc_score, roc_curve\n\nimport optuna\nfrom functools import partial\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import cross_validate","metadata":{"execution":{"iopub.status.busy":"2022-08-07T02:09:01.005520Z","iopub.execute_input":"2022-08-07T02:09:01.005974Z","iopub.status.idle":"2022-08-07T02:09:02.503964Z","shell.execute_reply.started":"2022-08-07T02:09:01.005938Z","shell.execute_reply":"2022-08-07T02:09:02.503041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv', index_col='id')\ntest = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/test.csv', index_col='id')\nboth = pd.concat([train[test.columns], test])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T02:09:26.964415Z","iopub.execute_input":"2022-08-07T02:09:26.964950Z","iopub.status.idle":"2022-08-07T02:09:27.311397Z","shell.execute_reply.started":"2022-08-07T02:09:26.964902Z","shell.execute_reply":"2022-08-07T02:09:27.310218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thanks for good Notebook<br>\nRef: https://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense","metadata":{}},{"cell_type":"code","source":"%%time\ndef objective(trial):\n    \n    # Tuned params here\n    n_estimators     = trial.suggest_int('n_estimators', 500, 600)\n    max_depth        = trial.suggest_int('max_depth', 6, 9)\n    min_samples_leaf = trial.suggest_int('min_samples_leaf', 80, 100)\n    n_neighbors      = trial.suggest_int('n_neighbors', 2, 4)\n    clipval          = trial.suggest_int('clipval', 9, 11)\n\n    auc_list = []\n    test_pred_list = []\n    kf = GroupKFold(n_splits=5)\n    for fold, (idx_tr, idx_va) in enumerate(kf.split(train, train.failure, train.product_code)):\n        X_tr = train.iloc[idx_tr][test.columns]\n        X_va = train.iloc[idx_va][test.columns]\n        X_te = test.copy()\n        y_tr = train.iloc[idx_tr].failure\n        y_va = train.iloc[idx_va].failure\n\n        # We fill the missing values\n        features = [f for f in X_tr.columns if f == 'loading' or f.startswith('measurement')]\n        imputer = KNNImputer(n_neighbors=n_neighbors)\n        imputer.fit(X_tr[features])\n        for df in [X_tr, X_va, X_te]:\n            df[features] = imputer.transform(df[features])\n\n    for df in [X_tr, X_va, X_te]:\n        df['measurement_2_2'] = df['measurement_2'].clip(clipval, None)\n        df['a2xa3']   = df['attribute_2'] * df['attribute_3']\n    \n    # feature engineering ['a2xa3'] here\n    features2 = ['loading', 'attribute_3', 'measurement_2', 'measurement_2_2', 'measurement_4', 'measurement_17', 'a2xa3']\n    model = ExtraTreesClassifier(n_estimators=n_estimators, max_depth=max_depth, min_samples_leaf=min_samples_leaf, n_jobs=-1, random_state=1)\n    model.fit(X_tr[features2], y_tr)\n    \n    # We validate the model\n    y_va_pred = model.predict_proba(X_va[features2])[:,1]\n    score = roc_auc_score(y_va, y_va_pred)\n    auc_list.append(score)\n\n    test_pred_list.append(model.predict_proba(X_te[features2])[:,1])\n\n    return sum(auc_list) / len(auc_list)\n\n# use SQLite DB to run in parallel or resume from an interruption. Significant time is needed to test a wide range of values.\nstudy = optuna.create_study(study_name='tps202208_001', storage='sqlite:///./kaggle.db', direction='maximize', load_if_exists=True)\n\n# n_trials is set to 3 to make sure it can be executed. additional trials need for better params\nstudy.optimize(objective, n_trials=3)\n\nprint('params:%s' % (study.best_params))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T02:13:04.556676Z","iopub.execute_input":"2022-08-07T02:13:04.557140Z","iopub.status.idle":"2022-08-07T02:23:28.031227Z","shell.execute_reply.started":"2022-08-07T02:13:04.557106Z","shell.execute_reply":"2022-08-07T02:23:28.029237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# copy & paste params here\nbp = {'clipval': 9, 'max_depth': 7, 'min_samples_leaf': 93, 'n_estimators': 533, 'n_neighbors': 3}\n\nauc_list = []\ntest_pred_list = []\nkf = GroupKFold(n_splits=5)\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(train, train.failure, train.product_code)):\n    X_tr = train.iloc[idx_tr][test.columns]\n    X_va = train.iloc[idx_va][test.columns]\n    X_te = test.copy()\n    y_tr = train.iloc[idx_tr].failure\n    y_va = train.iloc[idx_va].failure\n\n    ohe_attributes = ['attribute_0', 'attribute_1']\n    ohe_output = ['ohe0_7', 'ohe1_6', 'ohe1_8']\n    ohe = OneHotEncoder(categories=[['material_5', 'material_7'],\n                                    ['material_5', 'material_6', 'material_8']],\n                        drop='first', sparse=False, handle_unknown='ignore')\n    ohe.fit(X_tr[ohe_attributes])\n    for df in [X_tr, X_va, X_te]:\n        with warnings.catch_warnings():\n            warnings.filterwarnings('ignore', category=UserWarning)\n            df[ohe_output] = ohe.transform(df[ohe_attributes])\n        df.drop(columns=ohe_attributes, inplace=True)\n\n    # We fill the missing values\n    features = [f for f in X_tr.columns if f == 'loading' or f.startswith('measurement')]\n    imputer = KNNImputer(n_neighbors=bp['n_neighbors'])\n    imputer.fit(X_tr[features])\n    for df in [X_tr, X_va, X_te]:\n        df[features] = imputer.transform(df[features])\n                \n    for df in [X_tr, X_va, X_te]:\n        df['measurement_2_2'] = df['measurement_2'].clip(bp['clipval'], None)\n        df['a2xa3']   = df['attribute_2'] * df['attribute_3']\n    \n    # We fit a model using only the most important features\n    features2 = ['loading', 'attribute_3', 'measurement_2', 'measurement_2_2', 'measurement_4', 'measurement_17', 'a2xa3']\n    model = ExtraTreesClassifier(n_estimators=bp['n_estimators'], max_depth=bp['max_depth'], min_samples_leaf=bp['min_samples_leaf'], n_jobs=-1, random_state=1)\n    model.fit(X_tr[features2], y_tr)\n    \n    # We validate the model\n    y_va_pred = model.predict_proba(X_va[features2])[:,1]\n    score = roc_auc_score(y_va, y_va_pred)\n    print(f\"Fold {fold}: auc = {score:.5f}\")\n    auc_list.append(score)\n\n    test_pred_list.append(model.predict_proba(X_te[features2])[:,1])\nprint(f\"{Fore.GREEN}{Style.BRIGHT}Average auc = {sum(auc_list) / len(auc_list):.5f}{Style.RESET_ALL}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T02:24:27.537600Z","iopub.execute_input":"2022-08-07T02:24:27.538140Z","iopub.status.idle":"2022-08-07T02:28:09.682675Z","shell.execute_reply.started":"2022-08-07T02:24:27.538100Z","shell.execute_reply":"2022-08-07T02:28:09.681252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'id': test.index,\n                           'failure': sum(test_pred_list)/len(test_pred_list)})\nsubmission.to_csv('submission000.csv', index=False)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-08-07T02:28:09.779540Z","iopub.execute_input":"2022-08-07T02:28:09.779998Z","iopub.status.idle":"2022-08-07T02:28:09.863400Z","shell.execute_reply.started":"2022-08-07T02:28:09.779952Z","shell.execute_reply":"2022-08-07T02:28:09.859102Z"},"trusted":true},"execution_count":null,"outputs":[]}]}