{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Summer Holidays Out of Office Notebook\n![Process Schema](https://raw.githubusercontent.com/jxtrbtk/kaggle/master/tabular-playground-series-jul-2022.png)\n\nThis is summer and I will be busy outside enjoying the sun and have less time to participate to Kaggle's competitions. So I've designed this notebook that will do the work for me !\n\nThis is not a very serious, efficient or showing best practices notebook, it was more designed for experimental fun.\n\nThe idea is to schedule the notebooks so it will run everyday. It will collect the data from previous submissions, using the Kaggle CLI and some results published in a dataset that is synced with this notebook. So the loop is closed. The notebook update the dataset and consumn it the next time, this is the way to keep data from one execution to another. \nAlso, I've used \"secrets\" to store securely my api key (\"api_user\" and \"ai_key\"). \n\nAt the end, we select the 2 best supposed models for submission, then 2 randomly choosen models for data exploration and lastly a submission based on some good models voting. The notebook will submitt them.\n\nEnjoy.","metadata":{}},{"cell_type":"markdown","source":"## Reference the libs we need","metadata":{}},{"cell_type":"code","source":"!pip install timeout-decorator","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:11.220022Z","iopub.execute_input":"2022-07-18T09:57:11.220884Z","iopub.status.idle":"2022-07-18T09:57:26.591178Z","shell.execute_reply.started":"2022-07-18T09:57:11.220774Z","shell.execute_reply":"2022-07-18T09:57:26.590096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport json\nimport optuna\nimport timeout_decorator\nimport datetime\n\nfrom sklearn.ensemble import IsolationForest\nfrom sklearn.neighbors import LocalOutlierFactor\nfrom sklearn.svm import OneClassSVM\nfrom sklearn.linear_model import SGDOneClassSVM \nfrom sklearn.covariance import EllipticEnvelope\n\n\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import MaxAbsScaler\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import Normalizer\nfrom sklearn.preprocessing import PowerTransformer\nfrom sklearn.preprocessing import QuantileTransformer\nfrom sklearn.preprocessing import RobustScaler\n\nfrom sklearn.preprocessing import OneHotEncoder\n\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom sklearn.mixture import GaussianMixture\nfrom sklearn.cluster import KMeans\n\nfrom sklearn.metrics import calinski_harabasz_score\nfrom sklearn.metrics import davies_bouldin_score\nfrom sklearn.metrics import silhouette_score\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.linear_model import ElasticNet\nfrom sklearn.svm import LinearSVR, l1_min_c\n\nfrom sklearn.decomposition import PCA,TruncatedSVD,NMF,FastICA,FactorAnalysis\n\nfrom scipy.special import logit, expit\n\nimport importlib\nimport logging as log","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-18T09:57:26.594422Z","iopub.execute_input":"2022-07-18T09:57:26.594913Z","iopub.status.idle":"2022-07-18T09:57:28.137813Z","shell.execute_reply.started":"2022-07-18T09:57:26.594864Z","shell.execute_reply":"2022-07-18T09:57:28.136707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import IsolationForest\nfrom sklearn.neighbors import LocalOutlierFactor\nfrom sklearn.svm import OneClassSVM\nfrom sklearn.linear_model import SGDOneClassSVM \nfrom sklearn.covariance import EllipticEnvelope\n\n# get competition data\ndirectory = os.path.join(\"..\",\"input\",\"tabular-playground-series-jul-2022\")\ndata_path = os.path.join(directory, \"data.csv\")\ndf_data = pd.read_csv(data_path)\nfeatures_cols = [c for c in df_data.columns if c != \"id\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:28.139228Z","iopub.execute_input":"2022-07-18T09:57:28.139811Z","iopub.status.idle":"2022-07-18T09:57:29.486831Z","shell.execute_reply.started":"2022-07-18T09:57:28.139775Z","shell.execute_reply":"2022-07-18T09:57:29.485731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log = importlib.reload(log)\nlog.basicConfig(format=\"[%(levelname)1.1s %(asctime)s] %(message)s\", \n                datefmt='%m-%d-%Y %H:%M:%S', \n                level=log.INFO)\nlog.info(\"start\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:29.489014Z","iopub.execute_input":"2022-07-18T09:57:29.489361Z","iopub.status.idle":"2022-07-18T09:57:29.498009Z","shell.execute_reply.started":"2022-07-18T09:57:29.489329Z","shell.execute_reply":"2022-07-18T09:57:29.496722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Functions to get scores of the previous submissions","metadata":{}},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\n\ndef get_user_and_key():\n    log.info(\"get api and user key\")\n    api_user = UserSecretsClient().get_secret(\"api_user\") \n    api_key = UserSecretsClient().get_secret(\"api_key\")\n    return api_user, api_key\n\ndef create_kaggle_json():\n    log.info(\"create kaggl.json file\")\n    \n    # get the api value and prepare the json\n    api_user, api_key = get_user_and_key()\n    api_token = {\"username\":api_user,\"key\":api_key}\n\n    # create dir and file\n    if not os.path.exists(\"/root/.kaggle\"):\n        os.mkdir(\"/root/.kaggle\")\n    with open(\"/root/.kaggle/kaggle.json\", 'w') as file:\n        json.dump(api_token, file)\n    \n    # protect file\n    os.system(\"chmod 600 ~/.kaggle/kaggle.json\")\n    \ndef extract_previous_score():\n    log.info(\"extract submission scores\")\n\n    # check submission and put results in a file\n    os.system(\"kaggle competitions submissions -c tabular-playground-series-jul-2022 > \\\"sub_log.txt\\\"\")\n\n    # read the file\n    with open(\"sub_log.txt\") as f:\n        lines = f.readlines()\n\n    # extract score data for auto submissions\n    prev_scores = {}\n    for line in lines[2:]:\n        if \" ID \" in line:\n            line_id = line.split(\"ID\")[-1][:12].strip()\n            score = line.split(\"complete\")[-1][:16].strip()\n            prev_scores[int(line_id)] =  float(score)\n    \n    os.remove(\"sub_log.txt\")\n    \n    return prev_scores\n\ndef extract_previous_votes():\n    log.info(\"extract submission votes\")\n\n    # check submission and put results in a file\n    os.system(\"kaggle competitions submissions -c tabular-playground-series-jul-2022 > \\\"sub_log.txt\\\"\")\n\n    # read the file\n    with open(\"sub_log.txt\") as f:\n        lines = f.readlines()\n\n    # extract score data for auto submissions\n    prev_votes = []\n    for line in lines[2:]:\n        if \"  vote \" in line:\n            models = line.split(\"  vote \")[-1].split(\"complete\")[0].strip()\n            prev_votes.append(models)\n    \n    os.remove(\"sub_log.txt\")\n    \n    return list(set(prev_votes))","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:29.499696Z","iopub.execute_input":"2022-07-18T09:57:29.500085Z","iopub.status.idle":"2022-07-18T09:57:29.518174Z","shell.execute_reply.started":"2022-07-18T09:57:29.500055Z","shell.execute_reply":"2022-07-18T09:57:29.516882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Here we will build a model that aims to predict submission score based on odels metrics and some clustering model parameters","metadata":{}},{"cell_type":"code","source":"# aic, bic calculation, not very sure of the source and if it's right\ndef _n_parameters(model):\n    \"\"\"Return the number of free parameters in the model.\"\"\"\n    _, n_features = model.means_.shape\n    if model.covariance_type == \"full\":\n        cov_params = model.n_components * n_features * (n_features + 1) / 2.0\n    elif model.covariance_type == \"diag\":\n        cov_params = model.n_components * n_features\n    elif model.covariance_type == \"tied\":\n        cov_params = n_features * (n_features + 1) / 2.0\n    elif model.covariance_type == \"spherical\":\n        cov_params = model.n_components\n    mean_params = n_features * model.n_components\n    return int(cov_params + mean_params + model.n_components - 1)\n\ndef bic(model, X):\n    \"\"\"Bayesian information criterion for the current model on the input X.\n\n    Parameters\n    ----------\n    X : array of shape (n_samples, n_dimensions)\n\n    Returns\n    -------\n    bic : float\n        The lower the better.\n    \"\"\"\n    return (-2 * model.score(X) * X.shape[0] +\n            _n_parameters(model) * np.log(X.shape[0]))\n\ndef aic(model, X):\n    \"\"\"Akaike information criterion for the current model on the input X.\n\n    Parameters\n    ----------\n    X : array of shape (n_samples, n_dimensions)\n\n    Returns\n    -------\n    aic : float\n        The lower the better.\n    \"\"\"\n    return -2 * model.score(X) * X.shape[0] + 2 * _n_parameters(model)\n\ndef update_scores(df_score):\n    log.info(\"update scores\")\n    create_kaggle_json()\n    prev_scores = extract_previous_score()\n    \n    df_score[\"score\"] = np.nan\n    for i in prev_scores.keys():\n        log.debug(f\"{i} {prev_scores[i]}\")\n        df_score.loc[i, \"score\"] = prev_scores[i]\n    \n    return df_score\n\ndef format_score(df_score):\n    log.info(\"format score file\")\n    cols = [\"outliersdetection\", \"fraction\", \"preprocessing\", \"decomposition\", \"n_components\", \"model\", \"n_clusters\", \"config\", \"silhouette\", \"aic\", \"bic\", \"calinski_harabasz\", \"davies_bouldin\", \"score\", \"prediction\"] \n    for col in [c for c in cols if c not in df_score.columns]:\n        df_score[col] = np.nan\n\n    return df_score[cols].copy()\n\ndef generate_eval_features(df, df_score):\n    log.info(\"generate evaluation indictors features\")\n    # standardize and sign\n    df[\"f_silhouette\"] =         (df[\"silhouette\"] - df_score[\"silhouette\"].mean()) / df_score[\"silhouette\"].std()\n    df[\"f_aic\"] =               -(df[\"aic\"] - df_score[\"aic\"].mean()) / df_score[\"aic\"].std()\n    df[\"f_bic\"] =               -(df[\"bic\"] - df_score[\"bic\"].mean()) / df_score[\"bic\"].std()\n    df[\"f_calinski_harabasz\"] =  (df[\"calinski_harabasz\"] - df_score[\"calinski_harabasz\"].mean()) / df_score[\"calinski_harabasz\"].std()\n    df[\"f_davies_bouldin\"] =    -(df[\"davies_bouldin\"] - df_score[\"davies_bouldin\"].mean()) / df_score[\"davies_bouldin\"].std()\n    \n    return df\n\nencoder_model = None\ndef encode_model(df_score):\n    global encoder_model\n    cols = [\"GaussianMixture\", \"BayesianGaussianMixture\"]\n    cols_names = [\"f_\"+ str(c) for c in cols]\n    for c in cols_names:\n        if c in df_score.columns:\n            df_score.drop(columns=c, inplace=True)    \n    \n    if encoder_model is None: \n        encoder_model = OneHotEncoder(handle_unknown='ignore', categories=[np.array(cols)])\n        encoder_model.fit(df_score[\"model\"].values.reshape(-1, 1))\n    \n    transformed = encoder_model.transform(df_score[\"model\"].values.reshape(-1, 1)).toarray()\n    ohe_df = pd.DataFrame(transformed, columns=cols_names)\n    ohe_df\n\n    df_score = pd.concat([df_score, ohe_df], axis=1)\n    \n    return df_score\n\nencoder_outliersdetection = None\ndef encode_outliersdetection(df_score):\n    global encoder_outliersdetection\n    cols = [\"None\", \"IsolationForest\", \"OneClassSVM\", \"SGDOneClassSVM\", \"EllipticEnvelope\"]\n    cols_names = [\"f_o_\"+ str(c) for c in cols]\n    for c in cols_names:\n        if c in df_score.columns:\n            df_score.drop(columns=c, inplace=True)    \n    \n    if encoder_outliersdetection is None: \n        encoder_outliersdetection = OneHotEncoder(handle_unknown='ignore', categories=[np.array(cols)])\n        encoder_outliersdetection.fit(df_score[\"outliersdetection\"].values.reshape(-1, 1))\n    \n    transformed = encoder_outliersdetection.transform(df_score[\"outliersdetection\"].values.reshape(-1, 1)).toarray()\n    ohe_df = pd.DataFrame(transformed, columns=cols_names)\n    ohe_df\n\n    df_score = pd.concat([df_score, ohe_df], axis=1)\n    \n    return df_score\n\nencoder_preprocessing = None\ndef encode_preprocessing(df_score):\n    global encoder_preprocessing\n    cols = [\"None\", \"StandardScaler\", \"MaxAbsScaler\", \"MinMaxScaler\", \"Normalizer\", \"Normalizer 0-1\",\n         \"PowerTransformer yeo-johnson\", \"PowerTransformer box-cox\", \n         \"QuantileTransformer uniform\", \"QuantileTransformer normal\",\n         \"RobustScaler 25-75\", \"RobustScaler 10-90\"]\n    cols_names = [\"f_p_\"+ str(c) for c in cols]\n    for c in cols_names:\n        if c in df_score.columns:\n            df_score.drop(columns=c, inplace=True)    \n    \n    if encoder_preprocessing is None: \n        encoder_preprocessing = OneHotEncoder(handle_unknown='ignore', categories=[np.array(cols)])\n        encoder_preprocessing.fit(df_score[\"preprocessing\"].values.reshape(-1, 1))\n    \n    transformed = encoder_preprocessing.transform(df_score[\"preprocessing\"].values.reshape(-1, 1)).toarray()\n    ohe_df = pd.DataFrame(transformed, columns=cols_names)\n    ohe_df\n\n    df_score = pd.concat([df_score, ohe_df], axis=1)\n    \n    return df_score\n\nencoder_decomposition = None\ndef encode_decomposition(df_score):\n    global encoder_decomposition\n    cols = [\"None\", \"PCA\", \"TruncatedSVD\", \"NMF\", \"FastICA\", \"FactorAnalysis\"]\n    cols_names = [\"f_d_\"+ str(c) for c in cols]\n    for c in cols_names:\n        if c in df_score.columns:\n            df_score.drop(columns=c, inplace=True)    \n    \n    if encoder_decomposition is None: \n        encoder_decomposition = OneHotEncoder(handle_unknown='ignore', categories=[np.array(cols)])\n        encoder_decomposition.fit(df_score[\"decomposition\"].values.reshape(-1, 1))\n    \n    transformed = encoder_decomposition.transform(df_score[\"decomposition\"].values.reshape(-1, 1)).toarray()\n    ohe_df = pd.DataFrame(transformed, columns=cols_names)\n    ohe_df\n\n    df_score = pd.concat([df_score, ohe_df], axis=1)\n    \n    return df_score\n\nencoder_n_clusters = None\ndef encode_n_clusters(df_score):\n    global encoder_n_clusters\n    cols = [i for i in range(4, 31)]\n    cols_names = [\"f_n_\"+ str(c) for c in cols]\n    for c in cols_names:\n        if c in df_score.columns:\n            df_score.drop(columns=c, inplace=True)    \n    \n    if encoder_n_clusters is None:\n        encoder_n_clusters = OneHotEncoder(handle_unknown='ignore', categories=[np.array(cols)])\n        encoder_n_clusters.fit(df_score[\"n_clusters\"].values.reshape(-1, 1))\n    \n    transformed = encoder_n_clusters.transform(df_score[\"n_clusters\"].values.reshape(-1, 1)).toarray()\n    ohe_df = pd.DataFrame(transformed, columns=cols_names)\n    ohe_df\n\n    df_score = pd.concat([df_score, ohe_df], axis=1)\n    \n    return df_score\n\nencoder_n_components = None\ndef encode_n_components(df_score):\n    global encoder_n_components\n    cols = [i for i in range(6, 11)]\n    cols_names = [\"f_c_\"+ str(c) for c in cols]\n    for c in cols_names:\n        if c in df_score.columns:\n            df_score.drop(columns=c, inplace=True)    \n    \n    if encoder_n_components is None:\n        encoder_n_components = OneHotEncoder(handle_unknown='ignore', categories=[np.array(cols)])\n        encoder_n_components.fit(df_score[\"n_components\"].values.reshape(-1, 1))\n    \n    transformed = encoder_n_components.transform(df_score[\"n_components\"].values.reshape(-1, 1)).toarray()\n    ohe_df = pd.DataFrame(transformed, columns=cols_names)\n    ohe_df\n\n    df_score = pd.concat([df_score, ohe_df], axis=1)\n    \n    return df_score\n\nencoder_fraction = None\ndef encode_fraction(df_score):\n    global encoder_fraction\n    cols = [i for i in range(0, 6)]\n    cols_names = [\"f_f_\"+ str(c) for c in cols]\n    for c in cols_names:\n        if c in df_score.columns:\n            df_score.drop(columns=c, inplace=True)    \n    \n    if encoder_fraction is None:\n        encoder_fraction = OneHotEncoder(handle_unknown='ignore', categories=[np.array(cols)])\n        encoder_fraction.fit(df_score[\"fraction\"].values.reshape(-1, 1))\n    \n    transformed = encoder_fraction.transform(df_score[\"fraction\"].values.reshape(-1, 1)).toarray()\n    ohe_df = pd.DataFrame(transformed, columns=cols_names)\n    ohe_df\n\n    df_score = pd.concat([df_score, ohe_df], axis=1)\n    \n    return df_score\n\nlinear_regression = None\ndef make_evaluation_function(df_score):\n    global linear_regression\n    \n    score_features = [c for c in df_score.columns if str(c).startswith(\"f_\")]\n    linear_regression = LinearSVR(max_iter=10_000_000, C=3)\n    # linear_regression = LinearRegression()\n    # linear_regression = ElasticNet(alpha=0.1, l1_ratio=0.25)\n    mask = ~df_score[\"score\"].isna()\n    # mask = mask & (df_score[\"score\"]>0)\n    y = df_score.loc[mask,[\"score\"]].values.ravel()\n    y[y<=0] = 1e-2\n    y = logit(y)\n    linear_regression.fit(df_score.loc[mask,score_features].values,y)\n    \n    y = linear_regression.predict(df_score[score_features].values)\n    df_score[\"prediction\"] = expit(y)\n\n    return df_score","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:29.520746Z","iopub.execute_input":"2022-07-18T09:57:29.521075Z","iopub.status.idle":"2022-07-18T09:57:29.578293Z","shell.execute_reply.started":"2022-07-18T09:57:29.521045Z","shell.execute_reply":"2022-07-18T09:57:29.576806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get last scores and trials\ndf_score = pd.read_csv(\"../input/tabular-jul-2022-scores/scores.csv\", index_col=0)\nif \"outliersdetection\" not in df_score.columns: df_score.insert(0, \"outliersdetection\", \"None\")\nif \"fraction\" not in df_score.columns: df_score.insert(1, \"fraction\", 0)\nmask = (df_score[\"decomposition\"] == \"None\")\ndf_score.loc[mask, \"n_components\"] = 10\ndf_score = update_scores(df_score)\ndf_score = format_score(df_score)\ndf_score = generate_eval_features(df_score,df_score)\ndf_score = encode_model(df_score)\ndf_score = encode_outliersdetection(df_score)\ndf_score = encode_preprocessing(df_score)\ndf_score = encode_decomposition(df_score)\ndf_score = encode_fraction(df_score)\ndf_score = encode_n_clusters(df_score)\ndf_score = encode_n_components(df_score)\ndf_score = make_evaluation_function(df_score)\ndf_score","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:29.5801Z","iopub.execute_input":"2022-07-18T09:57:29.580489Z","iopub.status.idle":"2022-07-18T09:57:30.7868Z","shell.execute_reply.started":"2022-07-18T09:57:29.580457Z","shell.execute_reply":"2022-07-18T09:57:30.785431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Some visualizations from the model built","metadata":{}},{"cell_type":"code","source":"mask = ~df_score[\"score\"].isna()\n_ = df_score.loc[mask,[\"score\", \"prediction\"]].plot.scatter(x=\"score\", y=\"prediction\", title=\"real score vs prediction\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:30.793185Z","iopub.execute_input":"2022-07-18T09:57:30.797556Z","iopub.status.idle":"2022-07-18T09:57:31.022712Z","shell.execute_reply.started":"2022-07-18T09:57:30.797481Z","shell.execute_reply":"2022-07-18T09:57:31.021499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = df_score[\"prediction\"].plot(kind=\"hist\", bins=100, title=\"scores distribution\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:31.024178Z","iopub.execute_input":"2022-07-18T09:57:31.025321Z","iopub.status.idle":"2022-07-18T09:57:31.326712Z","shell.execute_reply.started":"2022-07-18T09:57:31.025282Z","shell.execute_reply":"2022-07-18T09:57:31.325723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot feature importance for n_clusters\ncols = [i for i in range(4, 31)]\ncols_names = [\"f_n_\"+ str(c) for c in cols]\nscore_features = [c for c in df_score.columns if str(c).startswith(\"f_\")]\n\ndf_n_clusters = pd.DataFrame() \nfor (e,v) in zip(score_features, linear_regression.coef_ ):\n    if e in cols_names: \n        df_n_clusters = df_n_clusters.append({\"name\":e, \"value\":v}, ignore_index=True)\n\n_ = df_n_clusters.plot.bar(y=\"value\", x=\"name\", title=\"coefs per number of clusters\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:31.331074Z","iopub.execute_input":"2022-07-18T09:57:31.331696Z","iopub.status.idle":"2022-07-18T09:57:31.706839Z","shell.execute_reply.started":"2022-07-18T09:57:31.331632Z","shell.execute_reply":"2022-07-18T09:57:31.705685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = [i for i in range(4, 31)]\ncols_names = [\"f_n_\"+ str(c) for c in cols]\nscore_features = [c for c in df_score.columns if str(c).startswith(\"f_\")]\n\ndf_n_clusters = pd.DataFrame() \nfor (e,v) in zip(score_features, linear_regression.coef_ ):\n    if e in cols_names: \n        df_n_clusters = df_n_clusters.append({\"name\":e.replace(\"f_n_\", \"\"), \"value\":v}, ignore_index=True)\n\nbest_n_clusters = df_n_clusters.sort_values(by=\"value\", ascending=False).head(1)[\"name\"].values[0]\nbest_n_clusters = int(best_n_clusters)\n\nprint(\"Best number of clusters :\", best_n_clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:31.708845Z","iopub.execute_input":"2022-07-18T09:57:31.709655Z","iopub.status.idle":"2022-07-18T09:57:31.916592Z","shell.execute_reply.started":"2022-07-18T09:57:31.70959Z","shell.execute_reply":"2022-07-18T09:57:31.915178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_params = [\"f_silhouette\", \"f_aic\", \"f_bic\", \"f_calinski_harabasz\", \"f_davies_bouldin\"]\nscore_features = [c for c in df_score.columns if str(c).startswith(\"f_\")]\n\ndf_cols_params = pd.DataFrame() \nfor (e,v) in zip(score_features, linear_regression.coef_ ):\n    if e in cols_params: \n        df_cols_params = df_cols_params.append({\"name\":e, \"value\":v}, ignore_index=True)\n\n_ = df_cols_params.plot.bar(y=\"value\", x=\"name\", title=\"coefs per metric\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:31.918407Z","iopub.execute_input":"2022-07-18T09:57:31.919977Z","iopub.status.idle":"2022-07-18T09:57:32.148558Z","shell.execute_reply.started":"2022-07-18T09:57:31.919925Z","shell.execute_reply":"2022-07-18T09:57:32.147731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_params = [\"f_GaussianMixture\", \"f_BayesianGaussianMixture\"]\nscore_features = [c for c in df_score.columns if str(c).startswith(\"f_\")]\n\ndf_cols_params = pd.DataFrame() \nfor (e,v) in zip(score_features, linear_regression.coef_ ):\n    if e in cols_params: \n        df_cols_params = df_cols_params.append({\"name\":e, \"value\":v}, ignore_index=True)\n\n_ = df_cols_params.plot.bar(y=\"value\", x=\"name\", title=\"coefs per model type\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:32.150442Z","iopub.execute_input":"2022-07-18T09:57:32.151272Z","iopub.status.idle":"2022-07-18T09:57:32.412531Z","shell.execute_reply.started":"2022-07-18T09:57:32.151225Z","shell.execute_reply":"2022-07-18T09:57:32.411287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_params = [c for c in df_score.columns if str(c).startswith(\"f_o_\")]\nscore_features = [c for c in df_score.columns if str(c).startswith(\"f_\")]\n\ndf_cols_params = pd.DataFrame() \nfor (e,v) in zip(score_features, linear_regression.coef_ ):\n    if e in cols_params: \n        df_cols_params = df_cols_params.append({\"name\":e, \"value\":v}, ignore_index=True)\n\n_ = df_cols_params.plot.bar(y=\"value\", x=\"name\", title=\"coefs per outliers detection\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:32.413991Z","iopub.execute_input":"2022-07-18T09:57:32.414569Z","iopub.status.idle":"2022-07-18T09:57:32.653225Z","shell.execute_reply.started":"2022-07-18T09:57:32.414525Z","shell.execute_reply":"2022-07-18T09:57:32.652047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_params = [c for c in df_score.columns if str(c).startswith(\"f_p_\")]\nscore_features = [c for c in df_score.columns if str(c).startswith(\"f_\")]\n\ndf_cols_params = pd.DataFrame() \nfor (e,v) in zip(score_features, linear_regression.coef_ ):\n    if e in cols_params: \n        df_cols_params = df_cols_params.append({\"name\":e, \"value\":v}, ignore_index=True)\n\n_ = df_cols_params.plot.bar(y=\"value\", x=\"name\", title=\"coefs per pre-processing\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:32.654399Z","iopub.execute_input":"2022-07-18T09:57:32.655384Z","iopub.status.idle":"2022-07-18T09:57:32.959704Z","shell.execute_reply.started":"2022-07-18T09:57:32.655344Z","shell.execute_reply":"2022-07-18T09:57:32.958738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_params = [c for c in df_score.columns if str(c).startswith(\"f_d_\")]\nscore_features = [c for c in df_score.columns if str(c).startswith(\"f_\")]\n\ndf_cols_params = pd.DataFrame() \nfor (e,v) in zip(score_features, linear_regression.coef_ ):\n    if e in cols_params: \n        df_cols_params = df_cols_params.append({\"name\":e, \"value\":v}, ignore_index=True)\n\n_ = df_cols_params.plot.bar(y=\"value\", x=\"name\", title=\"coefs per decomposition\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:32.96104Z","iopub.execute_input":"2022-07-18T09:57:32.962131Z","iopub.status.idle":"2022-07-18T09:57:33.199432Z","shell.execute_reply.started":"2022-07-18T09:57:32.96208Z","shell.execute_reply":"2022-07-18T09:57:33.19803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot feature importance for n_clusters\ncols = [i for i in range(0, 6)]\ncols_names = [\"f_f_\"+ str(c) for c in cols]\nscore_features = [c for c in df_score.columns if str(c).startswith(\"f_\")]\n\ndf_n_comp = pd.DataFrame() \nfor (e,v) in zip(score_features, linear_regression.coef_ ):\n    if e in cols_names: \n        df_n_comp = df_n_comp.append({\"name\":e, \"value\":v}, ignore_index=True)\n\n_ = df_n_comp.plot.bar(y=\"value\", x=\"name\", title=\"coeff per outliers fraction\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:33.201104Z","iopub.execute_input":"2022-07-18T09:57:33.201799Z","iopub.status.idle":"2022-07-18T09:57:33.428934Z","shell.execute_reply.started":"2022-07-18T09:57:33.201754Z","shell.execute_reply":"2022-07-18T09:57:33.428069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot feature importance for n_clusters\ncols = [i for i in range(6, 11)]\ncols_names = [\"f_c_\"+ str(c) for c in cols]\nscore_features = [c for c in df_score.columns if str(c).startswith(\"f_\")]\n\ndf_n_comp = pd.DataFrame() \nfor (e,v) in zip(score_features, linear_regression.coef_ ):\n    if e in cols_names: \n        df_n_comp = df_n_comp.append({\"name\":e, \"value\":v}, ignore_index=True)\n\n_ = df_n_comp.plot.bar(y=\"value\", x=\"name\", title=\"coeff per decomposition ratio x10\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:57:33.430141Z","iopub.execute_input":"2022-07-18T09:57:33.430614Z","iopub.status.idle":"2022-07-18T09:57:33.631191Z","shell.execute_reply.started":"2022-07-18T09:57:33.430584Z","shell.execute_reply":"2022-07-18T09:57:33.630244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now we will use Optuna to search for the best parameters\nWe will load the data, reload scored submission and perform some searchs.","metadata":{}},{"cell_type":"code","source":"study_timeout = 60*60*5\ntrial_timeout = 60*43\n\n# study_timeout = 60*10\n# trial_timeout = 10","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:05:54.862571Z","iopub.execute_input":"2022-07-15T19:05:54.862933Z","iopub.status.idle":"2022-07-15T19:05:54.867193Z","shell.execute_reply.started":"2022-07-15T19:05:54.8629Z","shell.execute_reply":"2022-07-15T19:05:54.866455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = df_score.to_dict(\"records\")\nmax_score_index = df_score.index.max() \nmax_score_index","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:05:54.868126Z","iopub.execute_input":"2022-07-15T19:05:54.869017Z","iopub.status.idle":"2022-07-15T19:05:54.934157Z","shell.execute_reply.started":"2022-07-15T19:05:54.868981Z","shell.execute_reply":"2022-07-15T19:05:54.933068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get competition data\ndirectory = os.path.join(\"..\",\"input\",\"tabular-playground-series-jul-2022\")\ndata_path = os.path.join(directory, \"data.csv\")\ndf_data = pd.read_csv(data_path)\nfeatures_cols = [c for c in df_data.columns if c != \"id\"]\n# df_data = df_data.head(1_000).copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:05:54.935797Z","iopub.execute_input":"2022-07-15T19:05:54.936235Z","iopub.status.idle":"2022-07-15T19:05:55.693098Z","shell.execute_reply.started":"2022-07-15T19:05:54.936189Z","shell.execute_reply":"2022-07-15T19:05:55.691995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reload_trial(trial, df_score, idx):\n    r = df_score.iloc[idx]\n    config = r[\"config\"]\n    if type(config) == type({}):\n        config = config\n    else:\n        config = (config.replace(\"'\", \"\\\"\"))\n        config = json.loads(config.replace(\" None\", \" null\").replace(\" False\", \" false\").replace(\" True\", \" true\"))\n\n        choices = (\"None\", \"IsolationForest\", \"OneClassSVM\", \"SGDOneClassSVM\", \"EllipticEnvelope\")\n        trial.storage.set_trial_param(trial.number, \"outliersdetection\", choices.index(str(r[\"outliersdetection\"])), optuna.distributions.CategoricalDistribution(choices=choices))\n        choices = ('None', 'StandardScaler', 'MaxAbsScaler', 'MinMaxScaler', 'Normalizer', 'Normalizer 0-1', 'PowerTransformer yeo-johnson', 'PowerTransformer box-cox', 'QuantileTransformer uniform', 'QuantileTransformer normal', 'RobustScaler 25-75', 'RobustScaler 10-90')\n        trial.storage.set_trial_param(trial.number, \"preprocessing\", choices.index(str(r[\"preprocessing\"])), optuna.distributions.CategoricalDistribution(choices=choices))\n        choices = ('None', 'PCA', 'TruncatedSVD', 'NMF', 'FastICA', 'FactorAnalysis')\n        trial.storage.set_trial_param(trial.number, \"decomposition\", choices.index(str(r[\"decomposition\"])), optuna.distributions.CategoricalDistribution(choices=choices))\n        \n        choices = ('BayesianGaussianMixture', 'GaussianMixture')\n        trial.storage.set_trial_param(trial.number, \"model\", choices.index(str(r[\"model\"])), optuna.distributions.CategoricalDistribution(choices=choices))\n\n        choices = ('full', 'tied', 'diag', 'spherical')\n        trial.storage.set_trial_param(trial.number, \"cov_type\", choices.index(str(config[\"covariance_type\"])), optuna.distributions.CategoricalDistribution(choices=choices))\n        trial.storage.set_trial_param(trial.number, \"n_clusters\", int(r[\"n_clusters\"]), optuna.distributions.IntLogUniformDistribution(high=30, low=4, step=1))\n        trial.storage.set_trial_param(trial.number, \"n_components\", int(r[\"n_components\"]), optuna.distributions.IntUniformDistribution(high=10, low=6, step=1))\n        trial.storage.set_trial_param(trial.number, \"fraction\", int(r[\"fraction\"]), optuna.distributions.IntUniformDistribution(high=5, low=0, step=1))\n        trial.storage.set_trial_param(trial.number, \"tol\", float(config[\"tol\"]), optuna.distributions.LogUniformDistribution(high=0.1, low=1e-05))\n        trial.storage.set_trial_param(trial.number, \"reg_covar\", float(config[\"reg_covar\"]), optuna.distributions.LogUniformDistribution(high=0.0001, low=1e-08))\n        trial.storage.set_trial_param(trial.number, \"max_iter\", int(config[\"max_iter\"]), optuna.distributions.IntLogUniformDistribution(high=3000, low=100, step=1))\n        trial.storage.set_trial_param(trial.number, \"n_init\", int(config[\"n_init\"]), optuna.distributions.IntUniformDistribution(high=20, low=1, step=1))\n        choices = ('kmeans', 'random')\n        trial.storage.set_trial_param(trial.number, \"init_params\", choices.index(str(config[\"init_params\"])), optuna.distributions.CategoricalDistribution(choices=choices))\n        choices = ('dirichlet_process', 'dirichlet_distribution')\n        trial.storage.set_trial_param(trial.number, \"weight_concentration_prior_type\", choices.index(str(config[\"weight_concentration_prior_type\"]) if \"weight_concentration_prior_type\" in config.keys() else \"dirichlet_process\"), optuna.distributions.CategoricalDistribution(choices=choices))\n        \n    return trial\n\n\ndef process_data(x, outliersdetection_name, fraction, preprocessing_name, decomposition_name, n_components, clustering_obj):\n        log.info(f\"outliersdetection... {outliersdetection_name} {fraction}\")\n        \n        inliers_mask = np.full((x.shape[0],), True, dtype=bool)\n        n_cols = x.shape[0]\n        n_features = x.shape[1]\n\n        detection = None\n        if outliersdetection_name == \"None\": \n            detection = None\n        elif outliersdetection_name == \"IsolationForest\": \n            detection = IsolationForest(contamination=(fraction/10))\n            inliers_mask = (detection.fit_predict(x) == 1)\n        elif outliersdetection_name == \"OneClassSVM\": \n            detection = OneClassSVM(nu=(fraction/10))\n            inliers_mask = (detection.fit_predict(x) == 1)\n        elif outliersdetection_name == \"SGDOneClassSVM\": \n            detection = SGDOneClassSVM(nu=(fraction/10))\n            inliers_mask = (detection.fit_predict(x) == 1)\n        elif outliersdetection_name == \"EllipticEnvelope\": \n            detection = EllipticEnvelope(contamination=(fraction/10))\n            inliers_mask = (detection.fit_predict(x) == 1)\n        log.info(np.sum(inliers_mask))\n        \n        log.info(f\"preprocessing... {preprocessing_name}\")\n        scaler = None\n        if preprocessing_name == \"None\": \n            scaler = None\n            x = x\n        elif preprocessing_name == \"StandardScaler\": \n            scaler = StandardScaler()\n            scaler = scaler.fit(x[inliers_mask])\n            x = scaler.transform(x)\n        elif preprocessing_name == \"MaxAbsScaler\": \n            scaler = MaxAbsScaler()\n            scaler = scaler.fit(x[inliers_mask])\n            x = scaler.transform(x)\n        elif preprocessing_name == \"MinMaxScaler\": \n            scaler = MinMaxScaler()\n            scaler = scaler.fit(x[inliers_mask])\n            x = scaler.transform(x)\n        elif preprocessing_name == \"Normalizer\": \n            scaler = Normalizer(norm=\"max\")\n            scaler = scaler.fit(x[inliers_mask])\n            x = scaler.transform(x)\n        elif preprocessing_name == \"Normalizer 0-1\": \n            scaler = Normalizer(norm=\"max\")\n            scaler = scaler.fit(x)\n            x = (scaler.fit_transform(x) + 1.0) / 2 \n        elif preprocessing_name == \"PowerTransformer yeo-johnson\": \n            scaler = PowerTransformer(method=\"yeo-johnson\")\n            scaler = scaler.fit(x[inliers_mask])\n            x = scaler.transform(x)\n        elif preprocessing_name == \"PowerTransformer box-cox\": \n            scaler = Normalizer(norm=\"max\")\n            scaler = scaler.fit(x)\n            x = (scaler.fit_transform(x) + 1.0) / 2 \n            scaler = PowerTransformer(method=\"box-cox\")\n            scaler = scaler.fit(x[inliers_mask])\n            x = scaler.transform(x)\n        elif preprocessing_name == \"QuantileTransformer uniform\": \n            scaler = QuantileTransformer(output_distribution= \"uniform\")\n            x = scaler.fit_transform(x)\n        elif preprocessing_name == \"QuantileTransformer normal\": \n            scaler = QuantileTransformer(output_distribution= \"normal\")\n            scaler = scaler.fit(x[inliers_mask])\n            x = scaler.transform(x)\n        elif preprocessing_name == \"RobustScaler 25-75\": \n            scaler = RobustScaler(quantile_range=(25.0, 75.0))\n            scaler = scaler.fit(x[inliers_mask])\n            x = scaler.transform(x)\n        elif preprocessing_name == \"RobustScaler 10-90\": \n            scaler = RobustScaler(quantile_range=(10.0, 90.0))\n            scaler = scaler.fit(x[inliers_mask])\n            x = scaler.transform(x)\n\n        n_components_value = int(n_features * n_components / 10)\n        log.info(f\"decomposition... {decomposition_name} {n_components_value}\")\n        decomposition = None\n        if decomposition_name == \"None\":\n            n_components = len(cols)\n            decomposition = None\n            x = x\n        elif decomposition_name == \"PCA\": \n            decomposition = PCA(n_components=n_components_value)\n            decomposition = decomposition.fit(x[inliers_mask])\n            x = decomposition.transform(x)\n        elif decomposition_name == \"TruncatedSVD\": \n            decomposition = TruncatedSVD(n_components=n_components_value)\n            decomposition = decomposition.fit(x[inliers_mask])\n            x = decomposition.transform(x)\n        elif decomposition_name == \"NMF\": \n            decomposition = NMF(n_components=n_components_value)\n            decomposition = decomposition.fit(x[inliers_mask])\n            x = decomposition.transform(x)\n        elif decomposition_name == \"FastICA\": \n            decomposition = FastICA(n_components=n_components_value, max_iter=10_000)\n            decomposition = decomposition.fit(x[inliers_mask])\n            x = decomposition.transform(x)\n        elif decomposition_name == \"FactorAnalysis\": \n            decomposition = FactorAnalysis(n_components=n_components_value, max_iter=10_000)\n            decomposition = decomposition.fit(x[inliers_mask])\n            x = decomposition.transform(x)\n            \n        log.info(\"fit...\")\n        clustering_obj.fit(x[inliers_mask])\n\n        log.info(\"score...\")\n        labels = clustering_obj.predict(x)\n        \n        return x, labels\n            ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:05:55.694935Z","iopub.execute_input":"2022-07-15T19:05:55.695397Z","iopub.status.idle":"2022-07-15T19:05:55.731468Z","shell.execute_reply.started":"2022-07-15T19:05:55.695349Z","shell.execute_reply":"2022-07-15T19:05:55.730558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sorry, this is not really clean code below","metadata":{}},{"cell_type":"code","source":"@timeout_decorator.timeout(trial_timeout, timeout_exception=optuna.TrialPruned, use_signals=True)\ndef objective(trial):\n       \n    global scores\n    global df_score\n    global df_data\n    global linear_regression\n    global encode_n_clusters\n    global encode_model\n    \n    cols = [c for c in df_data.columns if c != \"id\"]\n    x = df_data[cols].values\n\n    pred = 0.0\n    mask = ~df_score[\"score\"].isna()\n    # reload previous trials\n    if trial.number < df_score[mask].shape[0]:\n        idx = df_score.loc[mask].head(trial.number+1).tail(1).index.values[0] \n        trial = reload_trial(trial, df_score, idx)\n        pred = df_score.iloc[idx][\"score\"].max()\n    else: \n        outliersdetection_name = trial.suggest_categorical(\"outliersdetection\", \n            [\"None\", \"IsolationForest\", \"OneClassSVM\", \"SGDOneClassSVM\", \"EllipticEnvelope\"])     \n        \n        preprocessing_name = trial.suggest_categorical(\"preprocessing\", \n            [\"None\", \"StandardScaler\", \"MaxAbsScaler\", \"MinMaxScaler\", \n             \"Normalizer\", \"Normalizer 0-1\",\n             \"PowerTransformer yeo-johnson\", \"PowerTransformer box-cox\", \n             \"QuantileTransformer uniform\", \"QuantileTransformer normal\",\n             \"RobustScaler 25-75\", \"RobustScaler 10-90\"])\n        decomposition_name = trial.suggest_categorical(\"decomposition\", \n            [\"None\", \"PCA\", \"TruncatedSVD\", \"NMF\", \"FastICA\", \"FactorAnalysis\"])\n        n_components = trial.suggest_int(\"n_components\", 6, 10, log=False)\n        if decomposition_name == \"None\": n_components = 10\n        if decomposition_name == \"NMF\": preprocessing_name = \"Normalizer 0-1\"\n        fraction = trial.suggest_int(\"fraction\", 0, 5, log=False)\n        if fraction == 0: outliersdetection_name = \"None\"\n        \n        clustering_name = trial.suggest_categorical(\"model\", [\"BayesianGaussianMixture\", \"GaussianMixture\"])\n        cov_type = trial.suggest_categorical(\"cov_type\", [\"full\", \"tied\", \"diag\", \"spherical\"])\n        n_clusters = trial.suggest_int(\"n_clusters\", 4, 30, log=True)\n        tol = trial.suggest_float(\"tol\", 1e-5, 1e-1, log=True)\n        reg_covar = trial.suggest_float(\"reg_covar\", 1e-8, 1e-4, log=True)\n        max_iter = trial.suggest_int(\"max_iter\", 100, 3000, log=True)\n        n_init = trial.suggest_int(\"n_init\", 1, 20, log=False)\n        init_params = trial.suggest_categorical(\"init_params\", [\"kmeans\" , \"random\"])\n        weight_concentration_prior_type = trial.suggest_categorical(\n            \"weight_concentration_prior_type\", [\"dirichlet_process\", \"dirichlet_distribution\"])\n        log.info(f\"{clustering_name}... {n_clusters}\")\n        if clustering_name == \"GaussianMixture\":\n            clustering_obj = GaussianMixture(\n                n_components=n_clusters,\n                covariance_type=cov_type,\n                tol=tol,\n                reg_covar=reg_covar,\n                max_iter=max_iter,\n                init_params=init_params,\n                n_init=n_init,\n            )\n        else:\n            clustering_obj = BayesianGaussianMixture(\n                n_components=n_clusters,\n                covariance_type=cov_type,\n                tol=tol,\n                reg_covar=reg_covar,\n                max_iter=max_iter,\n                init_params=init_params,\n                n_init=n_init,\n            )\n\n        x, labels = process_data(x, outliersdetection_name, fraction, preprocessing_name, decomposition_name, n_components, clustering_obj)\n        \n        silhouette = silhouette_score(x, labels, metric='euclidean')\n        aic_score = aic(clustering_obj, x)\n        bic_score = bic(clustering_obj, x)\n        calinski_harabasz = calinski_harabasz_score(x, labels)\n        davies_bouldin = davies_bouldin_score(x, labels)\n\n        result = {\n            \"outliersdetection\": outliersdetection_name,\n            \"fraction\": fraction,\n            \"preprocessing\": preprocessing_name,\n            \"decomposition\": decomposition_name,\n            \"n_components\": n_components,\n            \"model\": clustering_name, \n            \"config\":clustering_obj.get_params(),\n            \"n_clusters\":n_clusters, \n            \"silhouette\":silhouette, \n            \"aic\": aic_score,\"bic\":bic_score, \n            \"calinski_harabasz\": calinski_harabasz, \n            \"davies_bouldin\": davies_bouldin\n        }\n\n        log.info(\"local score prediction...\")\n        df_local = pd.DataFrame.from_dict([result])\n        df_local = format_score(df_local)\n        df_local = generate_eval_features(df_local, df_score)    \n        df_local = encode_model(df_local)\n        df_local = encode_outliersdetection(df_local)\n        df_local = encode_preprocessing(df_local)\n        df_local = encode_decomposition(df_local)\n        df_local = encode_fraction(df_local)\n        df_local = encode_n_clusters(df_local)\n        df_local = encode_n_components(df_local)\n        pred = linear_regression.predict(df_local[score_features].values)[0]\n        pred = expit(pred)\n        result[\"prediction\"] = pred \n\n        scores.append(result)\n        log.info(result)\n\n        df_score = pd.DataFrame.from_dict(scores)\n        df_score.to_csv(\"scores.csv\")\n    \n        nb_records = df_score.shape[0]\n        log.info(f\"nb records:{nb_records}\")\n    \n    \n    return pred\n\nstudy = optuna.create_study(direction=\"maximize\")\nstudy.optimize(objective, catch=(Exception,), timeout=study_timeout)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:05:55.733789Z","iopub.execute_input":"2022-07-15T19:05:55.734314Z","iopub.status.idle":"2022-07-15T19:21:02.210165Z","shell.execute_reply.started":"2022-07-15T19:05:55.734282Z","shell.execute_reply":"2022-07-15T19:21:02.209094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(study.best_trial)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:21:02.212495Z","iopub.execute_input":"2022-07-15T19:21:02.213004Z","iopub.status.idle":"2022-07-15T19:21:02.219099Z","shell.execute_reply.started":"2022-07-15T19:21:02.212958Z","shell.execute_reply":"2022-07-15T19:21:02.218074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Time to record the trials dataset","metadata":{}},{"cell_type":"code","source":"df_score = pd.DataFrame.from_dict(scores)\ndf_score = format_score(df_score)\ndf_score.to_csv(\"scores.csv\")\ndf_score","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:21:02.220909Z","iopub.execute_input":"2022-07-15T19:21:02.221313Z","iopub.status.idle":"2022-07-15T19:21:02.305145Z","shell.execute_reply.started":"2022-07-15T19:21:02.221273Z","shell.execute_reply":"2022-07-15T19:21:02.304186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## And now we can submit","metadata":{}},{"cell_type":"code","source":"def fit_predict_model_i(x, i):\n    \n    log.info(i)\n    index = i\n    log.info(f\"index {index}\")\n    \n    score = df_score.loc[i, \"prediction\"]\n    log.info(f\"prediction {score}\")\n    \n    outliersdetection_name = df_score.loc[i, \"outliersdetection\"]\n    log.info(f\"outliersdetection_name {outliersdetection_name}\")\n\n    fraction = df_score.loc[i, \"fraction\"]\n    log.info(f\"fraction {fraction}\")\n    \n    preprocessing_name = df_score.loc[i, \"preprocessing\"]\n    log.info(f\"preprocessing_name {preprocessing_name}\")\n             \n    decomposition_name = df_score.loc[i, \"decomposition\"]\n    log.info(f\"decomposition_name {decomposition_name}\")\n\n    n_components = df_score.loc[i, \"n_components\"]\n    log.info(f\"n_components {n_components}\")\n\n    model_name = df_score.loc[i, \"model\"]\n    log.info(f\"model_name {model_name}\")\n    \n    n_clusters = df_score.loc[i, \"n_clusters\"]\n    log.info(f\"n_clusters {n_clusters}\")\n\n    log.info(type(df_score.loc[i, \"config\"]))\n    if type(df_score.loc[i, \"config\"]) == type({}):\n        params = df_score.loc[i, \"config\"]\n    else:\n        params = (df_score.loc[i, \"config\"].replace(\"'\", \"\\\"\"))\n        params = json.loads(params.replace(\" None\", \" null\").replace(\" False\", \" false\").replace(\" True\", \" true\"))\n    log.info(params)\n    \n    model = BayesianGaussianMixture()\n    if model_name == \"GaussianMixture\": model = GaussianMixture(**params)\n    if model_name == \"BayesianGaussianMixture\": model = BayesianGaussianMixture(**params)\n\n    cols = [c for c in df_data.columns if c != \"id\"]\n    x = df_data[cols].values\n    \n    x, labels = process_data(x, outliersdetection_name, fraction, preprocessing_name, decomposition_name, n_components, model)\n\n    return labels    ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:21:02.306597Z","iopub.execute_input":"2022-07-15T19:21:02.306905Z","iopub.status.idle":"2022-07-15T19:21:02.319866Z","shell.execute_reply.started":"2022-07-15T19:21:02.306879Z","shell.execute_reply":"2022-07-15T19:21:02.318961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## First, let's use voting with our best max 5 models","metadata":{}},{"cell_type":"code","source":"# # keep already scored configs\n# mask = ~df_score[\"score\"].isna()\n# mask = mask & (~df_score[\"model\"].isna())\n# mask = mask & (~df_score[\"config\"].isna())\n# # more than 1 prediction\n# df = df_score[mask].groupby(by=\"n_clusters\").filter(lambda x: len(x)>1)[[\"n_clusters\",\"score\"]]\n# # best mean score\n# n_clusters = df.groupby(by=\"n_clusters\").mean().sort_values(by=\"score\").tail(1).index.max()\nlog.info(f\"best n_clusters {best_n_clusters}\")\n\nmask = ~df_score[\"score\"].isna()\nmask = mask & (~df_score[\"model\"].isna())\nmask = mask & (~df_score[\"config\"].isna())\nmask = mask & (df_score[\"n_clusters\"] == best_n_clusters)\n\nids = sorted(list(df_score[mask].head(5).index))\nlog.info(ids)\n\nvote = True\nprevious_votes = extract_previous_votes()\nif '-'.join([str(i) for i in ids]) in previous_votes:\n    vote = False #already submitted this vote\n    log.info(\"no vote this time (vote already submitted)\")\nelse:\n    log.info(\"vote OK\")\n    for n,r in df_score.loc[ids, [\"score\"]].iterrows():\n        log.info(f\"{r.name} {r.score}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T08:37:51.7617Z","iopub.execute_input":"2022-07-17T08:37:51.762108Z","iopub.status.idle":"2022-07-17T08:37:53.029767Z","shell.execute_reply.started":"2022-07-17T08:37:51.762076Z","shell.execute_reply":"2022-07-17T08:37:53.02874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def init_df(data_normalized, n_clusters):\n    log.info(\"init_df\")\n    clustering = KMeans(n_clusters=n_clusters, random_state=42)\n    clustering.fit(data_normalized)\n    preds = clustering.predict(data_normalized)\n    \n    df_solution = pd.DataFrame()\n    df_solution[\"id\"] = df_data[\"id\"]\n    df_solution[\"kmeans\"] = preds\n    df_solution[\"cluster\"] = preds\n    for c in range(n_clusters):\n        df_solution[f\"cluster_{c}\"] = 0\n    for c in range(n_clusters):\n        mask = (df_solution[\"kmeans\"]==c)\n        df_solution.loc[mask, f\"cluster_{c}\"] += .1\n    df_solution.set_index(\"id\", inplace=True)\n    \n    return df_solution\n\ndef predict(data, df_solution, n_clusters, i):\n    \n    log.info(\"predict\")\n    cols = [c for c in data.columns if c != \"id\"]\n    x = data[cols].values\n    preds = fit_predict_model_i(x, i)    \n    df_solution[\"predicted\"] = preds\n    return df_solution\n\ndef translate(df_solution, n_clusters):\n\n    log.info(\"translate\")\n    translation = {}\n\n    for c in range(n_clusters):\n        mask = (df_solution[\"cluster\"]==c)\n        mask = mask & (~df_solution[\"predicted\"].isna())\n        if mask.sum() > 0:\n            translated = df_solution.loc[mask, \"predicted\"].mode().values[0]\n            translation[translated] = c\n\n    df_solution[\"translated\"] = df_solution[\"predicted\"].map(translation)\n\n    return df_solution\n\ndef cluster_count(df_solution, n_clusters):\n    log.info(\"count\")\n    for c in range(n_clusters):\n        mask = (df_solution[\"translated\"]==c)\n        df_solution.loc[mask, f\"cluster_{c}\"] += 1\n    \n    return df_solution\n\ndef cluster_max(df_solution, n_clusters):\n    log.info(\"max\")\n    cols = [f\"cluster_{c}\" for c in range(n_clusters)]\n    mask = (~df_solution[\"predicted\"].isna())\n\n    df_solution.loc[mask, \"max\"] = df_solution.loc[mask, cols].max(axis=1)\n    df_solution.loc[mask, \"max_count\"] = df_solution.loc[mask].apply(\n        lambda x: sum([1 for c in cols if x[c] == x[\"max\"]]), \n        axis=1)\n    \n    return df_solution\n\ndef update_clusters(df_solution, n_clusters):\n    log.info(\"update\")\n    mask = (df_solution[\"max_count\"] == 1) # doubt\n    df_solution.loc[mask]\n\n    df_solution.loc[mask, \"cluster\"] = df_solution.loc[mask].apply(\n        lambda x: [c for c in range(n_clusters) if x[f\"cluster_{c}\"] == x[\"max\"]][0], \n        axis=1)\n\n    return df_solution\n\ndef evaluate(df_solution, threshold=3):\n    log.info(\"evaluate\")\n    mask = (df_solution[\"max\"] <= threshold)\n    mask = mask | (df_solution[\"max_count\"] != 1)\n    mask = mask | (df_solution[\"max\"].isna())\n    \n    return mask.sum()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:21:03.015935Z","iopub.execute_input":"2022-07-15T19:21:03.016653Z","iopub.status.idle":"2022-07-15T19:21:03.034111Z","shell.execute_reply.started":"2022-07-15T19:21:03.016599Z","shell.execute_reply":"2022-07-15T19:21:03.03308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if vote:\n    transformer = StandardScaler()\n    cols = [c for c in df_data.columns if c != \"id\"]\n    transformer.fit(df_data[cols])\n    data_normalized = transformer.transform(df_data[cols])\n    df_solution = init_df(data_normalized, best_n_clusters)\n    \n    for i in ids:\n        df_solution = predict(df_data, df_solution, best_n_clusters, i)\n        df_solution = translate(df_solution, best_n_clusters)\n        df_solution = cluster_count(df_solution, best_n_clusters)\n        df_solution = cluster_max(df_solution, best_n_clusters)\n        df_solution = update_clusters(df_solution, best_n_clusters)\n\n        score = evaluate(df_solution, threshold=min([len(ids)-0.9, 3]))\n        log.info(f\"score {score}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:21:03.039211Z","iopub.execute_input":"2022-07-15T19:21:03.040015Z","iopub.status.idle":"2022-07-15T19:21:21.050175Z","shell.execute_reply.started":"2022-07-15T19:21:03.039982Z","shell.execute_reply":"2022-07-15T19:21:21.048844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if vote:\n    log.info(\"submitting voted files\")\n    file_name = \"submission.csv\"\n    df_solution[\"Predicted\"] = df_solution[\"cluster\"]\n    df_solution[\"Predicted\"].to_csv(file_name)\n\n    message = f\"vote {'-'.join([str(i) for i in ids])}\"\n\n    os.system(f\"kaggle competitions submit -c tabular-playground-series-jul-2022 -f {file_name} -m \\\"{message}\\\"\")\n    os.remove(file_name)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:21:21.05154Z","iopub.status.idle":"2022-07-15T19:21:21.051984Z","shell.execute_reply.started":"2022-07-15T19:21:21.051787Z","shell.execute_reply":"2022-07-15T19:21:21.051807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let's pick some configs and submit to the competition","metadata":{}},{"cell_type":"markdown","source":"We pick the 2 best supposed models and randomly 2 other to explore and give data for the score model","metadata":{}},{"cell_type":"code","source":"import random\nexploitation  = 2 if vote else 3\nexploration   = 2\n\n# filter out already scored configs\nmask = df_score[\"score\"].isna()\nmask = mask & (~df_score[\"model\"].isna())\nmask = mask & (~df_score[\"config\"].isna())\n\n# pick the best models\ndf_score_sub = df_score[mask].sort_values(\"prediction\", ascending=False)\nids = list(df_score_sub.head(exploitation).index)\n\nlog.info(f\"exploitation {ids}\")\n\n# randomly pick some models\nother = list(df_score_sub.index)\nother = sorted(other)\nother = [o for o in other[-250:]] # last 250\nother = [o for o in other if o not in ids]\nlast_run = [o for o in other if o > max_score_index] \nother = other + last_run * int(250/(len(last_run)+1)) # this run trials equally weigthed with last 250\n\nlog.info(f\"max_score_index {max_score_index}\")\n\nfor _ in range(exploration):\n    random.shuffle(other)\n    ids = ids + other[:1] # 2 (if exploration is 2) randomly picked\n    other = [o for o in other if o != other[:1]]\n    log.info(f\"+ exploration {other[:1]}\")\n\nids","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:21:25.19616Z","iopub.execute_input":"2022-07-15T19:21:25.196531Z","iopub.status.idle":"2022-07-15T19:21:25.222809Z","shell.execute_reply.started":"2022-07-15T19:21:25.196503Z","shell.execute_reply":"2022-07-15T19:21:25.221957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json \nimport datetime\n\ncols = [c for c in df_data.columns if c != \"id\"]\nx = df_data[cols].values\n\nfor i in ids:\n\n    preds = fit_predict_model_i(x, i)\n\n    log.info(\"save predictions...\")\n    df_solution = pd.DataFrame()\n    df_solution[\"id\"] = df_data[\"id\"]\n    df_solution[\"Predicted\"] = preds\n    df_solution.set_index(\"id\", inplace=True)\n    file_name = f\"submission_{i}.csv\"\n    df_solution.to_csv(file_name)\n    log.info(file_name)\n    \n    message = f\"ID {i}\"\n    \n    os.system(f\"kaggle competitions submit -c tabular-playground-series-jul-2022 -f {file_name} -m \\\"{message}\\\"\")\n    os.remove(file_name)\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:21:21.05557Z","iopub.status.idle":"2022-07-15T19:21:21.055977Z","shell.execute_reply.started":"2022-07-15T19:21:21.055788Z","shell.execute_reply":"2022-07-15T19:21:21.055806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log.info(\"done, I will be back at the end of July and see the results...\")","metadata":{"execution":{"iopub.status.busy":"2022-07-15T19:21:21.057987Z","iopub.status.idle":"2022-07-15T19:21:21.058821Z","shell.execute_reply.started":"2022-07-15T19:21:21.058574Z","shell.execute_reply":"2022-07-15T19:21:21.058595Z"},"trusted":true},"execution_count":null,"outputs":[]}]}