{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-29T01:19:23.418507Z","iopub.execute_input":"2022-07-29T01:19:23.419248Z","iopub.status.idle":"2022-07-29T01:19:23.450563Z","shell.execute_reply.started":"2022-07-29T01:19:23.419143Z","shell.execute_reply":"2022-07-29T01:19:23.449665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nimport optuna.integration.lightgbm as lgbo\n\nfrom sklearn import preprocessing\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler, MaxAbsScaler, RobustScaler, PowerTransformer, QuantileTransformer\nmmscaler = MinMaxScaler(feature_range=(0, 1), copy=True)\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_absolute_error # 平均絶対誤差\n#from sklearn.metrics import mean_squared_error # 平均二乗誤差\n#from sklearn.metrics import mean_squared_log_error # 対数平均二乗誤差\nfrom sklearn.metrics import r2_score # 決定係数\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set()\n\nimport missingno as msno\nimport plotly.express as px\n\nimport json\nfrom collections import OrderedDict\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:23.452287Z","iopub.execute_input":"2022-07-29T01:19:23.452909Z","iopub.status.idle":"2022-07-29T01:19:27.731896Z","shell.execute_reply.started":"2022-07-29T01:19:23.452875Z","shell.execute_reply":"2022-07-29T01:19:27.730531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pandas setting to display more dataset rows and columns\npd.set_option('display.max_rows', 150)\npd.set_option('display.max_columns', 600)\npd.set_option('display.max_colwidth', None)\npd.set_option('display.float_format', lambda x: '%.5f' % x)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:27.733605Z","iopub.execute_input":"2022-07-29T01:19:27.734733Z","iopub.status.idle":"2022-07-29T01:19:27.741824Z","shell.execute_reply.started":"2022-07-29T01:19:27.734690Z","shell.execute_reply":"2022-07-29T01:19:27.740404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Import data","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/tabular-playground-series-jul-2022/sample_submission.csv\")\ndata = pd.read_csv(\"/kaggle/input/tabular-playground-series-jul-2022/data.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:27.745331Z","iopub.execute_input":"2022-07-29T01:19:27.745840Z","iopub.status.idle":"2022-07-29T01:19:29.131248Z","shell.execute_reply.started":"2022-07-29T01:19:27.745793Z","shell.execute_reply":"2022-07-29T01:19:29.129978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:29.132562Z","iopub.execute_input":"2022-07-29T01:19:29.132992Z","iopub.status.idle":"2022-07-29T01:19:29.153554Z","shell.execute_reply.started":"2022-07-29T01:19:29.132949Z","shell.execute_reply":"2022-07-29T01:19:29.152374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:29.155602Z","iopub.execute_input":"2022-07-29T01:19:29.156261Z","iopub.status.idle":"2022-07-29T01:19:29.196150Z","shell.execute_reply.started":"2022-07-29T01:19:29.156213Z","shell.execute_reply":"2022-07-29T01:19:29.195150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:29.197644Z","iopub.execute_input":"2022-07-29T01:19:29.198131Z","iopub.status.idle":"2022-07-29T01:19:29.233876Z","shell.execute_reply.started":"2022-07-29T01:19:29.198082Z","shell.execute_reply":"2022-07-29T01:19:29.232969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. EDA","metadata":{}},{"cell_type":"code","source":"# 特徴量ごとの分布を確認\n# Check the distribution for each feature\n\ndata.drop(columns=['id']).describe().T\\\n        .style.bar(subset=['mean'], color=px.colors.qualitative.G10[0])\\\n        .background_gradient(subset=['std'], cmap='Greens')\\\n        .background_gradient(subset=['50%'], cmap='BuGn')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:29.235319Z","iopub.execute_input":"2022-07-29T01:19:29.235954Z","iopub.status.idle":"2022-07-29T01:19:29.526571Z","shell.execute_reply.started":"2022-07-29T01:19:29.235916Z","shell.execute_reply":"2022-07-29T01:19:29.525303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"f_07～f_13は正の整数値、それ以外は0付近を中心とした数値のようです。<br>\nIt seems that f_07 to f_13 are positive integer values, and other values are centered around 0.","metadata":{}},{"cell_type":"code","source":"# 特徴量ごとの分布を可視化\n# Visualization of distribution for each feature\n\nfigure = plt.figure(figsize=(16, 18))\nfeatCount = 29\nfor i in range(featCount):\n    if i < 10:\n        feat_name = 'f_0' + str(i)\n    else:\n        feat_name = 'f_' + str(i)\n    plt.subplot(10, 3, i+1)\n    plt.hist(data[feat_name], bins=100)\n    plt.title(f'{feat_name}')\nfigure.tight_layout(h_pad=1.0, w_pad=1.0)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:29.528183Z","iopub.execute_input":"2022-07-29T01:19:29.528685Z","iopub.status.idle":"2022-07-29T01:19:39.990923Z","shell.execute_reply.started":"2022-07-29T01:19:29.528640Z","shell.execute_reply":"2022-07-29T01:19:39.989649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"整数値の特徴量項目以外は概ね正規分布しているようです。<br>\nExcept for the integer features, they are normally distributed.","metadata":{}},{"cell_type":"code","source":"# 特徴量間の相関を可視化\n# Heatmap\n\ncorr = data.corr().round(2)\nplt.figure(figsize=(20,10))\nsns.heatmap(corr, vmin=-1, vmax=1, center=0, square=False, annot=True, cmap='coolwarm')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:39.995265Z","iopub.execute_input":"2022-07-29T01:19:39.995918Z","iopub.status.idle":"2022-07-29T01:19:43.805755Z","shell.execute_reply.started":"2022-07-29T01:19:39.995880Z","shell.execute_reply":"2022-07-29T01:19:43.804619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"f_07～f_13（整数値の特徴量）とf_22～f_28の部分において、その部分内および部分間での相関があります。<br>\nこれらの部分を抽出しクラスター分類の根拠とするのが有効と思われます。<br>\nIn the parts of f_07 to f_13 (features of integer values) and f_22 to f_28, there are correlations within each part and between parts.<br>\nIt seems effective to extract these parts and use them as the basis for clustering.","metadata":{}},{"cell_type":"code","source":"# Extract columns\ncols_ex = ['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26', 'f_27', 'f_28', ]","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:43.807562Z","iopub.execute_input":"2022-07-29T01:19:43.808606Z","iopub.status.idle":"2022-07-29T01:19:43.815457Z","shell.execute_reply.started":"2022-07-29T01:19:43.808558Z","shell.execute_reply":"2022-07-29T01:19:43.813993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = data[cols_ex]\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:43.816851Z","iopub.execute_input":"2022-07-29T01:19:43.817208Z","iopub.status.idle":"2022-07-29T01:19:43.846196Z","shell.execute_reply.started":"2022-07-29T01:19:43.817177Z","shell.execute_reply":"2022-07-29T01:19:43.844959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df = pd.DataFrame(data = data, columns = cols_ex)\nplt.figure(figsize=(20,10)) \nsns.boxplot(x=\"variable\", y=\"value\", data=pd.melt(tmp_df)).set_title('Boxplot of each feature',size=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:43.847557Z","iopub.execute_input":"2022-07-29T01:19:43.848022Z","iopub.status.idle":"2022-07-29T01:19:45.428053Z","shell.execute_reply.started":"2022-07-29T01:19:43.847988Z","shell.execute_reply":"2022-07-29T01:19:45.426652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"外れ値がやや見られます。<br>\nThere are some outliers.","metadata":{}},{"cell_type":"markdown","source":"# 3. Standardize data\nUse scaler and transformer","metadata":{}},{"cell_type":"code","source":"# Select scaler and transformer\n\n#scaler = StandardScaler()\n#df = scaler.fit_transform(df)\n\n#Abs_scaler = MaxAbsScaler().fit(df)\n#df = Abs_scaler.fit_transform(df)\n\n#rob_scaler = RobustScaler().fit(df)\n#df = rob_scaler.fit_transform(df)\n\npower_transformer = PowerTransformer().fit(df)\ndf = power_transformer.transform(df)\n\n#quantile_transformer = QuantileTransformer(output_distribution='normal').fit(df)\n#df = quantile_transformer.transform(df)\n\nX_scaled = pd.DataFrame(df, columns=cols_ex)\nX_scaled","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:45.431500Z","iopub.execute_input":"2022-07-29T01:19:45.431867Z","iopub.status.idle":"2022-07-29T01:19:47.283880Z","shell.execute_reply.started":"2022-07-29T01:19:45.431832Z","shell.execute_reply":"2022-07-29T01:19:47.282643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df = pd.DataFrame(data = X_scaled, columns = cols_ex)\nplt.figure(figsize=(20,10)) \nsns.boxplot(x=\"variable\", y=\"value\", data=pd.melt(tmp_df)).set_title('Boxplot of each feature',size=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:47.285514Z","iopub.execute_input":"2022-07-29T01:19:47.286382Z","iopub.status.idle":"2022-07-29T01:19:48.847511Z","shell.execute_reply.started":"2022-07-29T01:19:47.286337Z","shell.execute_reply":"2022-07-29T01:19:48.846166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Outlier processing\n\nfor col in X_scaled.columns:\n    X_scaled[col]=X_scaled[col].apply(lambda x:4.00 if x>4.00 else x)\n    X_scaled[col]=X_scaled[col].apply(lambda x:-4.00 if x<-4.00 else x)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:48.849318Z","iopub.execute_input":"2022-07-29T01:19:48.849845Z","iopub.status.idle":"2022-07-29T01:19:50.014730Z","shell.execute_reply.started":"2022-07-29T01:19:48.849799Z","shell.execute_reply":"2022-07-29T01:19:50.013348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntmp_df = pd.DataFrame(data = X_scaled, columns = cols_ex)\nplt.figure(figsize=(20,10)) \nsns.boxplot(x=\"variable\", y=\"value\", data=pd.melt(tmp_df)).set_title('Boxplot of each feature',size=15)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:50.016280Z","iopub.execute_input":"2022-07-29T01:19:50.016651Z","iopub.status.idle":"2022-07-29T01:19:51.601992Z","shell.execute_reply.started":"2022-07-29T01:19:50.016618Z","shell.execute_reply":"2022-07-29T01:19:51.600697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"figure = plt.figure(figsize=(16, 18))\nfor i in range(len(cols_ex)):\n    feat_name = cols_ex[i]\n    plt.subplot(5, 3, i+1)\n    plt.hist(X_scaled[feat_name], bins=100)\n    plt.title(f'{feat_name}')\nfigure.tight_layout(h_pad=1.0, w_pad=1.0)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:51.603179Z","iopub.execute_input":"2022-07-29T01:19:51.603519Z","iopub.status.idle":"2022-07-29T01:19:57.442993Z","shell.execute_reply.started":"2022-07-29T01:19:51.603488Z","shell.execute_reply":"2022-07-29T01:19:57.441793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Modeling and visualization","metadata":{}},{"cell_type":"code","source":"%%time\n# Find the optimal number of n_components\n\nfrom yellowbrick.cluster import KElbowVisualizer\n\nElbow_M = KElbowVisualizer(KMeans(), k=(4,13))\nElbow_M.fit(X_scaled)\nElbow_M.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:19:57.444576Z","iopub.execute_input":"2022-07-29T01:19:57.444906Z","iopub.status.idle":"2022-07-29T01:20:53.934212Z","shell.execute_reply.started":"2022-07-29T01:19:57.444878Z","shell.execute_reply":"2022-07-29T01:20:53.933312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Modeling and prediction\n# Extract the most frequent result while changing random_state\n\n##################################\n# param for \"random_state\"\nn_seed = 7\n##################################\n\npredsAry = []\nfor i in range(n_seed):\n    bgm = BayesianGaussianMixture(\n        n_components = 7,\n        covariance_type = 'full',\n#        n_init=10,\n        max_iter=200,\n        random_state = i+1\n    )\n    preds = bgm.fit_predict(X_scaled)\n    predsAry.append(preds)\n    print(str(i+1) + ' / {}'.format(n_seed))\n    \ndf_predsAry = pd.DataFrame(predsAry)\ndf_predsAry","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:20:53.935451Z","iopub.execute_input":"2022-07-29T01:20:53.936407Z","iopub.status.idle":"2022-07-29T01:26:39.355558Z","shell.execute_reply.started":"2022-07-29T01:20:53.936364Z","shell.execute_reply":"2022-07-29T01:26:39.354236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"BayesianGaussianMixtureで算出されるインデックス（数値：0,1,2･････）はその値自体は意味を持たず、”random_state”を変化させる毎に別のインデックスで振り分けるため、インデックスを統一する必要があります。<br>\nThe index (numerical value: 0,1,2 ...) calculated by BayesianGaussianMixture has no meaning in itself.\nEvery time \"random_state\" is changed, it is sorted by another index, so it is necessary to unify the indexes.","metadata":{}},{"cell_type":"code","source":"# 頻出度を基準にインデックスを振り直す\n# Re-index based on frequency\n\ndf_predsAry_copy = df_predsAry.copy()\n\nfor i in range(n_seed):\n    index_old = df_predsAry_copy.iloc[i,:].value_counts().keys()\n    for j in range(len(index_old)):\n        df_predsAry_copy.iloc[i,:] = df_predsAry_copy.iloc[i,:].replace(index_old[j],100+j)\ndf_predsAry = df_predsAry_copy - 100\ndf_predsAry","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:26:39.357454Z","iopub.execute_input":"2022-07-29T01:26:39.357905Z","iopub.status.idle":"2022-07-29T01:26:39.697119Z","shell.execute_reply.started":"2022-07-29T01:26:39.357859Z","shell.execute_reply":"2022-07-29T01:26:39.695925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"0～97999の項目を複数のクラスターに分類し、各クラスターに振り分けられた項目の個数の多い順に採番しているので、seedが別になると同じ項目であるにも関わらず、違う番号が振られる可能性があります。<br>\nこれを修正するために、再採番したseed同士を項目単位で比較し、同じ番号が振られている項目の個数をカウントします。<br>\nすべてのseed同士を比較した結果、他のseedと比べて合計カウントが低いseedは間違った番号が振られている可能性が高いので、これらを削除したうえで多数決を採ります。<br>\nItems from 0 to 97999 are classified into multiple clusters and numbered in descending order of the number of items assigned to each cluster, so different seeds are assigned different numbers even though they are the same items. May be.<br>\nTo fix this, compare the renumbered seeds on an item-by-item basis and count the number of items with the same number.<br>\nAs a result of comparing all seeds, it is highly possible that seeds with a lower total count than other seeds are numbered incorrectly, so delete these and then make a majority vote.","metadata":{}},{"cell_type":"code","source":"# seed同士を比較する\n# Compare seed\n\nobj = {}\nfor i in range(n_seed):\n    count = 0\n    for j in range(n_seed):\n        if i != j:\n            each_count = np.count_nonzero(df_predsAry.iloc[i,:] == df_predsAry.iloc[j,:])\n            count += each_count\n    obj[i] = count\nsortedObj = sorted(obj.items(), key=lambda i: i[1], reverse=True)\nsortedObj","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:26:39.698500Z","iopub.execute_input":"2022-07-29T01:26:39.698932Z","iopub.status.idle":"2022-07-29T01:26:39.766271Z","shell.execute_reply.started":"2022-07-29T01:26:39.698887Z","shell.execute_reply":"2022-07-29T01:26:39.764935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 上位2/3を採用\n# Adopt the top majority\nif int(n_seed * 2 / 3) < 1:\n    adopt_count = 1\nelse:\n    adopt_count = int(n_seed * 2 / 3)\nlist = pd.DataFrame(sortedObj[0:adopt_count])[0]\ndf_predsAry = df_predsAry.iloc[list]\ndf_predsAry","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:26:39.767919Z","iopub.execute_input":"2022-07-29T01:26:39.768383Z","iopub.status.idle":"2022-07-29T01:26:39.980789Z","shell.execute_reply.started":"2022-07-29T01:26:39.768339Z","shell.execute_reply":"2022-07-29T01:26:39.979627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# 多数決\n# Majority vote\n\npreds_mode = []\nfor i in range(98000):\n    if df_predsAry.iloc[:,i].mode().count() == 1:\n        preds_mode.append(df_predsAry.iloc[:,i].mode()[0])\n    else:\n        preds_mode.append(df_predsAry.iloc[0,i])\npreds = preds_mode.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:26:39.982570Z","iopub.execute_input":"2022-07-29T01:26:39.983146Z","iopub.status.idle":"2022-07-29T01:27:21.996918Z","shell.execute_reply.started":"2022-07-29T01:26:39.983111Z","shell.execute_reply":"2022-07-29T01:27:21.995866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualization\n\npca = PCA(n_components=2)\nreduced_data = pca.fit_transform(X_scaled)\n\ndf = pd.DataFrame(\n    {\n        \"x\": reduced_data[:,0],\n        \"y\": reduced_data[:,1],\n        \"clusters\" : preds\n    }\n)\nplt.figure(figsize=(20, 20))\nsns.scatterplot(x=df[\"x\"], y=df[\"y\"], hue=df[\"clusters\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:27:21.998231Z","iopub.execute_input":"2022-07-29T01:27:21.998656Z","iopub.status.idle":"2022-07-29T01:27:25.679991Z","shell.execute_reply.started":"2022-07-29T01:27:21.998623Z","shell.execute_reply":"2022-07-29T01:27:25.678531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pl = sns.countplot(x=df[\"clusters\"])\npl.set_title(\"Distribution of clusters\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:27:25.681588Z","iopub.execute_input":"2022-07-29T01:27:25.682670Z","iopub.status.idle":"2022-07-29T01:27:25.886080Z","shell.execute_reply.started":"2022-07-29T01:27:25.682621Z","shell.execute_reply":"2022-07-29T01:27:25.884867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Make submission file","metadata":{}},{"cell_type":"code","source":"sample_submission[\"Predicted\"] = preds\nsample_submission","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:27:25.887662Z","iopub.execute_input":"2022-07-29T01:27:25.888364Z","iopub.status.idle":"2022-07-29T01:27:25.965310Z","shell.execute_reply.started":"2022-07-29T01:27:25.888315Z","shell.execute_reply":"2022-07-29T01:27:25.964269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T01:27:25.970468Z","iopub.execute_input":"2022-07-29T01:27:25.971177Z","iopub.status.idle":"2022-07-29T01:27:26.138731Z","shell.execute_reply.started":"2022-07-29T01:27:25.971126Z","shell.execute_reply":"2022-07-29T01:27:26.137615Z"},"trusted":true},"execution_count":null,"outputs":[]}]}