{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T04:30:28.244392Z","iopub.execute_input":"2022-07-28T04:30:28.245096Z","iopub.status.idle":"2022-07-28T04:30:28.271419Z","shell.execute_reply.started":"2022-07-28T04:30:28.244997Z","shell.execute_reply":"2022-07-28T04:30:28.270459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\ntrain = df.drop(columns=['id'])\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-28T04:30:28.273020Z","iopub.execute_input":"2022-07-28T04:30:28.273558Z","iopub.status.idle":"2022-07-28T04:30:29.542595Z","shell.execute_reply.started":"2022-07-28T04:30:28.273525Z","shell.execute_reply":"2022-07-28T04:30:29.541477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_sample = train.sample(frac=0.1, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:37:13.386491Z","iopub.execute_input":"2022-07-25T07:37:13.387567Z","iopub.status.idle":"2022-07-25T07:37:13.392822Z","shell.execute_reply.started":"2022-07-25T07:37:13.387523Z","shell.execute_reply":"2022-07-25T07:37:13.391687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clipping values to remove outliers\nfor col in train.columns:\n    lower = np.percentile(train[col], 10)\n    upper = np.percentile(train[col], 90)\n    train[col] = train[col].clip(lower=lower, upper=upper)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:37:13.396065Z","iopub.execute_input":"2022-07-25T07:37:13.397112Z","iopub.status.idle":"2022-07-25T07:37:13.559523Z","shell.execute_reply.started":"2022-07-25T07:37:13.397063Z","shell.execute_reply":"2022-07-25T07:37:13.558260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# standardizing values\nfor col in train.columns:\n    mu = train[col].mean()\n    std = train[col].std()\n    train[col] = (train[col] - mu)/std","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:37:13.560951Z","iopub.execute_input":"2022-07-25T07:37:13.562110Z","iopub.status.idle":"2022-07-25T07:37:13.628687Z","shell.execute_reply.started":"2022-07-25T07:37:13.562062Z","shell.execute_reply":"2022-07-25T07:37:13.627555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:37:13.630396Z","iopub.execute_input":"2022-07-25T07:37:13.630871Z","iopub.status.idle":"2022-07-25T07:37:13.636611Z","shell.execute_reply.started":"2022-07-25T07:37:13.630827Z","shell.execute_reply":"2022-07-25T07:37:13.635284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.decomposition import PCA\n# pca = PCA(n_components=29)\n# pca.fit(train).explained_variance_ratio_","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:37:13.638519Z","iopub.execute_input":"2022-07-25T07:37:13.639026Z","iopub.status.idle":"2022-07-25T07:37:13.646410Z","shell.execute_reply.started":"2022-07-25T07:37:13.638985Z","shell.execute_reply":"2022-07-25T07:37:13.645382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.cluster import KMeans","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:37:13.648149Z","iopub.execute_input":"2022-07-25T07:37:13.649221Z","iopub.status.idle":"2022-07-25T07:37:13.657896Z","shell.execute_reply.started":"2022-07-25T07:37:13.649177Z","shell.execute_reply":"2022-07-25T07:37:13.656856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# from tqdm import tqdm\n\n# min_inertia = 1e10\n# best_param = 0\n# best_model = KMeans()\n# for i in tqdm(np.arange(2, 20)):\n#     model = KMeans(n_clusters=i)\n#     model.fit(train)\n#     inertia = model.inertia_\n#     if inertia < min_inertia:\n#         min_inertia = inertia\n#         best_param = i\n#         best_model = model\n# model = best_model\n# i, min_inertia","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:37:13.659862Z","iopub.execute_input":"2022-07-25T07:37:13.660710Z","iopub.status.idle":"2022-07-25T07:37:13.669028Z","shell.execute_reply.started":"2022-07-25T07:37:13.660664Z","shell.execute_reply":"2022-07-25T07:37:13.667772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"i, min_inertia for non-clipped = (19, 8048080.92)\n\ni, min_inertia for clipped = (19, 5317854.65)\n\ni, min_inertia for scaled data (non-clipped) = (19, 2328996.97)\n\ni, min_inertia for scaled data (clipped) = (19, 2328311.80)","metadata":{}},{"cell_type":"code","source":"# model = KMeans(n_clusters=7)\n# model.fit(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:37:13.672490Z","iopub.execute_input":"2022-07-25T07:37:13.672913Z","iopub.status.idle":"2022-07-25T07:37:13.682422Z","shell.execute_reply.started":"2022-07-25T07:37:13.672877Z","shell.execute_reply":"2022-07-25T07:37:13.681083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.mixture import BayesianGaussianMixture\nmodel = BayesianGaussianMixture(n_components=7).fit(train)\nmodel.predict(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T07:37:13.683902Z","iopub.execute_input":"2022-07-25T07:37:13.684598Z","iopub.status.idle":"2022-07-25T07:38:27.098354Z","shell.execute_reply.started":"2022-07-25T07:37:13.684556Z","shell.execute_reply":"2022-07-25T07:38:27.096999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df['Predicted'] = model.labels_ # for KMeans\ndf['Predicted'] = np.random.randint(7, size=(98000,))\ndf[['id', 'Predicted']].rename(columns={'id': 'Id'}).to_csv('./submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T04:32:15.467180Z","iopub.execute_input":"2022-07-28T04:32:15.467588Z","iopub.status.idle":"2022-07-28T04:32:15.582335Z","shell.execute_reply.started":"2022-07-28T04:32:15.467557Z","shell.execute_reply":"2022-07-28T04:32:15.581044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('./submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T04:32:16.212437Z","iopub.execute_input":"2022-07-28T04:32:16.213314Z","iopub.status.idle":"2022-07-28T04:32:16.235710Z","shell.execute_reply.started":"2022-07-28T04:32:16.213238Z","shell.execute_reply":"2022-07-28T04:32:16.234601Z"},"trusted":true},"execution_count":null,"outputs":[]}]}