{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This code shows one way to determine important features of the data.","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:37:20.505971Z","iopub.execute_input":"2022-07-28T02:37:20.506489Z","iopub.status.idle":"2022-07-28T02:37:20.512985Z","shell.execute_reply.started":"2022-07-28T02:37:20.506444Z","shell.execute_reply":"2022-07-28T02:37:20.511896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T02:28:18.581699Z","iopub.execute_input":"2022-07-28T02:28:18.582266Z","iopub.status.idle":"2022-07-28T02:28:18.592170Z","shell.execute_reply.started":"2022-07-28T02:28:18.582220Z","shell.execute_reply":"2022-07-28T02:28:18.591165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:18.601779Z","iopub.execute_input":"2022-07-28T02:28:18.602878Z","iopub.status.idle":"2022-07-28T02:28:18.609316Z","shell.execute_reply.started":"2022-07-28T02:28:18.602830Z","shell.execute_reply":"2022-07-28T02:28:18.608050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:18.611626Z","iopub.execute_input":"2022-07-28T02:28:18.612100Z","iopub.status.idle":"2022-07-28T02:28:18.623452Z","shell.execute_reply.started":"2022-07-28T02:28:18.612058Z","shell.execute_reply":"2022-07-28T02:28:18.622155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv ('/kaggle/input/tabular-playground-series-jul-2022/data.csv',index_col='id')\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:18.625475Z","iopub.execute_input":"2022-07-28T02:28:18.625917Z","iopub.status.idle":"2022-07-28T02:28:19.668007Z","shell.execute_reply.started":"2022-07-28T02:28:18.625876Z","shell.execute_reply":"2022-07-28T02:28:19.666648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\ndf_scaled = scaler.fit_transform(df)\ndf_scaled1 = pd.DataFrame(df_scaled, columns=df.columns)\ndf_scaled1 ","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:24.256355Z","iopub.execute_input":"2022-07-28T02:28:24.257158Z","iopub.status.idle":"2022-07-28T02:28:24.366788Z","shell.execute_reply.started":"2022-07-28T02:28:24.257107Z","shell.execute_reply":"2022-07-28T02:28:24.364992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kmeans = KMeans(n_clusters=5, random_state=42).fit(df_scaled1)\nkmeans.labels_","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:19.669870Z","iopub.execute_input":"2022-07-28T02:28:19.670625Z","iopub.status.idle":"2022-07-28T02:28:24.253597Z","shell.execute_reply.started":"2022-07-28T02:28:19.670579Z","shell.execute_reply":"2022-07-28T02:28:24.252316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nclf1 = RandomForestClassifier(random_state=10)\nclf1.fit(df_scaled1,kmeans.labels_ )","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:24.368879Z","iopub.execute_input":"2022-07-28T02:28:24.369574Z","iopub.status.idle":"2022-07-28T02:29:39.926041Z","shell.execute_reply.started":"2022-07-28T02:28:24.369454Z","shell.execute_reply":"2022-07-28T02:29:39.924484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16,6))\nsns.lineplot(x=df_scaled1.columns, y=clf1.feature_importances_, marker='v')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:29:39.927886Z","iopub.execute_input":"2022-07-28T02:29:39.928520Z","iopub.status.idle":"2022-07-28T02:29:40.323880Z","shell.execute_reply.started":"2022-07-28T02:29:39.928473Z","shell.execute_reply":"2022-07-28T02:29:40.322484Z"},"trusted":true},"execution_count":null,"outputs":[]}]}