{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Define","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\n\nTRAIN_PATH = \"../input/titanic/train.csv\"\n\nID = \"PassengerId\"\nTARGET = \"Survived\"\n\ntrain = pd.read_csv(TRAIN_PATH)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T22:55:05.663269Z","iopub.execute_input":"2022-07-27T22:55:05.663795Z","iopub.status.idle":"2022-07-27T22:55:05.680295Z","shell.execute_reply.started":"2022-07-27T22:55:05.663747Z","shell.execute_reply":"2022-07-27T22:55:05.678285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Function","metadata":{}},{"cell_type":"code","source":"def getFeatureImportance(df,ID,TARGET,num = -1):\n    \n    # null process\n    def checkNull_fillData(df):\n        for col in df.columns:\n            if len(df.loc[df[col].isnull() == True]) != 0:\n                if df[col].dtype == \"str\" or df[col].dtype == \"object\":\n                    df.loc[df[col].isnull() == True,col] = df[col].mode()[0]\n                else:\n                    df.loc[df[col].isnull() == True,col] = df[col].mean()\n                \n    checkNull_fillData(df)\n    \n    # get string column\n    str_list = [] \n    for colname, colvalue in df.iteritems():\n        if type(colvalue[1]) == str:\n            str_list.append(colname)\n    \n    #Label Encoding\n    for col in str_list:\n        encoder = LabelEncoder()\n        encoder.fit(train[col])\n        df[col] = encoder.transform(train[col])\n    \n    # correlation \n    corrTrain = df.corr()\n    \n    # abs \n    corrTrain = np.abs(corrTrain)\n    \n    # to dataframe \n    corrTarget = corrTrain[[TARGET]]\n    \n    # sort \n    corrTarget = corrTarget.sort_values(TARGET,ascending=False)\n    corrTargetT = corrTarget.T\n    \n    # delete target\n    corrTargetTImprortance = [col for col in corrTargetT.columns if col != TARGET and col != ID]\n    \n    # select feature importance top num\n    if num != -1:\n         selCorrTargetTImprortance = corrTargetTImprortance[:num]\n    else:\n        selCorrTargetTImprortance = corrTargetTImprortance\n        \n    result = corrTargetT[selCorrTargetTImprortance]\n    \n    return result.T","metadata":{"execution":{"iopub.status.busy":"2022-07-27T22:55:05.794254Z","iopub.execute_input":"2022-07-27T22:55:05.794720Z","iopub.status.idle":"2022-07-27T22:55:05.808178Z","shell.execute_reply.started":"2022-07-27T22:55:05.794683Z","shell.execute_reply":"2022-07-27T22:55:05.806883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# All","metadata":{}},{"cell_type":"code","source":"result = getFeatureImportance(train,ID,TARGET)\nresult","metadata":{"execution":{"iopub.status.busy":"2022-07-27T22:55:05.944450Z","iopub.execute_input":"2022-07-27T22:55:05.945442Z","iopub.status.idle":"2022-07-27T22:55:05.980210Z","shell.execute_reply.started":"2022-07-27T22:55:05.945393Z","shell.execute_reply":"2022-07-27T22:55:05.979315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# top 3","metadata":{}},{"cell_type":"code","source":"result = getFeatureImportance(train,ID,TARGET,3)\nresult","metadata":{"execution":{"iopub.status.busy":"2022-07-27T22:55:06.026530Z","iopub.execute_input":"2022-07-27T22:55:06.027188Z","iopub.status.idle":"2022-07-27T22:55:06.050360Z","shell.execute_reply.started":"2022-07-27T22:55:06.027154Z","shell.execute_reply":"2022-07-27T22:55:06.049037Z"},"trusted":true},"execution_count":null,"outputs":[]}]}