{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd\nfrom sklearn.metrics import make_scorer, roc_auc_score,classification_report\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.utils import class_weight\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedShuffleSplit","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train =  pd.read_csv(\"../input/tabular-melonoma/trainmod.csv\")\ntest = pd.read_csv(\"../input/tabular-melonoma/test_sub.csv\")\nlocing = [\"l\"+str(i) for i in range(10)]\ncolors_table = [\"Color\"+str(canal)+str(znach) for znach in range(3) for canal in range(3)]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"def label_e(dataframe):\n    dataframe.loc[dataframe[\"sex\"].isnull(),[\"sex\"]] = \"male\"\n    dataframe.loc[dataframe[\"age_approx\"].isnull(),[\"age_approx\"]] = 50\n    dataframe.loc[dataframe[\"anatom_site_general_challenge\"].isnull(),[\"anatom_site_general_challenge\"]] = \"torso\"\n    dataframe[\"split\"] = 0\n\n    dataframe.loc[dataframe[\"age_approx\"]<=40,[\"split\"]] = 1\n    dataframe.loc[(dataframe[\"age_approx\"]>40) & (dataframe[\"age_approx\"]<=76),[\"split\"]] = 2\n    dataframe.loc[dataframe[\"age_approx\"]>76,[\"split\"]] = 3\n    patient_id = LabelEncoder()\n    sex = LabelEncoder()\n    # age_approx = LabelEncoder()\n    anatom_site_general_challenge = LabelEncoder()\n\n    patient_id.fit(dataframe[\"patient_id\"].unique())\n    sex.fit(dataframe[\"sex\"].unique())\n    # age_approx.fit(train[\"age_approx\"].unique())\n    anatom_site_general_challenge.fit(dataframe[\"anatom_site_general_challenge\"].unique())\n\n    dataframe[\"patient_id\"] = patient_id.transform(dataframe[\"patient_id\"])\n    dataframe[\"sex\"] = sex.transform(dataframe[\"sex\"])\n    # train[\"age_approx\"] = age_approx.transform(train[\"age_approx\"])\n    dataframe[\"anatom_site_general_challenge\"] = anatom_site_general_challenge.transform(dataframe[\"anatom_site_general_challenge\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label_e(train)\nlabel_e(test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_c = train.copy()\ntrain_split = 0\ntrain_val_split = 0\n\nsplit = StratifiedShuffleSplit(n_splits=1, test_size=0.2, random_state=6)\nfor train_index, test_index in split.split(train_c,train_c[\"target\"]):\n    train_split = train_c.loc[train_index].copy()\n    train_val_split = train_c.loc[test_index].copy()\n    train_split.drop([\"split\"], axis=1, inplace=True)\n    train_val_split.drop([\"split\"], axis=1, inplace=True)\nlocing2 = np.hstack((locing,[\"age_approx\",\"veil\",\"width\",\"height\",\"globuli\",\"patient_id\",\"anatom_site_general_challenge\",\"sex\"]))\nlocing2 = np.hstack((locing2,colors_table))\ntrain_x = train_split[locing2]\ntrain_y = train_split[\"target\"]\nval_x = train_val_split[locing2]\nval_y = train_val_split[\"target\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"CW = class_weight.compute_class_weight('balanced',\n                                                 np.unique(train[\"target\"]),\n                                                 train[\"target\"])\nclases = [0,1]\nclass_weights = dict(zip(clases,CW))\nclass_weights","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tree = RandomForestClassifier(n_estimators=69, max_depth=50, min_samples_split=9,  min_samples_leaf=12, class_weight=class_weights)\ntree.fit(train_x,train_y)\ny_pred1 = tree.predict_proba(val_x)\nprint(make_scorer(roc_auc_score, needs_proba=True)(tree, val_x, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_x = test[locing2]\ntest_x.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_pred = tree.predict_proba(test_x)\nprediction = pd.DataFrame(test_pred,columns=[\"t\",\"target\"])\ntest[\"target\"] = prediction[\"target\"]\nsubmission = test[[\"image_name\",\"target\"]]\nsubmission.to_csv(\"submit.csv\", index=False, line_terminator=\"\\n\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}