{"cells":[{"metadata":{"trusted":true,"id":"bt7iedfykhEN"},"cell_type":"code","source":"from skimage import color","execution_count":null,"outputs":[]},{"metadata":{"id":"VoQnactlz7yw","trusted":true,"outputId":"0fd3516b-a07f-4cb8-f0af-7b685021d9bc"},"cell_type":"code","source":"import pandas as pd\nimport re\nimport random\nimport numpy as np\nimport os\nimport keras\n!pip install metrics\nimport metrics\nimport keras.backend as k\nfrom cv2 import imread,resize\nfrom time import time\n#from scipy.misc import imread\nfrom sklearn.metrics import accuracy_score,normalized_mutual_info_score\nfrom sklearn.preprocessing import LabelEncoder\ndf_train=pd.read_csv('/content/final_train.csv')\ndf_test=pd.read_csv('/content/test.csv')\n","execution_count":null,"outputs":[]},{"metadata":{"id":"KO7Hj05O1ACQ","trusted":true,"outputId":"bf99c189-46e1-4390-af1d-4e8259ead12b"},"cell_type":"code","source":"df_train.head()","execution_count":null,"outputs":[]},{"metadata":{"id":"N8vqKIAtCGZH","trusted":true,"outputId":"332f9a3c-72cb-46f9-bd36-e0e620684da4"},"cell_type":"code","source":"df_test.head()","execution_count":null,"outputs":[]},{"metadata":{"id":"Ze7I4ctk1CC9","trusted":true,"outputId":"762bdac0-e988-4dd1-e81d-9f52f372a271"},"cell_type":"code","source":"df_test.columns","execution_count":null,"outputs":[]},{"metadata":{"id":"Fe1S7aSh1E3y","trusted":true,"outputId":"6135a86c-26f8-4f32-95fb-b2ff6224a02d"},"cell_type":"code","source":"print(len(df_train['anatom_site_general_challenge'].unique()))","execution_count":null,"outputs":[]},{"metadata":{"id":"Imnj1-0z1XQT","trusted":true,"outputId":"f4a299a0-2ff5-4519-e501-bde8094bd354"},"cell_type":"code","source":"print(len(df_train['sex'].unique()))","execution_count":null,"outputs":[]},{"metadata":{"id":"mxwY_6Fr1cbx","trusted":true,"outputId":"4f1ac775-8c34-4cfd-bb22-00aafe135b2e"},"cell_type":"code","source":"np.random.seed(1234)\nboxplot = df_train.boxplot(column=['age_approx'])","execution_count":null,"outputs":[]},{"metadata":{"id":"uLY0REqZ11Zp","trusted":true},"cell_type":"code","source":"Q1 = df_train.quantile(0.25)\nQ3 = df_train.quantile(0.75)\nIQR = Q3 - Q1","execution_count":null,"outputs":[]},{"metadata":{"id":"n79dKd362VOC","trusted":true,"outputId":"7cedf952-d6d7-4f1e-cef2-485431fee13f"},"cell_type":"code","source":"IQR","execution_count":null,"outputs":[]},{"metadata":{"id":"neSLu-_l2WtC","trusted":true,"outputId":"5ad981e8-5315-4ede-d3d3-cfebee2120e0"},"cell_type":"code","source":"df_train[df_train['age_approx']==0]","execution_count":null,"outputs":[]},{"metadata":{"id":"fIhsFtJl2m8C","trusted":true,"outputId":"98841a81-3005-426e-f5c3-01883e92bf8d"},"cell_type":"code","source":"for i in df_train.columns:\n  column=df_train[i]\n  print(i,column.isna().sum())","execution_count":null,"outputs":[]},{"metadata":{"id":"L0-MU-Bq4sdT","trusted":true,"outputId":"0ff206c4-21ba-4fe6-b9ed-c3d6f331b339"},"cell_type":"code","source":"df_train['target'][df_train['sex'].isna()]","execution_count":null,"outputs":[]},{"metadata":{"id":"uS78FDHb4oMF","trusted":true,"outputId":"e0461e9a-8d07-47ee-b486-7e866c1d94e5"},"cell_type":"code","source":"df_train['target'][df_train['age_approx'].isna()]","execution_count":null,"outputs":[]},{"metadata":{"id":"9r6Ril2W66jt","trusted":true},"cell_type":"code","source":"def max_count(df,col_1):\n  return max(set(df[col_1]), key = list(df[col_1]).count) #.unique()\n","execution_count":null,"outputs":[]},{"metadata":{"id":"R-I6XAhQ3CKZ","trusted":true,"outputId":"fb27c1d7-f9c7-478f-8c66-989f4b10430b"},"cell_type":"code","source":"max(set(df_train['target'][df_train['anatom_site_general_challenge'].isna()]), key = list(df_train['target'][df_train['anatom_site_general_challenge'].isna()]).count) #.unique()","execution_count":null,"outputs":[]},{"metadata":{"id":"vHKC9sYG5Jc_","trusted":true,"outputId":"84e67e2f-c26f-4762-8df2-53408b5c1edb"},"cell_type":"code","source":"max(set(df_train['target']), key = list(df_train['target']).count)","execution_count":null,"outputs":[]},{"metadata":{"id":"hFBQ2Oic3Wu0","trusted":true},"cell_type":"code","source":"anatom_mx=max_count(df_train,'anatom_site_general_challenge')\n","execution_count":null,"outputs":[]},{"metadata":{"id":"roRu60nK58hf","trusted":true},"cell_type":"code","source":"sex_mx=max_count(df_train,'sex')","execution_count":null,"outputs":[]},{"metadata":{"id":"39NCOQta9p8Z","trusted":true},"cell_type":"code","source":"#index = df_train['age_approx'].index[df_train['age_approx'].apply(np.isnan)]","execution_count":null,"outputs":[]},{"metadata":{"id":"8lN7vtzS8EjK","trusted":true},"cell_type":"code","source":"def replace_val(df_train,column,val):\n  index = df_train[column].index[df_train[column].isna()]\n  for Index in index:\n    df_train[column][Index]=val\n    df_train[column][Index]=val\n  return df_train\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"6laxlgzzkhFa"},"cell_type":"code","source":"sex_le = LabelEncoder()\nanatom_le=LabelEncoder()","execution_count":null,"outputs":[]},{"metadata":{"id":"iaW2Rjm7PhG1","trusted":true},"cell_type":"code","source":"\n\ndef preprocess(df_train):\n  global sex_le\n  global anatom_le\n  df_train=replace_val(df_train,'anatom_site_general_challenge',anatom_mx)\n  df_train=replace_val(df_train,'sex',sex_mx)\n  q_low = df_train[\"age_approx\"].quantile(0.01)\n  q_hi  = df_train[\"age_approx\"].quantile(0.99)\n\n  df_filtered = df_train[(df_train[\"age_approx\"] < q_hi) & (df_train[\"age_approx\"] > q_low)]\n  \n  try:\n      df_filtered=df_filtered.drop(['diagnosis','benign_malignant'],axis=1)\n      sex_le.fit(df_filtered['sex'])\n      anatom_le.fit(df_filtered['anatom_site_general_challenge'])\n      df_filtered['sex']=sex_le.transform(df_filtered['sex'])\n      df_filtered['anatom_site_general_challenge']=anatom_le.transform(df_filtered['anatom_site_general_challenge'])\n  except Exception as e:\n      print(e)\n      df_filtered['sex']=sex_le.transform(df_filtered['sex'])\n      df_filtered['anatom_site_general_challenge']=anatom_le.transform(df_filtered['anatom_site_general_challenge'])\n  return df_filtered","execution_count":null,"outputs":[]},{"metadata":{"id":"Om-h7yAXQ_EU","trusted":true,"outputId":"b5cbe755-f3f6-43ac-e576-49c86786e550"},"cell_type":"code","source":"df_train1=preprocess(df_train)","execution_count":null,"outputs":[]},{"metadata":{"id":"xzNUIAllRDka","trusted":true,"outputId":"0d795aa8-d31f-426b-9541-f21ee187724e"},"cell_type":"code","source":"df_train1.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"g6CzrT4ukhFl"},"cell_type":"code","source":"import cv2\nimport glob\nimport random","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"aLq1pQDQkhFo"},"cell_type":"code","source":"len(list(df_train.index[df_train[\"target\"]==0]))","execution_count":null,"outputs":[]},{"metadata":{"id":"s_Cric1vRE6U","trusted":true},"cell_type":"code","source":"\n\ndef get_images(directory,df_train):\n    Images=[]\n    label=0\n    os.chdir(directory)\n    indx1=list(df_train[\"image_name\"].index[df_train[\"target\"]==1][:584])\n    indx1.extend(list(df_train[\"image_name\"].index[df_train[\"target\"]==0])[:2000])\n    indexes=random.sample(indx1,len(indx1))\n    #print(type(indexes))\n    print(len(indexes))\n    for image_file in indexes:\n        image=imread(df_train1[\"image_name\"][image_file]+\".jpg\")\n        image = color.rgb2gray(image) \n        image=cv2.resize(image,(150,150))\n        Images.append(image)\n        label=label+1\n        if label%100==0:\n            print(label)\n            print(np.array(Images).shape)\n    return Images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"bTwZ-l_zkhFv"},"cell_type":"code","source":"Images=get_images(\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\",df_train1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"OrJeZlwbkhFy"},"cell_type":"code","source":"#Images=np.array(Images)\n#Images.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"LTnT4ibnkhF0"},"cell_type":"code","source":"len(Images)\nfor i in range(len(Images)):\n    Images[i]=Images[i].astype('float32')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"veZtBcB3khF3"},"cell_type":"code","source":"from sklearn.cluster import KMeans\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"L9YBZywEkhF6"},"cell_type":"code","source":"train_x = np.stack(Images)\ntrain_x /= 255.0\ntrain_x = train_x.reshape(-1, 22500).astype('float32')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"B9I1-UT1khF8","outputId":"9d55fb3c-6d3d-420e-ea3d-e1c007b108d6"},"cell_type":"code","source":"train_x.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"qXRJ4zt6khGA","outputId":"077d9319-c205-44ff-cab0-11dce0f6a684"},"cell_type":"code","source":"km = KMeans(n_jobs=-1, n_clusters=2, n_init=20)\nkm.fit(train_x)\n#pred = km.predict(val_x)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"2mbcuhvFkhGF","outputId":"e5b7370e-dd50-45ef-9b39-36c0f8aa2011"},"cell_type":"code","source":"km","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"SPEugr8dkhGJ"},"cell_type":"code","source":"import pickle\nos.chdir(\"/kaggle/working\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"0aUCCfOdkhGP"},"cell_type":"code","source":"\npickle.dump(km, open(\"kmeans_cluster.pkl\", \"wb\"))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"I2aDgRHZkhGT"},"cell_type":"code","source":"\nkmeans = pickle.load(open(\"/kaggle/input/kmeans-model/kmeans_cluster1.pkl\", \"rb\"))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"igzim2BHkhGY","outputId":"69fa7db1-9282-4a3f-a8b7-ca36be6d9818"},"cell_type":"code","source":"km=kmeans","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"-250trS5khGf","outputId":"d143bc93-0c06-429a-ddb6-bb96c1e61562"},"cell_type":"code","source":"from IPython.display import FileLink\nos.chdir(\"/kaggle/working\")\nFileLink(\"kmeans_cluster.pkl\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"jYVxZypZkhGn"},"cell_type":"code","source":"index = ['Row'+str(i) for i in range(1, len(train_x)+1)]\n\ndf = pd.DataFrame(train_x, index=index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"hl8oiAxLkhGt"},"cell_type":"code","source":"df.to_csv(\"Images.csv\",index=False)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"AOxUZculkhGw","outputId":"a4f66f1d-3c1b-469a-e14a-a3cf953fc04c"},"cell_type":"code","source":"len(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"t1YU8eZNkhGz","outputId":"06981dd2-42b1-4234-836e-f4cd25757f30"},"cell_type":"code","source":"len(df_train1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"rN0f8Vn1khG3","outputId":"4cc5d21c-1da4-404f-b8d3-e67d247db411"},"cell_type":"code","source":"df_train.index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"rdJuATprkhG8"},"cell_type":"code","source":"df_train2=df_train1.reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"NuIn6b9DkhHC","outputId":"c0218954-1b97-4dc0-fe93-7feb9f1a5cbc"},"cell_type":"code","source":"len(df_train2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"t4s6DdZfkhHH","outputId":"09ccb026-6a97-4f95-e9f8-2a1bc63066fd"},"cell_type":"code","source":"df_train2.index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"212yn31_khHL"},"cell_type":"code","source":"Images=[]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"kJhQ8zttkhHQ"},"cell_type":"code","source":"\ndef get_images_cluster(directory,df):\n    global Images\n    label=0\n    os.chdir(directory)\n    #print(len(indexes))\n    for image_file in range(len(Images),len(df)):\n        try:\n            os.chdir(directory)\n            image=imread(df[\"image_name\"][image_file]+\".jpg\")\n            image = color.rgb2gray(image) \n            image=cv2.resize(image,(150,150))\n            image=[image]\n            test = np.stack(image)\n            test /= 255.0\n            test = test.reshape(-1, 22500).astype('float32')\n            pred = km.predict(test)\n            Images.append(pred[0])\n            label=label+1\n            if label%100==0:\n                print(label)\n                os.chdir(\"/kaggle/working\")\n                with open(\"prediction.txt\",\"w\") as f:\n                    for item in Images:\n                        f.write(\"{}\\n\".format(item))\n        except Exception as e:\n            print(e)\n            Images.append(0)\n    return Images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"eO48C33lkhHW","outputId":"c57564b9-b6d3-4070-caca-aa2e5d17c222"},"cell_type":"code","source":"Images=get_images_cluster(\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\",df_train2)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"HiL5-pDCkhHc","outputId":"be9d3697-aac7-4d85-f367-ca4386243413"},"cell_type":"code","source":"Images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"gf_d7bzTkhHi"},"cell_type":"code","source":"df_train2[\"cluster\"]=Images\nos.chdir(\"/kaggle/working\")\ndf_train2.to_csv(\"final_train.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"9OfJNYXWkhHp","outputId":"5f0d9ab0-a946-4a7f-a8e8-508c65f4fbd1"},"cell_type":"code","source":"df_test1=preprocess(df_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"tGfCNQnVkhHw","outputId":"0a202a01-73bd-4ff2-fb27-72c24c879f1a"},"cell_type":"code","source":"df_test1.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"8MbN9F7ckhH2"},"cell_type":"code","source":"df_test2=df_test1.reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"BdJqTO2xkhH-","outputId":"52c9949d-8d4f-4946-b8c7-a48149acf843"},"cell_type":"code","source":"df_test2.index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"476AvGCokhIC","outputId":"eacbe6da-4a8d-45af-c09a-76e9d4d4af7f"},"cell_type":"code","source":"Images=get_images_cluster(\"/kaggle/input/siim-isic-melanoma-classification/jpeg/test\",df_test2)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"AiMWB7_ykhIT","outputId":"db713077-9729-42ba-9e11-5b103d686651"},"cell_type":"code","source":"#os.chdir(\"/kaggle/input/siim-isic-melanoma-classification/jpeg/test\")\nImages","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"L-m3a4XskhIV"},"cell_type":"code","source":"df_test1[\"cluster\"]=Images\nos.chdir(\"/kaggle/working\")\ndf_test1.to_csv(\"final_test.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"wXlcZDjZkhIY","outputId":"53ba0b97-e495-4797-d233-b6fee4b153fb"},"cell_type":"code","source":"test.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"id":"uMt17aFXkhIf"},"cell_type":"code","source":"import xgboost as xgb\nfrom xgboost import XGBClassifier, XGBRegressor\nfrom sklearn.ensemble import (RandomForestClassifier, AdaBoostClassifier,GradientBoostingClassifier, ExtraTreesClassifier)","execution_count":null,"outputs":[]},{"metadata":{"id":"hrT7Y0thm4iy","outputId":"7e5c9814-a479-46b8-e706-a8f34a0c8f5e","trusted":false},"cell_type":"code","source":"!pip install catboost\nfrom catboost import CatBoostClassifier","execution_count":null,"outputs":[]},{"metadata":{"id":"QNPFPMwQm4eO","trusted":false},"cell_type":"code","source":"import xgboost as xgb\nfrom sklearn import model_selection\nfrom sklearn.preprocessing import LabelEncoder\nimport re","execution_count":null,"outputs":[]},{"metadata":{"id":"EPGEYKhXsypG","trusted":false},"cell_type":"code","source":"#len(df_train['patient_id'].unique())\ndf_train=pd.read_csv('/content/final_train.csv')\ndf_test=pd.read_csv('/content/final_test1.csv')","execution_count":null,"outputs":[]},{"metadata":{"id":"xDtwzI_kJIk8","outputId":"f12f1d94-08ef-4b7e-e263-1eabeb3ade5c","trusted":false},"cell_type":"code","source":"type(df_train['sex'][0])","execution_count":null,"outputs":[]},{"metadata":{"id":"KuLvakKotLH4","trusted":false},"cell_type":"code","source":"patient=LabelEncoder()\nvals=list(df_train['patient_id'])\nvals.extend(df_test['patient_id'])\npatient.fit(vals)\n\n\ndf_train['patient_id']=patient.transform(df_train['patient_id'])","execution_count":null,"outputs":[]},{"metadata":{"id":"6Yge3icPij_t","trusted":false},"cell_type":"code","source":"'''patient_se=LabelEncoder()\nvals1=list(df_test['sex'])\nvals1.extend(df_test['sex'])\npatient_se.fit(vals1)\n\n\ndf_train['sex']=patient_se.transform(df_train['sex'])'''","execution_count":null,"outputs":[]},{"metadata":{"id":"8SQgF4Aoinas","trusted":false},"cell_type":"code","source":"'''patient_an=LabelEncoder()\nvals2=list(df_train['anatom_site_general_challenge'])\nvals2.extend(df_test['anatom_site_general_challenge'])\npatient_an.fit(vals2)\n\n\ndf_train['anatom_site_general_challenge']=patient_an.transform(df_train['anatom_site_general_challenge'])'''","execution_count":null,"outputs":[]},{"metadata":{"id":"78h5-uNMLUqW","trusted":false},"cell_type":"code","source":"df_test['patient_id']=patient.transform(df_test['patient_id'])\n#df_test['sex']=patient_se.transform(df_test['sex'])\n#df_test['anatom_site_general_challenge']=patient_an.transform(df_test['anatom_site_general_challenge'])","execution_count":null,"outputs":[]},{"metadata":{"id":"jvGCQOWJtlM9","outputId":"9b7c5ab4-b14d-4b68-885c-64b3059029d4","trusted":false},"cell_type":"code","source":"type(df_test['patient_id'][0])","execution_count":null,"outputs":[]},{"metadata":{"id":"mjxxAdTyHFiA","outputId":"b9cd4de6-dbc6-411e-b7f2-f1dacce58991","trusted":false},"cell_type":"code","source":"df_test.head()","execution_count":null,"outputs":[]},{"metadata":{"id":"zeSjV-NR-f7d","outputId":"728bb774-c2ba-450f-b900-838319279991","trusted":false},"cell_type":"code","source":"import seaborn as sns\nsns.distplot(df_train['cluster'])","execution_count":null,"outputs":[]},{"metadata":{"id":"M1COjwIo_Z4C","trusted":false},"cell_type":"code","source":"Y=df_train['target']\nX=df_train[['patient_id', 'sex', 'age_approx','anatom_site_general_challenge', 'cluster']]\n#X[\"A\"] = X[\"A\"] / X[\"A\"].max()'''\nseed = 7\ntest_size = 0.33\nY=df_train['target']\nX=df_train[['patient_id', 'sex', 'age_approx','anatom_site_general_challenge', 'cluster']]\nX_train=X\ny_train=Y\nX_test=df_test[['patient_id', 'sex', 'age_approx','anatom_site_general_challenge', 'cluster']]","execution_count":null,"outputs":[]},{"metadata":{"id":"kJOA6ZB5-_tt","outputId":"1be1d3aa-d6da-421d-d2fe-3d031ed9d1aa","trusted":false},"cell_type":"code","source":"\ntrain_DMatrix = xgb.DMatrix(X_train, label= y_train)\ntest_DMatrix = xgb.DMatrix(X_test)\nparam = {\n    'booster':'gbtree', \n    'eta': 0.3,\n    'num_class': 2,\n    'max_depth': 100\n}\n\nepochs = 100\nclf = xgb.XGBClassifier(n_estimators=2000, \n                        max_depth=8, \n                        objective='multi:softprob',\n                        seed=0,  \n                        nthread=-1, \n                        learning_rate=0.15, \n                        num_class = 2, \n                        scale_pos_weight = (len(X_train)/584))\nclf.fit(X_train, y_train)\n#clf.predict_proba(X_test)[:,1]\n# clf.predict(x_test)\n","execution_count":null,"outputs":[]},{"metadata":{"id":"Ab7e5x7xgEhX","trusted":false},"cell_type":"code","source":"\ndf=pd.DataFrame()\ndf['image_name']=df_test['image_name']\n\ntarget = clf.predict_proba(X_test)[:,1]\ndf['target']=target\ndf.head()\ndf.to_csv('Submission_xgboost_cluster.csv',index=False)\n#sub_tabular = sub.copy()","execution_count":null,"outputs":[]},{"metadata":{"id":"C4_xnxdCuJ_1","trusted":false},"cell_type":"code","source":"Y=df_train['target']\nX=df_train[['patient_id', 'sex', 'age_approx','anatom_site_general_challenge']]\n#X[\"A\"] = X[\"A\"] / X[\"A\"].max()'''\nseed = 7\ntest_size = 0.33\nY=df_train['target']\nX=df_train[['patient_id', 'sex', 'age_approx','anatom_site_general_challenge']]\nX_train=X\ny_train=Y\nX_test=df_test[['patient_id', 'sex', 'age_approx','anatom_site_general_challenge']]\ntrain_DMatrix = xgb.DMatrix(X_train, label= y_train)\ntest_DMatrix = xgb.DMatrix(X_test)\nparam = {\n    'booster':'gbtree', \n    'eta': 0.3,\n    'num_class': 2,\n    'max_depth': 100\n}\n\nepochs = 100\nclf = xgb.XGBClassifier(n_estimators=2000, \n                        max_depth=8, \n                        objective='multi:softprob',\n                        seed=0,  \n                        nthread=-1, \n                        learning_rate=0.15, \n                        num_class = 2, \n                        scale_pos_weight = (len(X_train)/584))\nclf.fit(X_train, y_train)\n#clf.predict_proba(X_test)[:,1]\n# clf.predict(x_test)\n\ndf=pd.DataFrame()\ndf['image_name']=df_test['image_name']\n\ntarget = clf.predict_proba(X_test)[:,1]\ndf['target']=target\ndf.head()\ndf.to_csv('Submission_xgboost_.csv',index=False)\n#sub_tabular = sub.copy()","execution_count":null,"outputs":[]},{"metadata":{"id":"H33w00fCm4VL","outputId":"af6a0486-0b99-4f0a-aad2-ce99a7f7def1","trusted":false},"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nclf=LogisticRegression(random_state=0).fit(X_train,y_train)\nclf.predict(X_test)\ndf=pd.DataFrame()\ndf['image_name']=df_test['image_name']\n\ntarget = clf.predict_proba(X_test)[:,1]\ndf['target']=target\ndf.head()\ndf.to_csv('Submission_Logistic_.csv',index=False)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"id":"_XSJCJs7m4RM","outputId":"1c681107-2ac7-4351-8ca1-04277816b830","trusted":false},"cell_type":"code","source":"#prediction of the test set\n#importing keras and other required library's\n#importing and splitting data into training and testing \nimport keras\nfrom sklearn.model_selection import train_test_split\n\n# Feature Scaling\nfrom sklearn.preprocessing import StandardScaler\n\nsc = StandardScaler()\nX_train = sc.fit_transform(X_train)\nX_test = sc.transform(X_test)\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.preprocessing import StandardScaler # Used for scaling of data\nfrom sklearn.model_selection import train_test_split\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout\nfrom keras import metrics\nfrom sklearn import preprocessing\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom keras import backend as K\nfrom keras.wrappers.scikit_learn import KerasRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom math import sqrt\n#creating new neural network model\nclassifier=Sequential()\nclassifier.add(Dense(output_dim=100,init='uniform',activation='relu',input_dim=X.shape[1]))\nclassifier.add(Dropout(p=0.2))\n\n\nclassifier.add(Dense(output_dim=50,init='uniform',activation='relu'))\nclassifier.add(Dropout(p=0.2))\n\nclassifier.add(Dense(output_dim=1,init='uniform',activation='sigmoid'))\n\n\nclassifier.compile(optimizer='adam',loss='binary_crossentropy',metrics=['accuracy'])\n\n\n\n\n#trining or fitting the model\nclassifier.fit(X_train,y_train,batch_size=100,nb_epoch=100)\ny_pred = classifier.predict(X_test)\ny_pred\n","execution_count":null,"outputs":[]},{"metadata":{"id":"EMvfnAs2m4Kx","outputId":"93f530e6-b7e1-425b-f399-f89e41e87b63","trusted":false},"cell_type":"code","source":"df=pd.DataFrame()\ndf['image_name']=df_test['image_name']\n\ntarget = clf.predict_proba(X_test)[:,1]\ndf['target']=y_pred\ndf.to_csv('Submission_NN.csv',index=False)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"id":"Civ3-qRDuzEH","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}