{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pickle\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport pydicom as dicom\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom statistics import mean\nfrom sklearn.metrics import accuracy_score, confusion_matrix\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import model_selection\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn import svm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.ensemble import RandomForestClassifier, StackingClassifier\nfrom sklearn.base import BaseEstimator, TransformerMixin","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SEED = 42","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_melanoma = df[df['target'] == 1]\ntrain_benign = df[df['target'] == 0].sample(n=len(train_melanoma), random_state=SEED)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_melanoma.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.concat([train_melanoma, train_benign], ignore_index=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def pull_images(image_names):\n    results = []\n    for image_name in image_names:\n        image = '../input/siim-isic-melanoma-classification/train/' + image_name +'.dcm'\n        ds = dicom.dcmread(image)\n        pixels = ds.pixel_array\n        results.append(pixels.flatten())\n    results = tf.keras.preprocessing.sequence.pad_sequences(\n      results,\n      maxlen = 720,\n      dtype = \"int32\",\n      padding = \"pre\",\n      truncating = \"pre\",\n      value = 0\n    )\n    return results\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df.rename(columns={\"anatom_site_general_challenge\":\"site\", \"age_approx\": \"age\"})\ndf = df.drop([\"patient_id\"], axis=1)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df.dropna(axis=0, how=\"any\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(x = 'sex', data = df, hue = 'target')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.distplot(df['age'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cancer_patients = df[df['target'] == 1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cancer_patients_dist = cancer_patients[[\"age\"]]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.distplot(cancer_patients_dist)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.subplot(1,2,2)\nsns.countplot(y=cancer_patients_dist['age'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dummies = pd.get_dummies(df, columns=['site'])\ndummies.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dummies = pd.get_dummies(dummies, columns=['sex'], drop_first=True)\ndummies = pd.get_dummies(dummies, columns=['diagnosis'], drop_first=True)\n\ndummies","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def pull_details(X):  \n    return X.reindex(columns=['sex_male', 'age', 'site_lower extremity', 'site_torso', 'site_upper extremity', 'site_head/neck'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = pull_details(dummies)\nX","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y = dummies['target']\ny","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nkf = KFold(n_splits=4, shuffle=True)\n\n\ndef kfold_test(X, y, model):   \n    scores = []\n#     X['diagnosis_melanoma'] = 0\n#     X['diagnosis_unknown'] = 1\n    for train_index, test_index in kf.split(X):\n        X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n        y_train, y_test = y.iloc[train_index], y.iloc[test_index]\n        model = model.fit(X_train, y_train)\n        scores.append(accuracy_score(y_test, model.predict(X_test)))\n    return mean(scores)\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"details_lr = kfold_test(X, y, LogisticRegression())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"details_svc = kfold_test(X, y, svm.SVC())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"details_dtc = kfold_test(X, y, DecisionTreeClassifier())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"details_rfc = kfold_test(X, y, RandomForestClassifier())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\ndetails_gbc = kfold_test(X, y, GradientBoostingClassifier())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ndetails_abc = kfold_test(X, y, AdaBoostClassifier())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_l = ['Logistic Regression', 'Support Vector Machine', 'Decison Tree', 'Random Forest', 'Gradient Boosting', 'Adaptive Boosting' ]\ny_l = [details_lr, details_svc, details_dtc, details_rfc, details_gbc, details_abc]\n\nprint(list(zip(x_l, y_l)))\nsns.barplot(y_l, x_l,palette=\"rocket\")\nplt.xlim([0.5, 1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file = \"details_comparison.txt\"\nwith open(file, \"wb\") as f:\n    pickle.dump([x_l, y_l], f)\n    \nfile = \"columns.txt\"\nwith open(file, \"wb\") as f:\n    pickle.dump(X.columns, f)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nX.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_details_model = AdaBoostClassifier()\npatient_details_model.fit(X, y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"s0 = df.target[df.target.eq(0)].sample(100, random_state=SEED).index\ns1 = df.target[df.target.eq(1)].sample(100,random_state=SEED).index \n\n\nimage_data_sampled = df.loc[s0.union(s1)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_data_sampled","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images = pull_images(image_data_sampled['image_name'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def kfold_test_image(X, y, model):   \n    scores = []\n    for train_index, test_index in kf.split(X):\n        X_train, X_test = X[train_index], X[test_index]\n        y_train, y_test = y[train_index], y[test_index]\n        model = model.fit(X_train, y_train)\n        scores.append(accuracy_score(y_test, model.predict(X_test)))\n    return mean(scores)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = images\ny = np.array(image_data_sampled['target'])\n\n\n\n\nimage_lr = kfold_test_image(X, y, LogisticRegression(max_iter=4000))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_svc = kfold_test_image(X, y, svm.SVC())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_dtc = kfold_test_image(X, y, DecisionTreeClassifier())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_rfc = kfold_test_image(X, y, RandomForestClassifier())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_abc = kfold_test_image(X, y, AdaBoostClassifier())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_gbc = kfold_test_image(X, y, GradientBoostingClassifier())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_l = ['Logistic Regression', 'Support Vector Machine', 'Decison Tree', 'Random Forest', 'Adaptive Boosting', 'Gradient Boosting']\ny_l = [image_lr, image_svc, image_dtc, image_rfc, image_abc, image_gbc]\n\nsns.barplot(y_l, x_l,palette=\"rocket\")\nplt.xlim([0.5, 1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file = \"image_comparison.txt\"\nwith open(file, \"wb\") as f:\n    pickle.dump([x_l, y_l], f)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_image_model = RandomForestClassifier()\npatient_image_model.fit(X,y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file1 = \"model_patient_details.pkl\"\nwith open(file1, \"wb\") as f:\n    pickle.dump(patient_details_model, f)\n    \n    \nfile2 = \"model_patient_image.pkl\"\nwith open(file2, \"wb\") as f:\n    pickle.dump(patient_image_model, f)\n    \n    \n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}