{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":30840,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:35:55.131896Z","iopub.execute_input":"2025-01-24T04:35:55.132172Z","iopub.status.idle":"2025-01-24T04:39:33.632374Z","shell.execute_reply.started":"2025-01-24T04:35:55.132151Z","shell.execute_reply":"2025-01-24T04:39:33.630619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:40:12.819620Z","iopub.execute_input":"2025-01-24T04:40:12.819919Z","iopub.status.idle":"2025-01-24T04:40:13.459115Z","shell.execute_reply.started":"2025-01-24T04:40:12.819899Z","shell.execute_reply":"2025-01-24T04:40:13.458475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:40:23.156125Z","iopub.execute_input":"2025-01-24T04:40:23.156422Z","iopub.status.idle":"2025-01-24T04:40:23.160932Z","shell.execute_reply.started":"2025-01-24T04:40:23.156399Z","shell.execute_reply":"2025-01-24T04:40:23.160210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:40:56.566326Z","iopub.execute_input":"2025-01-24T04:40:56.566669Z","iopub.status.idle":"2025-01-24T04:40:56.610709Z","shell.execute_reply.started":"2025-01-24T04:40:56.566641Z","shell.execute_reply":"2025-01-24T04:40:56.609975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:41:04.906604Z","iopub.execute_input":"2025-01-24T04:41:04.906908Z","iopub.status.idle":"2025-01-24T04:41:04.942246Z","shell.execute_reply.started":"2025-01-24T04:41:04.906885Z","shell.execute_reply":"2025-01-24T04:41:04.941281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Missing values:\\n\", data.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:41:28.171097Z","iopub.execute_input":"2025-01-24T04:41:28.171419Z","iopub.status.idle":"2025-01-24T04:41:28.186875Z","shell.execute_reply.started":"2025-01-24T04:41:28.171391Z","shell.execute_reply":"2025-01-24T04:41:28.186107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encode categorical variables\nle = LabelEncoder()\ncategorical_columns = ['sex', 'anatom_site_general_challenge', 'diagnosis', 'benign_malignant']\nfor col in categorical_columns:\n    # Fill NaNs with a placeholder before encoding\n    data[col] = data[col].fillna('unknown')\n    data[col] = le.fit_transform(data[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:42:21.894291Z","iopub.execute_input":"2025-01-24T04:42:21.894640Z","iopub.status.idle":"2025-01-24T04:42:21.912307Z","shell.execute_reply.started":"2025-01-24T04:42:21.894611Z","shell.execute_reply":"2025-01-24T04:42:21.911574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = data.drop(columns=['image_name', 'patient_id', 'target', 'benign_malignant'])\ny = data['target']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:42:46.355281Z","iopub.execute_input":"2025-01-24T04:42:46.355602Z","iopub.status.idle":"2025-01-24T04:42:46.362184Z","shell.execute_reply.started":"2025-01-24T04:42:46.355576Z","shell.execute_reply":"2025-01-24T04:42:46.361534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X.shape)\nprint(y.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:43:40.741093Z","iopub.execute_input":"2025-01-24T04:43:40.741380Z","iopub.status.idle":"2025-01-24T04:43:40.745761Z","shell.execute_reply.started":"2025-01-24T04:43:40.741359Z","shell.execute_reply":"2025-01-24T04:43:40.744832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipeline = Pipeline([\n    ('imputer', SimpleImputer(strategy='median')),  \n    ('scaler', StandardScaler()),  \n    ('classifier', RandomForestClassifier(random_state=42, n_estimators=100))\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:44:08.148396Z","iopub.execute_input":"2025-01-24T04:44:08.148783Z","iopub.status.idle":"2025-01-24T04:44:08.152747Z","shell.execute_reply.started":"2025-01-24T04:44:08.148748Z","shell.execute_reply":"2025-01-24T04:44:08.151877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:44:29.898908Z","iopub.execute_input":"2025-01-24T04:44:29.899189Z","iopub.status.idle":"2025-01-24T04:44:29.908240Z","shell.execute_reply.started":"2025-01-24T04:44:29.899168Z","shell.execute_reply":"2025-01-24T04:44:29.907546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipeline.fit(X_train,y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:44:46.860198Z","iopub.execute_input":"2025-01-24T04:44:46.860511Z","iopub.status.idle":"2025-01-24T04:44:47.428481Z","shell.execute_reply.started":"2025-01-24T04:44:46.860485Z","shell.execute_reply":"2025-01-24T04:44:47.427626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = pipeline.predict(X_test)\ny_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:45:58.432525Z","iopub.execute_input":"2025-01-24T04:45:58.432822Z","iopub.status.idle":"2025-01-24T04:45:58.482323Z","shell.execute_reply.started":"2025-01-24T04:45:58.432802Z","shell.execute_reply":"2025-01-24T04:45:58.481529Z"}},"outputs":[],"execution_count":null},{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"code","source":"print(\"\\nClassification Report:\")\nprint(classification_report(y_test, y_pred))\n\nprint(\"\\nConfusion Matrix:\")\nprint(confusion_matrix(y_test, y_pred))\n\nprint(f\"\\nAccuracy Score: {accuracy_score(y_test, y_pred):.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:46:38.424268Z","iopub.execute_input":"2025-01-24T04:46:38.424652Z","iopub.status.idle":"2025-01-24T04:46:38.452768Z","shell.execute_reply.started":"2025-01-24T04:46:38.424622Z","shell.execute_reply":"2025-01-24T04:46:38.451831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"clf = pipeline.named_steps['classifier']\nfeature_importance = pd.DataFrame({\n    'feature': X.columns,\n    'importance': clf.feature_importances_\n}).sort_values('importance', ascending=False)\n\nprint(\"\\nFeature Importance:\")\nprint(feature_importance)\n\n# Visualize feature importance\nplt.figure(figsize=(10, 6))\nsns.barplot(x='importance', y='feature', data=feature_importance)\nplt.title('Feature Importance in Melanoma Classification')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-24T04:47:17.456547Z","iopub.execute_input":"2025-01-24T04:47:17.456848Z","iopub.status.idle":"2025-01-24T04:47:17.704930Z","shell.execute_reply.started":"2025-01-24T04:47:17.456826Z","shell.execute_reply":"2025-01-24T04:47:17.704154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}