{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data Loading and Cleaning","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\n\ndf = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\n\ndf['label'].value_counts().plot(kind='bar', title='Label Distribution')\nplt.show()\n\nsample_df = df.sample(n=5000, random_state=3)\n\nsample_ids = sample_df.sample(9, random_state=3)['id'].values\nfig, axes = plt.subplots(3, 3, figsize=(8, 5))\nfor ax, img_id in zip(axes.flatten(), sample_ids):\n    img = Image.open(f'/kaggle/input/histopathologic-cancer-detection/train/{img_id}.tif')\n    ax.imshow(img)\n    ax.axis('off')\nplt.tight_layout()\nplt.show()\n\ndef load_and_preprocess_image(img_id, img_size=(32,32)):\n    img_path = f'/kaggle/input/histopathologic-cancer-detection/train/{img_id}.tif'\n    img = Image.open(img_path)\n    img = img.resize(img_size)\n    img = np.array(img)\n    return img.flatten()\n\nX = []\ny = []\n\nfor idx, row in sample_df.iterrows():\n    img_vector = load_and_preprocess_image(row['id'])\n    X.append(img_vector)\n    y.append(int(row['label']))\n\nX = np.array(X)\ny = np.array(y)\n\nprint(\"X shape:\", X.shape)\nprint(\"y shape:\", y.shape)\n\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\nX_train, X_test, y_train, y_test = train_test_split(X_scaled, y, test_size=0.2, random_state=3, stratify=y)\n\nprint(\"Train set:\", X_train.shape)\nprint(\"Test set:\", X_test.shape)\n\nnp.save('X_train.npy', X_train)\nnp.save('X_test.npy', X_test)\nnp.save('y_train.npy', y_train)\nnp.save('y_test.npy', y_test)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-28T15:53:03.676757Z","iopub.execute_input":"2025-04-28T15:53:03.677086Z","iopub.status.idle":"2025-04-28T15:53:12.268155Z","shell.execute_reply.started":"2025-04-28T15:53:03.677063Z","shell.execute_reply":"2025-04-28T15:53:12.267333Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Naive Bayes","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. CART with Bagging and Cost-Complexity-Pruning","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Random forest","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Logistic regression","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Neural nets","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}