{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":5048,"databundleVersionId":868335,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pickle\nimport numpy as np\nimport seaborn as sns\nfrom sklearn.datasets import load_files\nimport matplotlib.pyplot as plt\nfrom keras.layers import Conv2D, MaxPooling2D, GlobalAveragePooling2D\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom keras.layers import Dropout, Flatten, Dense\nfrom keras.models import Sequential\nfrom tensorflow.keras import models\nfrom tensorflow.keras.optimizers import Adam, SGD, RMSprop\nfrom tensorflow.keras import layers\nfrom keras.utils import plot_model\nfrom keras.callbacks import ModelCheckpoint\nfrom keras.utils import to_categorical\nfrom sklearn.metrics import confusion_matrix\nfrom keras.preprocessing import image                  \nfrom tqdm import tqdm\n\nimport seaborn as sns\nfrom sklearn.metrics import accuracy_score,precision_score,recall_score,f1_score\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-17T17:01:18.735941Z","iopub.execute_input":"2025-09-17T17:01:18.736113Z","iopub.status.idle":"2025-09-17T17:01:33.450003Z","shell.execute_reply.started":"2025-09-17T17:01:18.736097Z","shell.execute_reply":"2025-09-17T17:01:33.449222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dir = \"../input/state-farm-distracted-driver-detection/imgs\"\ntrain_dir = os.path.join(data_dir,\"train\")\ntest_dir = os.path.join(data_dir,\"test\")\nmodel1_path = os.path.join(os.getcwd(),\"model1\",\"self_trained\")\npickle_dir = os.path.join(os.getcwd(),\"pickle_files\")\ncsv_dir = os.path.join(os.getcwd(),\"csv_files\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T17:01:36.614708Z","iopub.execute_input":"2025-09-17T17:01:36.615056Z","iopub.status.idle":"2025-09-17T17:01:36.619543Z","shell.execute_reply.started":"2025-09-17T17:01:36.615029Z","shell.execute_reply":"2025-09-17T17:01:36.618831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_size = (224, 224) \nbatch_size = 32\nseed = 42\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,         \n    validation_split=0.2,   \n    rotation_range=15,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    shear_range=0.1,\n    zoom_range=0.1,\n    horizontal_flip=True,\n    fill_mode=\"nearest\"\n)\n\ntest_val_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = train_datagen.flow_from_directory(\n    train_dir,\n    target_size=img_size,\n    batch_size=batch_size,\n    class_mode=\"categorical\",\n    subset=\"training\",\n    seed=seed\n)\n\nval_generator = train_datagen.flow_from_directory(\n    train_dir,\n    target_size=img_size,\n    batch_size=batch_size,\n    class_mode=\"categorical\",\n    subset=\"validation\",\n    seed=seed\n)\n\n# test_generator = test_val_datagen.flow_from_directory(\n#     directory=os.path.dirname(test_dir),                # parent dir\n#     classes=['test'],         # only test folder\n#     target_size=(224,224),\n#     batch_size=32,\n#     class_mode=None,          # no labels\n#     shuffle=False\n# )\n#print(\"\\nClass indices:\", train_generator.class_indices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T17:01:42.042380Z","iopub.execute_input":"2025-09-17T17:01:42.043018Z","execution_failed":"2025-09-17T17:04:31.370Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dir = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/test\"\n\n\ntest_files = os.listdir(test_dir)\ntest_df = pd.DataFrame({\"filename\": test_files})\n\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\ntest_generator = test_datagen.flow_from_dataframe(\n    test_df,\n    directory=test_dir,\n    x_col=\"filename\",\n    y_col=None,          # no labels in test set\n    target_size=(224,224),\n    batch_size=32,\n    class_mode=None,\n    shuffle=False\n)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Visualize Random Training Images","metadata":{}},{"cell_type":"code","source":"activity_map = {'c0': 'Safe driving', \n                'c1': 'Texting - right', \n                'c2': 'Talking on the phone - right', \n                'c3': 'Texting - left', \n                'c4': 'Talking on the phone - left', \n                'c5': 'Operating the radio', \n                'c6': 'Drinking', \n                'c7': 'Reaching behind', \n                'c8': 'Hair and makeup', \n                'c9': 'Talking to passenger'}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T13:04:56.923252Z","iopub.execute_input":"2025-09-17T13:04:56.923487Z","iopub.status.idle":"2025-09-17T13:04:56.927511Z","shell.execute_reply.started":"2025-09-17T13:04:56.923470Z","shell.execute_reply":"2025-09-17T13:04:56.926729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images, labels = next(train_generator)\nplt.figure(figsize=(14, 14))\nfor i in range(9):\n    plt.subplot(3, 3, i+1)\n    plt.imshow(images[i])\n    label_index = np.argmax(labels[i])\n    class_code = list(train_generator.class_indices.keys())[label_index]\n    activity_name = activity_map[class_code]\n    plt.title(f\"{class_code}: {activity_name}\", fontsize=10)\n    plt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T13:04:56.928885Z","iopub.execute_input":"2025-09-17T13:04:56.929073Z","iopub.status.idle":"2025-09-17T13:04:58.902474Z","shell.execute_reply.started":"2025-09-17T13:04:56.929059Z","shell.execute_reply":"2025-09-17T13:04:58.901482Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Show Class Distribution","metadata":{}},{"cell_type":"code","source":"class_counts = {cls: len(os.listdir(os.path.join(train_dir, cls))) \n                for cls in os.listdir(train_dir)}\ndf_counts = pd.DataFrame.from_dict(class_counts, orient='index', columns=['count'])\ndf_counts.plot(kind='bar', figsize=(10,5), legend=False)\nplt.title(\"Number of Images per Class in Training Set\")\nplt.xlabel(\"Class\")\nplt.ylabel(\"Image Count\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T13:04:58.903631Z","iopub.execute_input":"2025-09-17T13:04:58.904411Z","iopub.status.idle":"2025-09-17T13:04:59.121678Z","shell.execute_reply.started":"2025-09-17T13:04:58.904383Z","shell.execute_reply":"2025-09-17T13:04:59.120974Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"VGG16","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import VGG16\nconv_base = VGG16(weights='imagenet',\n                  include_top=False,\n                  input_shape=(224, 224, 3))\n\nconv_base.trainable = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T13:07:52.941586Z","iopub.execute_input":"2025-09-17T13:07:52.942194Z","iopub.status.idle":"2025-09-17T13:07:53.193236Z","shell.execute_reply.started":"2025-09-17T13:07:52.942167Z","shell.execute_reply":"2025-09-17T13:07:53.192008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conv_base.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T13:07:54.834406Z","iopub.execute_input":"2025-09-17T13:07:54.834900Z","iopub.status.idle":"2025-09-17T13:07:54.857014Z","shell.execute_reply.started":"2025-09-17T13:07:54.834875Z","shell.execute_reply":"2025-09-17T13:07:54.856480Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"fine tuning","metadata":{}},{"cell_type":"code","source":"set_trainable = False\nfor layer in conv_base.layers:\n    if layer.name == 'block5_conv1':\n        set_trainable = True\n    if set_trainable:\n        layer.trainable = True\n    else:\n        layer.trainable = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T13:07:58.536225Z","iopub.execute_input":"2025-09-17T13:07:58.536464Z","iopub.status.idle":"2025-09-17T13:07:58.540827Z","shell.execute_reply.started":"2025-09-17T13:07:58.536447Z","shell.execute_reply":"2025-09-17T13:07:58.540019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conv_base.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T13:08:00.861956Z","iopub.execute_input":"2025-09-17T13:08:00.862420Z","iopub.status.idle":"2025-09-17T13:08:00.884363Z","shell.execute_reply.started":"2025-09-17T13:08:00.862397Z","shell.execute_reply":"2025-09-17T13:08:00.883741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_1 = models.Sequential()\nmodel_1.add(conv_base)\nmodel_1.add(layers.Flatten())\nmodel_1.add(layers.Dense(512, activation='relu'))\nmodel_1.add(layers.Dense(64, activation='relu'))\nmodel_1.add(Dense(10, activation='softmax'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T13:08:04.096956Z","iopub.execute_input":"2025-09-17T13:08:04.097195Z","iopub.status.idle":"2025-09-17T13:08:04.133065Z","shell.execute_reply.started":"2025-09-17T13:08:04.097177Z","shell.execute_reply":"2025-09-17T13:08:04.132539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_1.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T13:08:06.041227Z","iopub.execute_input":"2025-09-17T13:08:06.041778Z","iopub.status.idle":"2025-09-17T13:08:06.055590Z","shell.execute_reply.started":"2025-09-17T13:08:06.041750Z","shell.execute_reply":"2025-09-17T13:08:06.055008Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"callbacks","metadata":{}},{"cell_type":"code","source":"early_stop = EarlyStopping(\n    monitor=\"val_loss\",\n    patience=5,\n    restore_best_weights=True,\n    verbose=1\n)\ncheckpoint = ModelCheckpoint(\n    filepath=\"model1/best_model.weights.h5\",\n    monitor=\"val_loss\",\n    save_best_only=True,\n    save_weights_only=True,          \n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T13:08:08.205202Z","iopub.execute_input":"2025-09-17T13:08:08.205917Z","iopub.status.idle":"2025-09-17T13:08:08.209754Z","shell.execute_reply.started":"2025-09-17T13:08:08.205884Z","shell.execute_reply":"2025-09-17T13:08:08.209030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import optimizers\nmodel_1.compile(loss='categorical_crossentropy',\n              optimizer=RMSprop(learning_rate=1e-5),\n              metrics=['acc'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T13:08:10.199163Z","iopub.execute_input":"2025-09-17T13:08:10.199422Z","iopub.status.idle":"2025-09-17T13:08:10.208344Z","shell.execute_reply.started":"2025-09-17T13:08:10.199401Z","shell.execute_reply":"2025-09-17T13:08:10.207528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_1 = model_1.fit(\n    train_generator,\n    steps_per_epoch=len(train_generator), \n    validation_data=val_generator,\n    validation_steps=len(val_generator),         \n    epochs=30,\n    callbacks=[early_stop, checkpoint]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T13:08:12.978083Z","iopub.execute_input":"2025-09-17T13:08:12.978347Z","iopub.status.idle":"2025-09-17T14:27:12.313978Z","shell.execute_reply.started":"2025-09-17T13:08:12.978326Z","shell.execute_reply":"2025-09-17T14:27:12.313371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"acc = history_1.history['acc']\nval_acc = history_1.history['val_acc']\nloss = history_1.history['loss']\nval_loss = history_1.history['val_loss']\n\nepochs = range(len(acc))\n\nplt.plot(epochs, acc, 'r', label='Training acc')\nplt.plot(epochs, val_acc, 'b', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.legend()\n\nplt.figure()\n\nplt.plot(epochs, loss, 'r', label='Training loss')\nplt.plot(epochs, val_loss, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.legend()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T14:27:20.758465Z","iopub.execute_input":"2025-09-17T14:27:20.759041Z","iopub.status.idle":"2025-09-17T14:27:21.093373Z","shell.execute_reply.started":"2025-09-17T14:27:20.759015Z","shell.execute_reply":"2025-09-17T14:27:21.092750Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_1.load_weights(\"model1/best_model.weights.h5\")\npreds=model_1.predict(test_generator, verbose=1)  # shape: (79726, 10)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-09-17T17:04:31.371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filenames = test_generator.filenames\nfilenames = [f.split('/')[-1] for f in filenames] \n\n\nsubmission = pd.DataFrame(preds, columns=[f'c{i}' for i in range(10)])\nsubmission.insert(0, 'img', filenames)\n\nsubmission.to_csv(\"submission.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T15:27:02.488110Z","iopub.execute_input":"2025-09-17T15:27:02.488385Z","iopub.status.idle":"2025-09-17T15:27:03.412487Z","shell.execute_reply.started":"2025-09-17T15:27:02.488363Z","shell.execute_reply":"2025-09-17T15:27:03.411663Z"}},"outputs":[],"execution_count":null}]}