{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":5337,"sourceType":"datasetVersion","datasetId":3258},{"sourceId":140900240,"sourceType":"kernelVersion"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout, Input\nfrom tensorflow.keras.utils import to_categorical","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:08:37.079554Z","iopub.execute_input":"2025-06-09T06:08:37.080048Z","iopub.status.idle":"2025-06-09T06:08:49.378255Z","shell.execute_reply.started":"2025-06-09T06:08:37.080004Z","shell.execute_reply":"2025-06-09T06:08:49.377678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/sign-language-mnist/sign_mnist_train.csv')\ntest_df = pd.read_csv('/kaggle/input/sign-language-mnist/sign_mnist_test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:08:53.261738Z","iopub.execute_input":"2025-06-09T06:08:53.262302Z","iopub.status.idle":"2025-06-09T06:08:57.282619Z","shell.execute_reply.started":"2025-06-09T06:08:53.262271Z","shell.execute_reply":"2025-06-09T06:08:57.282073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = train_df.iloc[:, 1:].values\ny_train = train_df.iloc[:, 0].values\nX_test = test_df.iloc[:, 1:].values\ny_test = test_df.iloc[:, 0].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:08:59.647289Z","iopub.execute_input":"2025-06-09T06:08:59.647557Z","iopub.status.idle":"2025-06-09T06:08:59.652535Z","shell.execute_reply.started":"2025-06-09T06:08:59.647536Z","shell.execute_reply":"2025-06-09T06:08:59.651723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = X_train / 255.0\nX_test = X_test / 255.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:09:02.625273Z","iopub.execute_input":"2025-06-09T06:09:02.625534Z","iopub.status.idle":"2025-06-09T06:09:02.709326Z","shell.execute_reply.started":"2025-06-09T06:09:02.625513Z","shell.execute_reply":"2025-06-09T06:09:02.708617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = X_train.reshape(-1, 28, 28)\nX_test = X_test.reshape(-1, 28, 28)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:09:04.063712Z","iopub.execute_input":"2025-06-09T06:09:04.064520Z","iopub.status.idle":"2025-06-09T06:09:04.067763Z","shell.execute_reply.started":"2025-06-09T06:09:04.064492Z","shell.execute_reply":"2025-06-09T06:09:04.067161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if y_train.ndim == 3:\n    y_train = np.argmax(y_train, axis=-1)\n    y_test = np.argmax(y_test, axis=-1)\n\n# 정수형인 경우에만 인코딩 적용\nif y_train.ndim == 1:\n    y_train = to_categorical(y_train, num_classes=25)\n    y_test = to_categorical(y_test, num_classes=25)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:09:06.885524Z","iopub.execute_input":"2025-06-09T06:09:06.886292Z","iopub.status.idle":"2025-06-09T06:09:06.891769Z","shell.execute_reply.started":"2025-06-09T06:09:06.886264Z","shell.execute_reply":"2025-06-09T06:09:06.891065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Sequential([\n    Input(shape=(28, 28)),            \n    LSTM(128, return_sequences=False),\n    Dropout(0.3),\n    Dense(64, activation='relu'),\n    Dropout(0.3),\n    Dense(25, activation='softmax')   \n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:09:10.946743Z","iopub.execute_input":"2025-06-09T06:09:10.947497Z","iopub.status.idle":"2025-06-09T06:09:13.188960Z","shell.execute_reply.started":"2025-06-09T06:09:10.947466Z","shell.execute_reply":"2025-06-09T06:09:13.188424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:09:16.362534Z","iopub.execute_input":"2025-06-09T06:09:16.362900Z","iopub.status.idle":"2025-06-09T06:09:16.379287Z","shell.execute_reply.started":"2025-06-09T06:09:16.362869Z","shell.execute_reply":"2025-06-09T06:09:16.378397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=10, batch_size=64, validation_split=0.2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:09:18.197493Z","iopub.execute_input":"2025-06-09T06:09:18.198216Z","iopub.status.idle":"2025-06-09T06:09:44.200947Z","shell.execute_reply.started":"2025-06-09T06:09:18.198177Z","shell.execute_reply":"2025-06-09T06:09:44.200227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_loss, test_acc = model.evaluate(X_test, y_test)\nprint(f'테스트 정확도: {test_acc:.4f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:09:47.112468Z","iopub.execute_input":"2025-06-09T06:09:47.112743Z","iopub.status.idle":"2025-06-09T06:09:47.973893Z","shell.execute_reply.started":"2025-06-09T06:09:47.112723Z","shell.execute_reply":"2025-06-09T06:09:47.973338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Train Acc')\nplt.plot(history.history['val_accuracy'], label='Val Acc')\nplt.title(\"Accuracy change\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Accuracy\")\nplt.legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:10:58.900323Z","iopub.execute_input":"2025-06-09T06:10:58.900618Z","iopub.status.idle":"2025-06-09T06:10:59.156424Z","shell.execute_reply.started":"2025-06-09T06:10:58.900599Z","shell.execute_reply":"2025-06-09T06:10:59.155766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Val Loss')\nplt.title(\"Loss over Epochs\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Loss\")\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:11:02.343486Z","iopub.execute_input":"2025-06-09T06:11:02.343959Z","iopub.status.idle":"2025-06-09T06:11:02.506953Z","shell.execute_reply.started":"2025-06-09T06:11:02.343936Z","shell.execute_reply":"2025-06-09T06:11:02.506401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\nnum_samples = 10\n\nindices = np.random.choice(len(X_test), num_samples, replace=False)\nsample_images = X_test[indices]\nsample_labels = np.argmax(y_test[indices], axis=1)\n\npredictions = model.predict(sample_images)\npredicted_labels = np.argmax(predictions, axis=1)\n\nfig, axes = plt.subplots(1, num_samples, figsize=(15, 3))\nfor i, ax in enumerate(axes):\n    ax.imshow(sample_images[i], cmap='gray')\n    ax.axis('off')\n    ax.set_title(f\"predict: {predicted_labels[i]}\\nanswer: {sample_labels[i]}\", fontsize=9)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:11:08.379721Z","iopub.execute_input":"2025-06-09T06:11:08.380222Z","iopub.status.idle":"2025-06-09T06:11:09.002304Z","shell.execute_reply.started":"2025-06-09T06:11:08.380200Z","shell.execute_reply":"2025-06-09T06:11:09.001539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.countplot(x=train_df['label'])\nplt.title(\"Sign Language MNIST Label Distribution\")\nplt.xlabel(\"Class (0=A, ..., 24=Z excluding J)\")\nplt.ylabel(\"Count\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:11:13.734044Z","iopub.execute_input":"2025-06-09T06:11:13.734323Z","iopub.status.idle":"2025-06-09T06:11:13.953694Z","shell.execute_reply.started":"2025-06-09T06:11:13.734302Z","shell.execute_reply":"2025-06-09T06:11:13.953075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\nsamples = train_df.sample(10)\nfig, axes = plt.subplots(1, 10, figsize=(15, 3))\nfor i, (idx, row) in enumerate(samples.iterrows()):\n    image = row[1:].values.reshape(28, 28)\n    label = row[0]\n    axes[i].imshow(image, cmap='gray')\n    axes[i].axis('off')\n    axes[i].set_title(label)\nplt.suptitle(\"Random Sign Language Images\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-09T06:11:17.161286Z","iopub.execute_input":"2025-06-09T06:11:17.161880Z","iopub.status.idle":"2025-06-09T06:11:17.531615Z","shell.execute_reply.started":"2025-06-09T06:11:17.161860Z","shell.execute_reply":"2025-06-09T06:11:17.530918Z"}},"outputs":[],"execution_count":null}]}