{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4829,"databundleVersionId":44847,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#ในkaggle\n#data inputได้แก่\n#/kaggle/input/yelp-restaurant-photo-classification/sample_submission.csv.tgz\n#/kaggle/input/yelp-restaurant-photo-classification/test_photo_to_biz.csv.tgz\n#/kaggle/input/yelp-restaurant-photo-classification/test_photos.tgz\n#/kaggle/input/yelp-restaurant-photo-classification/train.csv.tgz\n#/kaggle/input/yelp-restaurant-photo-classification/train_photo_to_biz_ids.csv.tgz\n#/kaggle/input/yelp-restaurant-photo-classification/train_photos.tgz\n#data outยังไม่มี\n\n#1. ต้องการclean data ใน test_photos.tgz ก่อน ที่จะคัดลอกไปที่ /kaggle/working/test_photos\n#2. ต้องการไฟล์รูปภาพที่เป็นนามสกุล jpg และขึ้นต้นด้วยตัวเลขจำนวน 100 ไฟล์\n#3. ต้องการไฟล์ที่ Pillow (PIL) สามารถจำแนกได้\n#4.ต้องการโค้ดที่เมื่อไฟล์ที่ต้องการเข้าเงื่อนไขและได้จำนวนครบแล้วก็หยุดไม่ต้องไปรันเพื่อเช็คที่ไฟล์ทั้งหมดใน test_photos.tgz\n#5. ต้องการการเช็คำจำนวนไฟล์ที่ได้มา และรายชื่อไฟล์ที่ได้มา\n#6. เมื่อต้องการแสดงรูปภาพทั้งหมดด้วย plt.show() ได้\n#7.ต้องไม่เป็นไฟล์ที่ขึ้นต้นด้วย ._ ต้องขึ้นต้นด้วยตัวเลขและเป็นนามสกุล jpgเท่านั้น","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:29:37.711601Z","iopub.execute_input":"2023-11-22T21:29:37.712052Z","iopub.status.idle":"2023-11-22T21:29:37.718895Z","shell.execute_reply.started":"2023-11-22T21:29:37.712020Z","shell.execute_reply":"2023-11-22T21:29:37.717781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import shutil\n\n# ลบไดเรกทอรี '/kaggle/working/test_photos' ทั้งหมด\n#shutil.rmtree('/kaggle/working/test_photos')\n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:29:37.721014Z","iopub.execute_input":"2023-11-22T21:29:37.721762Z","iopub.status.idle":"2023-11-22T21:29:37.738637Z","shell.execute_reply.started":"2023-11-22T21:29:37.721721Z","shell.execute_reply":"2023-11-22T21:29:37.737466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport tarfile\nimport shutil\n\n# ระบุที่อยู่ของไฟล์ .tgz\ninput_path = '/kaggle/input/yelp-restaurant-photo-classification/test_photos.tgz'\n\n# ระบุที่อยู่ที่ต้องการแตกไฟล์ไป\noutput_path = '/kaggle/working'\n\n# สร้างไดเรกทอรีที่ต้องการเก็บไฟล์ที่แตกออกมา\nos.makedirs(output_path, exist_ok=True)\n\n# กำหนดจำนวนไฟล์ที่ต้องการดึงมา (ตัวอย่างนี้คือ 10 ไฟล์)\nnum_files_to_extract = 1300\n\n# เปิดไฟล์ .tgz และดึงออกมาเพียงบางส่วน\nwith tarfile.open(input_path, 'r:gz') as tar:\n    extracted_files = 0\n    for member in tar.getmembers():\n        if member.name.lower().endswith('.jpg') and not member.name.startswith('._'):\n            tar.extract(member, output_path)\n            extracted_files += 1\n            if extracted_files >= num_files_to_extract:\n                break\n\ndirectory_path = '/kaggle/working/test_photos'\n\n# แสดงไฟล์ทั้งหมดในไดเรกทอรี\nfiles = os.listdir(directory_path)\n\n# แสดงชื่อไฟล์\nprint(len(files))","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:41:15.556977Z","iopub.execute_input":"2023-11-22T21:41:15.557564Z","iopub.status.idle":"2023-11-22T21:43:11.735179Z","shell.execute_reply.started":"2023-11-22T21:41:15.557526Z","shell.execute_reply":"2023-11-22T21:43:11.734281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# ระบุที่อยู่ของไดเรกทอรีที่ต้องการนับ\ndirectory_path = '/kaggle/working/test_photos'\n\n# นับจำนวนไฟล์ที่ขึ้นต้นด้วยตัวเลขและลงท้ายด้วย .jpg\ncount = 0\nfor file in os.listdir(directory_path):\n    if file[:-4].isdigit() and file.lower().endswith('.jpg'):\n        count += 1\n\n# แสดงจำนวนไฟล์ที่ตรงเงื่อนไข\nprint(f\"จำนวนไฟล์ที่ขึ้นต้นด้วยตัวเลขและลงท้ายด้วย .jpg: {count}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:43:11.737142Z","iopub.execute_input":"2023-11-22T21:43:11.737678Z","iopub.status.idle":"2023-11-22T21:43:11.746682Z","shell.execute_reply.started":"2023-11-22T21:43:11.737647Z","shell.execute_reply":"2023-11-22T21:43:11.745540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\n# ระบุที่อยู่ของไดเรกทอรีที่มีไฟล์รูปภาพ\nimage_directory = '/kaggle/working/test_photos'\n\n# ดึงรายชื่อไฟล์รูปภาพทั้งหมดที่มีชื่อขึ้นต้นด้วยตัวเลขและลงท้ายด้วย .jpg\nimage_files = [file for file in os.listdir(image_directory) if file[:-4].isdigit() and file.lower().endswith('.jpg')]\n\n# กำหนดขนาดของตารางรูปภาพ\nnum_rows = 3\nnum_cols = 3\n\n# สร้าง subplot\nfig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 15))\n\n# แสดงรูปภาพทั้งหมดใน subplot\nfor i in range(num_rows):\n    for j in range(num_cols):\n        index = i * num_cols + j\n        if index < len(image_files):\n            image_file = image_files[index]\n            image_path = os.path.join(image_directory, image_file)\n            img = mpimg.imread(image_path)\n            axes[i, j].imshow(img)\n            axes[i, j].axis('off')\n\n# ปรับระยะห่างของ subplot\nplt.subplots_adjust(wspace=0.2, hspace=0.5)\n\n# แสดง subplot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:48:32.709527Z","iopub.execute_input":"2023-11-22T21:48:32.709974Z","iopub.status.idle":"2023-11-22T21:48:33.898539Z","shell.execute_reply.started":"2023-11-22T21:48:32.709941Z","shell.execute_reply":"2023-11-22T21:48:33.897073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\n# ระบุที่อยู่ของไดเรกทอรีที่มีไฟล์รูปภาพ\nimage_directory = '/kaggle/working/test_photos'\n\n# ดึงรายชื่อไฟล์รูปภาพทั้งหมดที่มีชื่อขึ้นต้นด้วยตัวเลขและลงท้ายด้วย .jpg\nimage_files = [file for file in os.listdir(image_directory) if file[:-4].isdigit() and file.lower().endswith('.jpg')]\n\n# กำหนดขนาดของตารางรูปภาพ\nnum_rows = 25\nnum_cols = 25\n\n# สร้าง subplot\nfig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 15))\n\n# แสดงรูปภาพทั้งหมดใน subplot\nfor i in range(num_rows):\n    for j in range(num_cols):\n        index = i * num_cols + j\n        if index < len(image_files):\n            image_file = image_files[index]\n            image_path = os.path.join(image_directory, image_file)\n            img = mpimg.imread(image_path)\n            axes[i, j].imshow(img)\n            axes[i, j].axis('off')\n\n# ปรับระยะห่างของ subplot\nplt.subplots_adjust(wspace=0.2, hspace=0.5)\n\n# แสดง subplot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:43:12.228372Z","iopub.execute_input":"2023-11-22T21:43:12.228731Z","iopub.status.idle":"2023-11-22T21:43:50.425257Z","shell.execute_reply.started":"2023-11-22T21:43:12.228699Z","shell.execute_reply":"2023-11-22T21:43:50.423240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport tarfile\nimport shutil\n\n# ระบุที่อยู่ของไฟล์ .tgz\ninput_path = '/kaggle/input/yelp-restaurant-photo-classification/train_photos.tgz'\n\n# ระบุที่อยู่ที่ต้องการแตกไฟล์ไป\noutput_path = '/kaggle/working'\n\n# สร้างไดเรกทอรีที่ต้องการเก็บไฟล์ที่แตกออกมา\nos.makedirs(output_path, exist_ok=True)\n\n# กำหนดจำนวนไฟล์ที่ต้องการดึงมา (ตัวอย่างนี้คือ 10 ไฟล์)\nnum_files_to_extract = 1300\n\n# เปิดไฟล์ .tgz และดึงออกมาเพียงบางส่วน\nwith tarfile.open(input_path, 'r:gz') as tar:\n    extracted_files = 0\n    for member in tar.getmembers():\n        if member.name.lower().endswith('.jpg') and not member.name.startswith('._'):\n            tar.extract(member, output_path)\n            extracted_files += 1\n            if extracted_files >= num_files_to_extract:\n                break\n\ndirectory_path = '/kaggle/working/train_photos'\n\n# แสดงไฟล์ทั้งหมดในไดเรกทอรี\nfiles = os.listdir(directory_path)\n\n# แสดงชื่อไฟล์\nprint(len(files))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:31:58.574524Z","iopub.execute_input":"2023-11-22T21:31:58.574994Z","iopub.status.idle":"2023-11-22T21:34:14.999112Z","shell.execute_reply.started":"2023-11-22T21:31:58.574955Z","shell.execute_reply":"2023-11-22T21:34:14.998279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# ระบุที่อยู่ของไดเรกทอรีที่ต้องการนับ\ndirectory_path = '/kaggle/working/train_photos'\n\n# นับจำนวนไฟล์ที่ขึ้นต้นด้วยตัวเลขและลงท้ายด้วย .jpg\ncount = 0\nfor file in os.listdir(directory_path):\n    if file[:-4].isdigit() and file.lower().endswith('.jpg'):\n        count += 1\n\n# แสดงจำนวนไฟล์ที่ตรงเงื่อนไข\nprint(f\"จำนวนไฟล์ที่ขึ้นต้นด้วยตัวเลขและลงท้ายด้วย .jpg: {count}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:34:15.000493Z","iopub.execute_input":"2023-11-22T21:34:15.001041Z","iopub.status.idle":"2023-11-22T21:34:15.014887Z","shell.execute_reply.started":"2023-11-22T21:34:15.001010Z","shell.execute_reply":"2023-11-22T21:34:15.013446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\n# ระบุที่อยู่ของไดเรกทอรีที่มีไฟล์รูปภาพ\nimage_directory = '/kaggle/working/train_photos'\n\n# ดึงรายชื่อไฟล์รูปภาพทั้งหมดที่มีชื่อขึ้นต้นด้วยตัวเลขและลงท้ายด้วย .jpg\nimage_files = [file for file in os.listdir(image_directory) if file[:-4].isdigit() and file.lower().endswith('.jpg')]\n\n# กำหนดขนาดของตารางรูปภาพ\nnum_rows = 25\nnum_cols = 25\n\n# สร้าง subplot\nfig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 15))\n\n# แสดงรูปภาพทั้งหมดใน subplot\nfor i in range(num_rows):\n    for j in range(num_cols):\n        index = i * num_cols + j\n        if index < len(image_files):\n            image_file = image_files[index]\n            image_path = os.path.join(image_directory, image_file)\n            img = mpimg.imread(image_path)\n            axes[i, j].imshow(img)\n            axes[i, j].axis('off')\n\n# ปรับระยะห่างของ subplot\nplt.subplots_adjust(wspace=0.2, hspace=0.5)\n\n# แสดง subplot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:34:15.016475Z","iopub.execute_input":"2023-11-22T21:34:15.016835Z","iopub.status.idle":"2023-11-22T21:34:50.612284Z","shell.execute_reply.started":"2023-11-22T21:34:15.016805Z","shell.execute_reply":"2023-11-22T21:34:50.609776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# โหลด CSV เข้าสู่ DataFrame\nfile_path = '/kaggle/input/yelp-restaurant-photo-classification/sample_submission.csv.tgz'\ndf = pd.read_csv(file_path, compression='gzip')\n# ทำการแก้ไขข้อมูลใน DataFrame ตามที่ต้องการ\n\n# บันทึก DataFrame กลับเป็นไฟล์ CSV\ndf.to_csv('/kaggle/working/sample_submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:34:50.616291Z","iopub.execute_input":"2023-11-22T21:34:50.616697Z","iopub.status.idle":"2023-11-22T21:34:51.120743Z","shell.execute_reply.started":"2023-11-22T21:34:50.616666Z","shell.execute_reply":"2023-11-22T21:34:51.119665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# โหลด CSV เข้าสู่ DataFrame\nfile_path = '/kaggle/input/yelp-restaurant-photo-classification/test_photo_to_biz.csv.tgz'\ndf = pd.read_csv(file_path, compression='gzip')\n# ทำการแก้ไขข้อมูลใน DataFrame ตามที่ต้องการ\n\n# บันทึก DataFrame กลับเป็นไฟล์ CSV\ndf.to_csv('/kaggle/working/test_photo_to_biz.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:34:51.122143Z","iopub.execute_input":"2023-11-22T21:34:51.122544Z","iopub.status.idle":"2023-11-22T21:34:55.351356Z","shell.execute_reply.started":"2023-11-22T21:34:51.122513Z","shell.execute_reply":"2023-11-22T21:34:55.349965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# โหลด CSV เข้าสู่ DataFrame\nfile_path = '/kaggle/input/yelp-restaurant-photo-classification/train.csv.tgz'\ndf = pd.read_csv(file_path, compression='gzip', encoding='latin1')\n# ทำการแก้ไขข้อมูลใน DataFrame ตามที่ต้องการ\n\n# บันทึก DataFrame กลับเป็นไฟล์ CSV\ndf.to_csv('/kaggle/working/train.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:34:55.353376Z","iopub.execute_input":"2023-11-22T21:34:55.353752Z","iopub.status.idle":"2023-11-22T21:34:55.383616Z","shell.execute_reply.started":"2023-11-22T21:34:55.353717Z","shell.execute_reply":"2023-11-22T21:34:55.382278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# โหลด CSV เข้าสู่ DataFrame\nfile_path = '/kaggle/input/yelp-restaurant-photo-classification/train_photo_to_biz_ids.csv.tgz'\ndf = pd.read_csv(file_path, compression='gzip', encoding='latin1')\n# ทำการแก้ไขข้อมูลใน DataFrame ตามที่ต้องการ\n\n# บันทึก DataFrame กลับเป็นไฟล์ CSV\ndf.to_csv('/kaggle/working/train_photo_to_biz_ids.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-22T21:34:55.385131Z","iopub.execute_input":"2023-11-22T21:34:55.385535Z","iopub.status.idle":"2023-11-22T21:34:56.417263Z","shell.execute_reply.started":"2023-11-22T21:34:55.385504Z","shell.execute_reply":"2023-11-22T21:34:56.416229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}