{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nfrom wordcloud import WordCloud, STOPWORDS\nfrom PIL import Image\nimport random\n#Text Color\nfrom termcolor import colored\n\n#NLP\nfrom sklearn.feature_extraction.text import CountVectorizer\n\n#WordCloud\nfrom wordcloud import WordCloud, STOPWORDS\n\n#Text Processing\nimport re\nimport nltk","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:51:20.628282Z","iopub.execute_input":"2022-08-13T18:51:20.628649Z","iopub.status.idle":"2022-08-13T18:51:20.635238Z","shell.execute_reply.started":"2022-08-13T18:51:20.628612Z","shell.execute_reply":"2022-08-13T18:51:20.634278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/shopee-product-matching/train.csv')\ntest = pd.read_csv('../input/shopee-product-matching/test.csv')\nsample = pd.read_csv('../input/shopee-product-matching/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:51:22.493540Z","iopub.execute_input":"2022-08-13T18:51:22.493901Z","iopub.status.idle":"2022-08-13T18:51:22.692149Z","shell.execute_reply.started":"2022-08-13T18:51:22.493870Z","shell.execute_reply":"2022-08-13T18:51:22.691147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of images in train dataset:\",len(train))\nprint(\"Number of images in test dataset:\",len(test))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T19:32:08.991976Z","iopub.execute_input":"2022-08-13T19:32:08.992372Z","iopub.status.idle":"2022-08-13T19:32:08.999924Z","shell.execute_reply.started":"2022-08-13T19:32:08.992337Z","shell.execute_reply":"2022-08-13T19:32:08.998900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Image Folder Paths\ntrain_jpg_directory = '../input/shopee-product-matching/train_images'\ntest_jpg_directory = '../input/shopee-product-matching/test_images'","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:51:25.372146Z","iopub.execute_input":"2022-08-13T18:51:25.373185Z","iopub.status.idle":"2022-08-13T18:51:25.378013Z","shell.execute_reply.started":"2022-08-13T18:51:25.373142Z","shell.execute_reply":"2022-08-13T18:51:25.377086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def getImagePaths(path):\n    \"\"\"\n    Function to Combine Directory Path with individual Image Paths\n    \n    parameters: path(string) - Path of directory\n    returns: image_names(string) - Full Image Path\n    \"\"\"\n    image_names = []\n    for dirname, _, filenames in os.walk(path):\n        for filename in filenames:\n            fullpath = os.path.join(dirname, filename)\n            image_names.append(fullpath)\n    return image_names","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:51:27.133681Z","iopub.execute_input":"2022-08-13T18:51:27.134383Z","iopub.status.idle":"2022-08-13T18:51:27.140991Z","shell.execute_reply.started":"2022-08-13T18:51:27.134347Z","shell.execute_reply":"2022-08-13T18:51:27.139638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Get complete image paths for train and test datasets\ntrain_images_path = getImagePaths(train_jpg_directory)\ntest_images_path = getImagePaths(test_jpg_directory)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:51:28.702213Z","iopub.execute_input":"2022-08-13T18:51:28.703272Z","iopub.status.idle":"2022-08-13T18:51:41.055064Z","shell.execute_reply.started":"2022-08-13T18:51:28.703206Z","shell.execute_reply":"2022-08-13T18:51:41.054083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images_path","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:13.787740Z","iopub.execute_input":"2022-08-13T18:28:13.788811Z","iopub.status.idle":"2022-08-13T18:28:13.799760Z","shell.execute_reply.started":"2022-08-13T18:28:13.788751Z","shell.execute_reply":"2022-08-13T18:28:13.798756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:13.801206Z","iopub.execute_input":"2022-08-13T18:28:13.802369Z","iopub.status.idle":"2022-08-13T18:28:13.830318Z","shell.execute_reply.started":"2022-08-13T18:28:13.802330Z","shell.execute_reply":"2022-08-13T18:28:13.828830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:17.422658Z","iopub.execute_input":"2022-08-13T18:28:17.423072Z","iopub.status.idle":"2022-08-13T18:28:17.435329Z","shell.execute_reply.started":"2022-08-13T18:28:17.423040Z","shell.execute_reply":"2022-08-13T18:28:17.434408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking missing data\ntrain.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:18.388756Z","iopub.execute_input":"2022-08-13T18:28:18.389214Z","iopub.status.idle":"2022-08-13T18:28:18.407977Z","shell.execute_reply.started":"2022-08-13T18:28:18.389180Z","shell.execute_reply":"2022-08-13T18:28:18.406697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:19.244988Z","iopub.execute_input":"2022-08-13T18:28:19.245474Z","iopub.status.idle":"2022-08-13T18:28:19.253522Z","shell.execute_reply.started":"2022-08-13T18:28:19.245412Z","shell.execute_reply":"2022-08-13T18:28:19.252000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Column-wise unique values\nfor col in train.columns:\n    print(col + \":\" + colored(str(len(train[col].unique()))))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:20.096468Z","iopub.execute_input":"2022-08-13T18:28:20.096954Z","iopub.status.idle":"2022-08-13T18:28:20.140256Z","shell.execute_reply.started":"2022-08-13T18:28:20.096915Z","shell.execute_reply":"2022-08-13T18:28:20.139159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to display images:\ndef display_multiple_img(images_paths, rows, cols):\n    \"\"\"\n    Function to Display Images from Dataset.\n    \n    parameters: images_path(string) - Paths of Images to be displayed\n                rows(int) - No. of Rows in Output\n                cols(int) - No. of Columns in Output\n    \"\"\"\n    figure, ax = plt.subplots(nrows=rows,ncols=cols,figsize=(16,8) )\n    for ind,image_path in enumerate(images_paths):\n        image=cv2.imread(image_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB) \n        try:\n            ax.ravel()[ind].imshow(image)\n            ax.ravel()[ind].set_axis_off()\n        except:\n            continue;\n    plt.tight_layout()\n    plt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:21.020401Z","iopub.execute_input":"2022-08-13T18:28:21.020913Z","iopub.status.idle":"2022-08-13T18:28:21.029841Z","shell.execute_reply.started":"2022-08-13T18:28:21.020876Z","shell.execute_reply":"2022-08-13T18:28:21.028477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_multiple_img(train_images_path[10:30], 3,5)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:22.242646Z","iopub.execute_input":"2022-08-13T18:28:22.243113Z","iopub.status.idle":"2022-08-13T18:28:24.734286Z","shell.execute_reply.started":"2022-08-13T18:28:22.243077Z","shell.execute_reply":"2022-08-13T18:28:24.733376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_multiple_img(test_images_path, 1, 3)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:24.735889Z","iopub.execute_input":"2022-08-13T18:28:24.737193Z","iopub.status.idle":"2022-08-13T18:28:25.396337Z","shell.execute_reply.started":"2022-08-13T18:28:24.737155Z","shell.execute_reply":"2022-08-13T18:28:25.395130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top10 = pd.DataFrame(train.label_group.value_counts().head(10))\ntop10.reset_index(inplace=True)\ntop10.columns = ['label_group','count']\ntop10","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:25.397850Z","iopub.execute_input":"2022-08-13T18:28:25.398196Z","iopub.status.idle":"2022-08-13T18:28:25.415633Z","shell.execute_reply.started":"2022-08-13T18:28:25.398164Z","shell.execute_reply":"2022-08-13T18:28:25.414564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the most frequent landmark_ids\ntop20 = pd.DataFrame(train.label_group.value_counts().head(20))\ntop20.reset_index(inplace=True)\ntop20.columns = ['label_group','count']\ntop20\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nplt.figure(figsize = (22, 8))\nplt.title('Most Frequent Landmarks')\nsns.set_color_codes(\"muted\")\nsns.barplot(x=\"label_group\", y=\"count\", data=top20, label=\"Count\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:25.750004Z","iopub.execute_input":"2022-08-13T18:28:25.750909Z","iopub.status.idle":"2022-08-13T18:28:26.123071Z","shell.execute_reply.started":"2022-08-13T18:28:25.750869Z","shell.execute_reply":"2022-08-13T18:28:26.121774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bot10 = pd.DataFrame(train.label_group.value_counts().tail(10))\nbot10.reset_index(inplace=True)\nbot10.columns = ['label_group','count']\nbot10","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:26.875684Z","iopub.execute_input":"2022-08-13T18:28:26.876225Z","iopub.status.idle":"2022-08-13T18:28:26.894871Z","shell.execute_reply.started":"2022-08-13T18:28:26.876182Z","shell.execute_reply":"2022-08-13T18:28:26.893517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the least frequent products\nbot20 = pd.DataFrame(train.label_group.value_counts().tail(20))\nbot20.reset_index(inplace=True)\nbot20.columns = ['label_group','count']\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nplt.figure(figsize = (22, 8))\nplt.title('Most Frequent Landmarks')\nsns.set_color_codes(\"muted\")\nsns.barplot(x=\"label_group\", y=\"count\", data=bot20, label=\"Count\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:28.043594Z","iopub.execute_input":"2022-08-13T18:28:28.044536Z","iopub.status.idle":"2022-08-13T18:28:28.432091Z","shell.execute_reply.started":"2022-08-13T18:28:28.044483Z","shell.execute_reply":"2022-08-13T18:28:28.431137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Number of products with more than 10 images:\ncounts = train['label_group'].value_counts().sort_values(ascending=False)\nabove10 = counts[counts >10].index.shape[0]\nabove10","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:29.562932Z","iopub.execute_input":"2022-08-13T18:28:29.563695Z","iopub.status.idle":"2022-08-13T18:28:29.577911Z","shell.execute_reply.started":"2022-08-13T18:28:29.563639Z","shell.execute_reply":"2022-08-13T18:28:29.576512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Number of lproducts with more than 20 images:\ncounts = train['label_group'].value_counts().sort_values(ascending=False)\nabove10 = counts[counts >20].index.shape[0]\nabove10","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:30.674917Z","iopub.execute_input":"2022-08-13T18:28:30.675786Z","iopub.status.idle":"2022-08-13T18:28:30.689903Z","shell.execute_reply.started":"2022-08-13T18:28:30.675719Z","shell.execute_reply":"2022-08-13T18:28:30.688465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set()\nplt.title('Training set: number of images per class(line plot)')\nlandmarks_fold = pd.DataFrame(train['label_group'].value_counts())\nlandmarks_fold.reset_index(inplace=True)\nlandmarks_fold.columns = ['landmark_id','count']\nax = landmarks_fold['count'].plot(logy=True, grid=True)\nlocs, labels = plt.xticks()\nplt.setp(labels, rotation=30)\nax.set(xlabel=\"Products\", ylabel=\"Number of images\")","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:31.683405Z","iopub.execute_input":"2022-08-13T18:28:31.684880Z","iopub.status.idle":"2022-08-13T18:28:32.552935Z","shell.execute_reply.started":"2022-08-13T18:28:31.684830Z","shell.execute_reply":"2022-08-13T18:28:32.551515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Adding file paths to train and test datasets:\ndef get_train_file_path(image_id):\n    return \"../input/shopee-product-matching/train_images/{}\".format(image_id)\ntrain['file_path'] = train['image'].apply(get_train_file_path)\n\n\ndef get_test_file_path(image_id):\n    return \"../input/shopee-product-matching/test_images/{}\".format(image_id)\ntest['file_path'] = test['image'].apply(get_train_file_path)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:32.916845Z","iopub.execute_input":"2022-08-13T18:28:32.918203Z","iopub.status.idle":"2022-08-13T18:28:32.948249Z","shell.execute_reply.started":"2022-08-13T18:28:32.918138Z","shell.execute_reply":"2022-08-13T18:28:32.946922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets examine product with label_group is 994676122 which is having highest count :\nimport os\nimport glob\nimport cv2\ntrain1 = train[train.label_group==994676122]\nfig = plt.figure(figsize=(15,15))\nx=1\nfor i in train1.file_path[:16]:\n    image = cv2.imread(i)\n    fig.add_subplot(4, 4, x)\n    plt.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n    plt.axis('off')\n    x+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:34.710093Z","iopub.execute_input":"2022-08-13T18:28:34.710534Z","iopub.status.idle":"2022-08-13T18:28:37.721877Z","shell.execute_reply.started":"2022-08-13T18:28:34.710501Z","shell.execute_reply":"2022-08-13T18:28:37.720693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2 = train[train.label_group==562358068]\nfig = plt.figure(figsize=(15,15))\nx=1\nfor i in train2.file_path[:16]:\n    image = cv2.imread(i)\n    fig.add_subplot(4, 4, x)\n    plt.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n    plt.axis('off')\n    x+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:37.723992Z","iopub.execute_input":"2022-08-13T18:28:37.724666Z","iopub.status.idle":"2022-08-13T18:28:40.357133Z","shell.execute_reply.started":"2022-08-13T18:28:37.724624Z","shell.execute_reply":"2022-08-13T18:28:40.355778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train3 = train[train.label_group==3113678103]\nfig = plt.figure(figsize=(15,15))\nx=1\nfor i in train3.file_path[:16]:\n    image = cv2.imread(i)\n    fig.add_subplot(4, 4, x)\n    plt.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n    plt.axis('off')\n    x+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:40.359043Z","iopub.execute_input":"2022-08-13T18:28:40.360443Z","iopub.status.idle":"2022-08-13T18:28:43.623895Z","shell.execute_reply.started":"2022-08-13T18:28:40.360358Z","shell.execute_reply":"2022-08-13T18:28:43.622396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain4 = train[train.label_group==1395102007]\nfig = plt.figure(figsize=(15,15))\nx=1\nfor i in train4.file_path[:16]:\n    image = cv2.imread(i)\n    fig.add_subplot(4, 4, x)\n    plt.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n    plt.axis('off')\n    x+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:43.626353Z","iopub.execute_input":"2022-08-13T18:28:43.626692Z","iopub.status.idle":"2022-08-13T18:28:43.998285Z","shell.execute_reply.started":"2022-08-13T18:28:43.626663Z","shell.execute_reply":"2022-08-13T18:28:43.996843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain4 = train[train.label_group==3907914144]\nfig = plt.figure(figsize=(15,15))\nx=1\nfor i in train4.file_path[:16]:\n    image = cv2.imread(i)\n    fig.add_subplot(4, 4, x)\n    plt.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n    plt.axis('off')\n    x+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:43.999743Z","iopub.execute_input":"2022-08-13T18:28:44.000079Z","iopub.status.idle":"2022-08-13T18:28:44.391123Z","shell.execute_reply.started":"2022-08-13T18:28:44.000049Z","shell.execute_reply":"2022-08-13T18:28:44.390001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Exploring titles:\nprint(train.shape)\nprint(train['title'].nunique())","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:44.393193Z","iopub.execute_input":"2022-08-13T18:28:44.393565Z","iopub.status.idle":"2022-08-13T18:28:44.412593Z","shell.execute_reply.started":"2022-08-13T18:28:44.393533Z","shell.execute_reply":"2022-08-13T18:28:44.411520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t = train['title'].value_counts().sort_values(ascending=False).reset_index()\nt.columns = ['title','count']\nt","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:44.413822Z","iopub.execute_input":"2022-08-13T18:28:44.414712Z","iopub.status.idle":"2022-08-13T18:28:44.451485Z","shell.execute_reply.started":"2022-08-13T18:28:44.414674Z","shell.execute_reply":"2022-08-13T18:28:44.450462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t. loc[t['count'] >1]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:44.452908Z","iopub.execute_input":"2022-08-13T18:28:44.454006Z","iopub.status.idle":"2022-08-13T18:28:44.472353Z","shell.execute_reply.started":"2022-08-13T18:28:44.453967Z","shell.execute_reply":"2022-08-13T18:28:44.470794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Display images for First title = \"Koko syubbanul muslimin koko azzahir koko baju\"\npath=train.loc[train[\"title\"] == \"Koko syubbanul muslimin koko azzahir koko baju\",\"file_path\"]\ndisplay_multiple_img(path,3,3)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:44.476460Z","iopub.execute_input":"2022-08-13T18:28:44.477394Z","iopub.status.idle":"2022-08-13T18:28:46.338358Z","shell.execute_reply.started":"2022-08-13T18:28:44.477341Z","shell.execute_reply":"2022-08-13T18:28:46.337238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Monde Boromon Cookies 1 tahun+ 120gr\npath=train.loc[train[\"title\"] == \"Monde Boromon Cookies 1 tahun+ 120gr\",\"file_path\"]\ndisplay_multiple_img(path,2,3)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:46.339640Z","iopub.execute_input":"2022-08-13T18:28:46.339962Z","iopub.status.idle":"2022-08-13T18:28:47.547367Z","shell.execute_reply.started":"2022-08-13T18:28:46.339931Z","shell.execute_reply":"2022-08-13T18:28:47.545977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Viva Air Mawar\npath=train.loc[train[\"title\"] == \"Viva Air Mawar\",\"file_path\"]\ndisplay_multiple_img(path,2,3)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:47.549684Z","iopub.execute_input":"2022-08-13T18:28:47.550018Z","iopub.status.idle":"2022-08-13T18:28:48.918012Z","shell.execute_reply.started":"2022-08-13T18:28:47.549982Z","shell.execute_reply":"2022-08-13T18:28:48.916908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#wordcloud\nwc = WordCloud(\n    background_color='white',\n    max_words = 150,\n    random_state = 42,\n    max_font_size=80\n    )\nwc.generate(' '.join(train[\"title\"]))\nplt.figure(figsize=(50,7))\nplt.imshow(wc)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T19:33:09.318607Z","iopub.execute_input":"2022-08-13T19:33:09.319292Z","iopub.status.idle":"2022-08-13T19:33:12.232925Z","shell.execute_reply.started":"2022-08-13T19:33:09.319245Z","shell.execute_reply":"2022-08-13T19:33:12.232070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:28:52.070018Z","iopub.execute_input":"2022-08-13T18:28:52.070529Z","iopub.status.idle":"2022-08-13T18:28:52.087864Z","shell.execute_reply.started":"2022-08-13T18:28:52.070484Z","shell.execute_reply":"2022-08-13T18:28:52.086637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#First we will work on labels i.e title\n#From the fouth row we can see that the tile is not in english, on investigating it has been found that is in \n#Indonesian langauge\n\nstopwords = nltk.corpus.stopwords.words('english')\nstopwords1 = nltk.corpus.stopwords.words('indonesian')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:52:03.319837Z","iopub.execute_input":"2022-08-13T18:52:03.320208Z","iopub.status.idle":"2022-08-13T18:52:03.332580Z","shell.execute_reply.started":"2022-08-13T18:52:03.320175Z","shell.execute_reply":"2022-08-13T18:52:03.331591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_stopwords(text):\n    text1=\" \".join([word for word in str(text).split() if word not in stopwords])\n    return \" \".join([word for word in str(text1).split() if word not in stopwords])\ntrain[\"title\"] = train[\"title\"].apply(lambda text: remove_stopwords(text))\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:52:05.705382Z","iopub.execute_input":"2022-08-13T18:52:05.706030Z","iopub.status.idle":"2022-08-13T18:52:07.711308Z","shell.execute_reply.started":"2022-08-13T18:52:05.705995Z","shell.execute_reply":"2022-08-13T18:52:07.710373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#removing Punctuations\nimport string\nenglish_punctuations = string.punctuation\npunctuations_list = english_punctuations\ndef remove_punctuations(text):\n    translator = str.maketrans('', '', punctuations_list)\n    return text.translate(translator)\ntrain[\"title\"] = train[\"title\"].apply(lambda text: remove_punctuations(text))\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:52:08.815464Z","iopub.execute_input":"2022-08-13T18:52:08.816568Z","iopub.status.idle":"2022-08-13T18:52:09.071965Z","shell.execute_reply.started":"2022-08-13T18:52:08.816524Z","shell.execute_reply":"2022-08-13T18:52:09.070740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#convert to lower case\ntrain[\"title\"] = train[\"title\"].str.lower()\n# strip leading and trailing spaces\ntrain[\"title\"] = train[\"title\"].str.strip(' ')\ntrain.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:52:10.784375Z","iopub.execute_input":"2022-08-13T18:52:10.784743Z","iopub.status.idle":"2022-08-13T18:52:10.822994Z","shell.execute_reply.started":"2022-08-13T18:52:10.784707Z","shell.execute_reply":"2022-08-13T18:52:10.822127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gensim\nfrom gensim import corpora, models\nfrom pprint import pprint","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:52:14.071862Z","iopub.execute_input":"2022-08-13T18:52:14.072244Z","iopub.status.idle":"2022-08-13T18:52:14.298440Z","shell.execute_reply.started":"2022-08-13T18:52:14.072191Z","shell.execute_reply":"2022-08-13T18:52:14.297198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Dictionary and corpus \ndocuments = train[\"title\"].tolist()\n\n# remove common words and tokenize\ntexts = [\n    [word for word in document.lower().split() if word not in stopwords]\n    for document in documents\n]\n\ndictionary = corpora.Dictionary(texts)\ncorpus = [dictionary.doc2bow(text) for text in texts]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:52:22.816215Z","iopub.execute_input":"2022-08-13T18:52:22.817033Z","iopub.status.idle":"2022-08-13T18:52:24.707298Z","shell.execute_reply.started":"2022-08-13T18:52:22.816997Z","shell.execute_reply":"2022-08-13T18:52:24.706287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfidf = models.TfidfModel(corpus)\n\ncorpus_tfidf = tfidf[corpus]\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:52:29.392702Z","iopub.execute_input":"2022-08-13T18:52:29.393069Z","iopub.status.idle":"2022-08-13T18:52:29.530191Z","shell.execute_reply.started":"2022-08-13T18:52:29.393038Z","shell.execute_reply":"2022-08-13T18:52:29.529258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#TF-IDF VEctorization\nfrom sklearn.feature_extraction.text import TfidfVectorizer  \ntfidf_vect = TfidfVectorizer(min_df=5, max_df=0.7)\nX_tfidf = tfidf_vect.fit_transform(train[\"title\"])\nX_f = pd.DataFrame(X_tfidf.toarray())\nX_f.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:52:32.692633Z","iopub.execute_input":"2022-08-13T18:52:32.693379Z","iopub.status.idle":"2022-08-13T18:52:33.990649Z","shell.execute_reply.started":"2022-08-13T18:52:32.693347Z","shell.execute_reply":"2022-08-13T18:52:33.989605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#######################################\n# Add ","metadata":{}},{"cell_type":"code","source":"test_f=tfidf_vect.transform(test['title'])\ntest_f = pd.DataFrame(test_f.toarray())","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:52:35.524121Z","iopub.execute_input":"2022-08-13T18:52:35.524823Z","iopub.status.idle":"2022-08-13T18:52:35.532573Z","shell.execute_reply.started":"2022-08-13T18:52:35.524786Z","shell.execute_reply":"2022-08-13T18:52:35.531303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#small sample, 1/10 of X_f\nsmall_sample = X_f.sample(n=int(len(X_f)/10))\nlen(small_sample)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:52:37.740060Z","iopub.execute_input":"2022-08-13T18:52:37.740432Z","iopub.status.idle":"2022-08-13T18:52:37.847820Z","shell.execute_reply.started":"2022-08-13T18:52:37.740400Z","shell.execute_reply":"2022-08-13T18:52:37.846712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#After an initial silhouette_score with 25,50,75,100,125,150 we found out our ideal k is in the range of 75-100\n#silhouette on small subset shows 80 clusters\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics import silhouette_score\nrange_n_clusters = [65,80,95,110]\nsilhouette_avg = []\nfor num_clusters in range_n_clusters:\n \n # initialise kmeans\n kmeans = KMeans(n_clusters=num_clusters, init='k-means++',random_state=1374, max_iter=200)\n kmeans.fit(small_sample)\n cluster_labels = kmeans.labels_\n \n # silhouette score\n silhouette_avg.append(silhouette_score(small_sample, cluster_labels))\n    \nplt.plot(range_n_clusters,silhouette_avg,\"bx-\")\nplt.xlabel(\"Values of K\") \nplt.ylabel(\"Silhouette score\") \nplt.title(\"Silhouette analysis For Optimal k\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T18:56:44.712198Z","iopub.execute_input":"2022-08-13T18:56:44.712893Z","iopub.status.idle":"2022-08-13T19:02:47.271052Z","shell.execute_reply.started":"2022-08-13T18:56:44.712858Z","shell.execute_reply":"2022-08-13T19:02:47.270132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nmodel = KMeans(n_clusters=80, init='k-means++',random_state=1374, max_iter=200)\nmodel.fit(X_f)\nlabels=model.labels_\ntrain['text_cluster']=labels","metadata":{"execution":{"iopub.status.busy":"2022-08-13T19:03:02.706324Z","iopub.execute_input":"2022-08-13T19:03:02.707390Z","iopub.status.idle":"2022-08-13T19:19:49.479285Z","shell.execute_reply.started":"2022-08-13T19:03:02.707340Z","shell.execute_reply":"2022-08-13T19:19:49.478446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.naive_bayes import MultinomialNB\nfrom sklearn import metrics\nnaive_bayes_classifier = MultinomialNB()\nnaive_bayes_classifier.fit(X_f,train[\"text_cluster\"])\ny_predict_test = naive_bayes_classifier.predict(test_f)\ny_predict_test","metadata":{"execution":{"iopub.status.busy":"2022-08-13T19:32:31.105100Z","iopub.execute_input":"2022-08-13T19:32:31.106127Z","iopub.status.idle":"2022-08-13T19:32:32.375807Z","shell.execute_reply.started":"2022-08-13T19:32:31.106086Z","shell.execute_reply":"2022-08-13T19:32:32.374694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.text_cluster.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T19:32:36.038066Z","iopub.execute_input":"2022-08-13T19:32:36.038535Z","iopub.status.idle":"2022-08-13T19:32:36.057134Z","shell.execute_reply.started":"2022-08-13T19:32:36.038497Z","shell.execute_reply":"2022-08-13T19:32:36.056288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfirst=train[train['text_cluster']==30]['posting_id'].values\nsecond=train[train['text_cluster']==8]['posting_id'].values\nthird=train[train['text_cluster']==4]['posting_id'].values\ntest['predcited_by_text_kmean']=[first,second,third]\ntest","metadata":{"execution":{"iopub.status.busy":"2022-08-13T19:32:49.238782Z","iopub.execute_input":"2022-08-13T19:32:49.239140Z","iopub.status.idle":"2022-08-13T19:32:49.262126Z","shell.execute_reply.started":"2022-08-13T19:32:49.239109Z","shell.execute_reply":"2022-08-13T19:32:49.261027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#For example, what is in cluster 30? toy,puzzle,child\nb=train[train['text_cluster']==30]['title']\n\nwc.generate(' '.join(b))\nplt.figure(figsize=(50,7))\nplt.imshow(wc)\nplt.show()\ntest.title.tail(1)\n#in the title of first row of test we can see toys","metadata":{"execution":{"iopub.status.busy":"2022-08-13T19:37:27.937649Z","iopub.execute_input":"2022-08-13T19:37:27.938014Z","iopub.status.idle":"2022-08-13T19:37:28.491094Z","shell.execute_reply.started":"2022-08-13T19:37:27.937981Z","shell.execute_reply":"2022-08-13T19:37:28.490159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\nscore=naive_bayes_classifier.score(X_f,train['text_cluster'])\nprint('Accuracy of Naive Bayes and kmeans :')\nprint(score*100.0) #With 80 clusters we have 71 percent","metadata":{"execution":{"iopub.status.busy":"2022-08-13T19:39:07.366887Z","iopub.execute_input":"2022-08-13T19:39:07.368100Z","iopub.status.idle":"2022-08-13T19:39:08.581683Z","shell.execute_reply.started":"2022-08-13T19:39:07.368058Z","shell.execute_reply":"2022-08-13T19:39:08.580401Z"},"trusted":true},"execution_count":null,"outputs":[]}]}