{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-09-14T13:44:44.834678Z","iopub.execute_input":"2021-09-14T13:44:44.8355Z","iopub.status.idle":"2021-09-14T13:44:44.858895Z","shell.execute_reply.started":"2021-09-14T13:44:44.835451Z","shell.execute_reply":"2021-09-14T13:44:44.857443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Wikipedia - Image/Caption Matching**\n#### Retrieve captions based on images","metadata":{}},{"cell_type":"markdown","source":"<p> Description:\nA picture is worth a thousand words, yet sometimes a few will do. We all rely on online images for knowledge sharing, learning, and understanding. Even the largest websites are missing visual content and metadata to pair with their images. Captions and “alt text” increase accessibility and enable better search. The majority of images on Wikipedia articles, for example, don't have any written context connected to the image. Open models could help anyone improve accessibility and learning for all.\n\nCurrent solutions rely on simple methods based on translations or page interlinks, which have limited coverage. Even the most advanced computer vision image captioning isn't suitable for images with complex semantics.\n\nIn this competition, you’ll build a model that automatically retrieves the text closest to an image. Specifically, you'll train your model to associate given images with article titles or complex captions, in multiple languages. The best models will account for the semantic granularity of Wikipedia images.</p>","metadata":{}},{"cell_type":"code","source":"import os\nimport random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\n\nimport PIL.Image\n","metadata":{"execution":{"iopub.status.busy":"2021-09-14T13:44:44.873761Z","iopub.execute_input":"2021-09-14T13:44:44.87466Z","iopub.status.idle":"2021-09-14T13:44:44.880115Z","shell.execute_reply.started":"2021-09-14T13:44:44.874622Z","shell.execute_reply":"2021-09-14T13:44:44.879195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_data\ntrain_data=pd.read_csv('../input/wikipedia-image-caption/image_data_test/image_pixels/test_image_pixels_part-00000.csv', sep='\\t',names=['image_url', 'b64_bytes', 'metadata_url'])\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-14T13:44:44.936503Z","iopub.execute_input":"2021-09-14T13:44:44.936894Z","iopub.status.idle":"2021-09-14T13:44:48.555921Z","shell.execute_reply.started":"2021-09-14T13:44:44.936859Z","shell.execute_reply":"2021-09-14T13:44:48.554722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test data\ntest_data = pd.read_csv('../input/wikipedia-image-caption/test.tsv', sep='\\t')\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-14T13:44:48.558105Z","iopub.execute_input":"2021-09-14T13:44:48.559107Z","iopub.status.idle":"2021-09-14T13:44:48.85483Z","shell.execute_reply.started":"2021-09-14T13:44:48.559054Z","shell.execute_reply":"2021-09-14T13:44:48.853834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission File\nsub_file = pd.read_csv('../input/wikipedia-image-caption/sample_submission.csv')\nsub_file.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-09-14T13:44:48.8562Z","iopub.execute_input":"2021-09-14T13:44:48.856505Z","iopub.status.idle":"2021-09-14T13:44:49.222499Z","shell.execute_reply.started":"2021-09-14T13:44:48.856472Z","shell.execute_reply":"2021-09-14T13:44:49.221683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataset info\nprint(\"shape of train file\",train_data.shape)\nprint(\"shape of test file\",test_data.shape)\nprint(\"shape of Submission file\",sub_file.shape)","metadata":{"execution":{"iopub.status.busy":"2021-09-14T13:44:49.224925Z","iopub.execute_input":"2021-09-14T13:44:49.225233Z","iopub.status.idle":"2021-09-14T13:44:49.231074Z","shell.execute_reply.started":"2021-09-14T13:44:49.225195Z","shell.execute_reply":"2021-09-14T13:44:49.230093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2021-09-14T13:44:49.232592Z","iopub.execute_input":"2021-09-14T13:44:49.233224Z","iopub.status.idle":"2021-09-14T13:44:49.254156Z","shell.execute_reply.started":"2021-09-14T13:44:49.23318Z","shell.execute_reply":"2021-09-14T13:44:49.253018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#language distribution plot\nplt.figure(figsize=(18, 8))\nsns.set(style=\"darkgrid\")\nsns.countplot(test_data[\"language\"])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-14T13:44:49.25582Z","iopub.execute_input":"2021-09-14T13:44:49.256142Z","iopub.status.idle":"2021-09-14T13:44:50.966757Z","shell.execute_reply.started":"2021-09-14T13:44:49.256105Z","shell.execute_reply":"2021-09-14T13:44:50.965995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"description: most of the images have english caption. ","metadata":{}},{"cell_type":"code","source":"import urllib\ndef get_links(df, num):\n    return df.image_url[10:num].values\n\nlinks = get_links(train_data, 21)\n\ndef load_images(links):\n    images = []\n    \n    for link in links:\n        URL = link\n        try:\n\n            with urllib.request.urlopen(URL) as url:\n                with open('./temp.jpg', 'wb') as f:\n                    f.write(url.read())\n\n            img = PIL.Image.open('./temp.jpg')\n            img = np.asarray(img)\n            images.append(img)\n        except:\n            continue\n    return images\n\ndef display_images(images, title=None): \n    f, ax = plt.subplots(2,5, figsize=(18,12))\n    if title:\n        f.suptitle(title, fontsize = 30)\n\n    for i, image_id in enumerate(images):\n        ax[i//5, i%5].imshow(image_id) \n   \n        ax[i//5, i%5].axis('off')\n\n    plt.show() \n#code taken from https://www.kaggle.com/hijest/wikipedia-image-caption-matching-starter-eda ","metadata":{"execution":{"iopub.status.busy":"2021-09-14T13:44:50.967906Z","iopub.execute_input":"2021-09-14T13:44:50.96813Z","iopub.status.idle":"2021-09-14T13:44:50.980156Z","shell.execute_reply.started":"2021-09-14T13:44:50.968105Z","shell.execute_reply":"2021-09-14T13:44:50.978996Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = load_images(links)\ndisplay_images(images)","metadata":{"execution":{"iopub.status.busy":"2021-09-14T13:44:50.981431Z","iopub.execute_input":"2021-09-14T13:44:50.981707Z","iopub.status.idle":"2021-09-14T13:45:18.898489Z","shell.execute_reply.started":"2021-09-14T13:44:50.981678Z","shell.execute_reply":"2021-09-14T13:45:18.897798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport squarify    \nsquarify.plot(sizes=test_data['language'].value_counts().values, \n              label=test_data['language'].value_counts().index )\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-14T13:45:18.899692Z","iopub.execute_input":"2021-09-14T13:45:18.900618Z","iopub.status.idle":"2021-09-14T13:45:19.373925Z","shell.execute_reply.started":"2021-09-14T13:45:18.900576Z","shell.execute_reply":"2021-09-14T13:45:19.372967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/wikipedia-image-caption/train-00000-of-00005.tsv', sep='\\t',nrows=100)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-14T13:49:28.574241Z","iopub.execute_input":"2021-09-14T13:49:28.57461Z","iopub.status.idle":"2021-09-14T13:49:28.62517Z","shell.execute_reply.started":"2021-09-14T13:49:28.574571Z","shell.execute_reply":"2021-09-14T13:49:28.624019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from wordcloud import WordCloud, STOPWORDS \ncloud = WordCloud(collocations = False,stopwords={'nan'},background_color='white').generate(\" \".join(df['page_title'].astype(str)))\nplt.figure(figsize=(16, 10))\nplt.title('WordCloud ',fontsize=20,pad=40)\nplt.imshow(cloud,interpolation='bilinear')\nplt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2021-09-14T13:53:06.690475Z","iopub.execute_input":"2021-09-14T13:53:06.691595Z","iopub.status.idle":"2021-09-14T13:53:07.441963Z","shell.execute_reply.started":"2021-09-14T13:53:06.691507Z","shell.execute_reply":"2021-09-14T13:53:07.441392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Work in progress","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}