{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2021-11-18T07:59:12.972236Z","iopub.execute_input":"2021-11-18T07:59:12.972850Z","iopub.status.idle":"2021-11-18T07:59:13.242443Z","shell.execute_reply.started":"2021-11-18T07:59:12.972720Z","shell.execute_reply":"2021-11-18T07:59:13.241512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/wikipedia-image-caption/train-00000-of-00005.tsv', sep='\\t', usecols = ['image_url', 'page_title', 'language', 'caption_alt_text_description'])\ndf","metadata":{"execution":{"iopub.status.busy":"2021-11-18T07:59:22.948589Z","iopub.execute_input":"2021-11-18T07:59:22.948874Z","iopub.status.idle":"2021-11-18T08:03:48.393266Z","shell.execute_reply.started":"2021-11-18T07:59:22.948843Z","shell.execute_reply":"2021-11-18T08:03:48.392372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_img_dir = '/kaggle/working/train_images_fr/'\ncaptions_txt = '/kaggle/working/captions_fr.txt'\nif not os.path.exists(train_img_dir):\n    os.makedirs(train_img_dir)","metadata":{"execution":{"iopub.status.busy":"2021-11-18T08:03:48.404602Z","iopub.execute_input":"2021-11-18T08:03:48.404904Z","iopub.status.idle":"2021-11-18T08:03:48.411008Z","shell.execute_reply.started":"2021-11-18T08:03:48.404864Z","shell.execute_reply":"2021-11-18T08:03:48.410116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir(train_img_dir))","metadata":{"execution":{"iopub.status.busy":"2021-11-18T08:40:59.339111Z","iopub.execute_input":"2021-11-18T08:40:59.340074Z","iopub.status.idle":"2021-11-18T08:40:59.351477Z","shell.execute_reply.started":"2021-11-18T08:40:59.340001Z","shell.execute_reply":"2021-11-18T08:40:59.350505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lines = [l for l in open(captions_txt,'r').readlines() if not l=='\\n']","metadata":{"execution":{"iopub.status.busy":"2021-11-18T08:44:46.926958Z","iopub.execute_input":"2021-11-18T08:44:46.927920Z","iopub.status.idle":"2021-11-18T08:44:46.936768Z","shell.execute_reply.started":"2021-11-18T08:44:46.927863Z","shell.execute_reply":"2021-11-18T08:44:46.936017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(lines)","metadata":{"execution":{"iopub.status.busy":"2021-11-18T08:45:00.293890Z","iopub.execute_input":"2021-11-18T08:45:00.294219Z","iopub.status.idle":"2021-11-18T08:45:00.301026Z","shell.execute_reply.started":"2021-11-18T08:45:00.294178Z","shell.execute_reply":"2021-11-18T08:45:00.300150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[df.language=='en']\ndf.dropna(inplace=True)\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2021-11-18T08:03:48.468630Z","iopub.execute_input":"2021-11-18T08:03:48.468840Z","iopub.status.idle":"2021-11-18T08:03:50.421070Z","shell.execute_reply.started":"2021-11-18T08:03:48.468813Z","shell.execute_reply":"2021-11-18T08:03:50.420125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport urllib\nimport urllib.request as urll\nimport PIL.Image\n\ndef save_data(df, num):\n    txt = open(captions_txt,'w')\n    for i in range(num-10,0,-1) : \n        link = df.iloc[i, 1]\n        img_name = '_'.join(df.iloc[i,2].replace(',',' ').split())\n        desc = df.iloc[i, 3]\n        print(link)\n        try:\n            with urll.urlopen(link) as url:\n                with open(os.path.join(train_img_dir,img_name+'.jpg'), 'wb') as f:\n                    f.write(url.read())\n            img = cv2.imread(os.path.join(train_img_dir,img_name+'.jpg'))\n            img = cv2.resize(img,(300,300))\n            cv2.imwrite(os.path.join(train_img_dir,img_name+'.jpg'),img)\n            print(desc)\n            txt.write(img_name+'\\t\\t'+desc+'\\n')\n        except Exception as e:\n            print(\"Exception occured:\",e)\n    txt.close()\n\nsave_data(df, 11290)","metadata":{"execution":{"iopub.status.busy":"2021-11-18T08:05:06.019028Z","iopub.execute_input":"2021-11-18T08:05:06.019366Z","iopub.status.idle":"2021-11-18T08:36:41.254246Z","shell.execute_reply.started":"2021-11-18T08:05:06.019329Z","shell.execute_reply":"2021-11-18T08:36:41.252237Z"},"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}