{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install transliterate","metadata":{"execution":{"iopub.status.busy":"2022-11-06T20:04:22.265084Z","iopub.execute_input":"2022-11-06T20:04:22.265607Z","iopub.status.idle":"2022-11-06T20:04:33.419112Z","shell.execute_reply.started":"2022-11-06T20:04:22.265572Z","shell.execute_reply":"2022-11-06T20:04:33.417278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport urllib.parse\n\nfrom transliterate import translit\nimport fasttext","metadata":{"execution":{"iopub.status.busy":"2022-11-06T20:04:54.976007Z","iopub.execute_input":"2022-11-06T20:04:54.976326Z","iopub.status.idle":"2022-11-06T20:04:55.007697Z","shell.execute_reply.started":"2022-11-06T20:04:54.976300Z","shell.execute_reply":"2022-11-06T20:04:55.005745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ft_model = fasttext.load_model('../input/fasttextmodel/lid.176.ftz')","metadata":{"execution":{"iopub.status.busy":"2022-11-06T20:23:59.531269Z","iopub.execute_input":"2022-11-06T20:23:59.531578Z","iopub.status.idle":"2022-11-06T20:23:59.572233Z","shell.execute_reply.started":"2022-11-06T20:23:59.531555Z","shell.execute_reply":"2022-11-06T20:23:59.571304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PART = 0\nMIN, MAX = 4, 5\n\ncolumns = ['page_title', 'image_url', 'caption_reference_description']","metadata":{"execution":{"iopub.status.busy":"2022-11-06T20:24:15.019453Z","iopub.execute_input":"2022-11-06T20:24:15.019792Z","iopub.status.idle":"2022-11-06T20:24:15.024618Z","shell.execute_reply.started":"2022-11-06T20:24:15.019768Z","shell.execute_reply":"2022-11-06T20:24:15.023348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain0 = pd.read_csv('../input/wikipedia-image-caption/train-0000{}-of-00005.tsv'.format(PART), sep='\\t', usecols=columns)","metadata":{"execution":{"iopub.status.busy":"2022-11-06T20:31:27.699321Z","iopub.execute_input":"2022-11-06T20:31:27.699910Z","iopub.status.idle":"2022-11-06T20:33:55.251204Z","shell.execute_reply.started":"2022-11-06T20:31:27.699829Z","shell.execute_reply":"2022-11-06T20:33:55.249731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ok_images = train0[train0[columns].notnull().all(1)].copy()\nok_images","metadata":{"execution":{"iopub.status.busy":"2022-11-06T20:35:22.688583Z","iopub.execute_input":"2022-11-06T20:35:22.689103Z","iopub.status.idle":"2022-11-06T20:35:24.988447Z","shell.execute_reply.started":"2022-11-06T20:35:22.689075Z","shell.execute_reply":"2022-11-06T20:35:24.985995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def repair_filename(url):\n    \"\"\"\n    Repairs the filename from the specified image url.\n    \"\"\"\n    return urllib.parse.unquote('.'.join(url.split('/')[-1].split('.')[:-1]).replace('_', ' '))\n\n\ndef equip_df(df):\n    df['filename'] = df['image_url'].apply(repair_filename)\n    df[\"pured_filename\"] = df['filename'].str.replace(r'[^\\w\\s]', ' ', regex=True).replace(r'\\s{2,}', ' ', regex=True)\n\n    df[\"spaced_filename\"] = df[\"pured_filename\"].str. \\\n        replace(r'([a-z])([A-Z])', r'\\1 \\2', regex=True). \\\n        replace(r'([а-я])([А-Я])', r'\\1 \\2', regex=True). \\\n        replace(r'([a-z])(\\d)', r'\\1 \\2', regex=True). \\\n        replace(r'([а-я])(\\d)', r'\\1 \\2', regex=True). \\\n        replace(r'(\\d)([a-zA-Z])', r'\\1 \\2', regex=True). \\\n        replace(r'(\\d)([а-яА-Я])', r'\\1 \\2', regex=True)\n    \n    df[\"undigit_filename\"] = df['spaced_filename'].str.replace(r'[0-9]', '', regex=True)\n    df['undigit_filename'] = df['undigit_filename'].str.replace(r'\\s{2,}', ' ', regex=True)\n    df['undigit_filename'] = df['undigit_filename'].str.strip()    \n    \n    df['undigit_filename_empty'] = np.logical_or(df['undigit_filename'].str.isspace().values, (df['undigit_filename'] == '').values)\n    \n    \ndef extract_lang(s: str):\n    o = ft_model.predict(s)\n    return o[0][0].split('__')[-1], o[1][0]\n\n\ndef transliterate(s: str):\n    try:\n        return translit(s, reversed=True)\n    except:\n        return ''","metadata":{"execution":{"iopub.status.busy":"2022-11-06T20:36:08.475457Z","iopub.execute_input":"2022-11-06T20:36:08.475805Z","iopub.status.idle":"2022-11-06T20:36:08.492298Z","shell.execute_reply.started":"2022-11-06T20:36:08.475773Z","shell.execute_reply":"2022-11-06T20:36:08.490053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nimage_count = ok_images.groupby('image_url').size()\n\nfiltered = image_count[(MIN <= image_count) & (image_count <= MAX)]\nimages2train = pd.DataFrame({ 'image_url': filtered.index, 'count': filtered.values })\n\nequip_df(images2train)\nr = images2train['spaced_filename'].map(extract_lang)\nimages2train['filename_lang'], images2train['filename_lang_p'] = zip(*r)\nimages2train['filename_en'] = images2train['filename_lang'] == 'en'\n\nimages2train['section'] = images2train['image_url'].str.extract(r'wikipedia/([a-zA-Z]{1,})/')\nimages2train['commons'] = images2train['section'] == 'commons'\nimages2train['lang_ok'] = images2train['section'] == images2train['filename_lang']\n\nimages2train['spaced_filename_translit'] = images2train['spaced_filename'].map(transliterate)\nimages2train['spaced_filename_translited'] = images2train['spaced_filename_translit'] != ''\n\nimages2train['ext'] = images2train['image_url'].str.rsplit('.', 1).str[-1].str.lower()\n\nimages2train['filename_contains_digit'] = images2train['filename'].str.contains(r'\\d', regex=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-06T20:36:28.943400Z","iopub.execute_input":"2022-11-06T20:36:28.943696Z","iopub.status.idle":"2022-11-06T20:36:52.483755Z","shell.execute_reply.started":"2022-11-06T20:36:28.943672Z","shell.execute_reply":"2022-11-06T20:36:52.481848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images2train","metadata":{"execution":{"iopub.status.busy":"2022-11-06T20:36:52.485633Z","iopub.execute_input":"2022-11-06T20:36:52.485935Z","iopub.status.idle":"2022-11-06T20:36:52.512077Z","shell.execute_reply.started":"2022-11-06T20:36:52.485911Z","shell.execute_reply":"2022-11-06T20:36:52.510773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"matching_columns = ['image_url', 'page_title', 'caption_reference_description', 'count', 'spaced_filename', 'spaced_filename_translit', 'title_translit', 'caption_translit', \\\n                    'page_title_lang', 'page_title_lang_p', 'page_title_en', 'caption_lang', 'caption_lang_p', 'caption_en', \\\n                    'caption_contains_digit', 'undigit_caption', 'page_title_contains_digit', 'undigit_page_title',\n                    'target']\n\ntrain_matchings = pd.merge(ok_images, images2train, on='image_url')\ntrain_matchings['target'] = train_matchings['page_title'] + ' [SEP] ' + train_matchings['caption_reference_description']\n\ntrain_matchings['title_translit'] = train_matchings['page_title'].map(transliterate)\ntrain_matchings['caption_translit'] = train_matchings['caption_reference_description'].map(transliterate)\n\nr = train_matchings['page_title'].map(extract_lang)\ntrain_matchings['page_title_lang'], train_matchings['page_title_lang_p'] = zip(*r)\ntrain_matchings['page_title_en'] = train_matchings['page_title_lang'] == 'en'\n\nr = train_matchings['caption_reference_description'].map(lambda o: o.replace('\\n', ' ')).map(extract_lang)\ntrain_matchings['caption_lang'], train_matchings['caption_lang_p'] = zip(*r)\ntrain_matchings['caption_en'] = train_matchings['caption_lang'] == 'en'\n\ntrain_matchings['caption_contains_digit'] = train_matchings['caption_reference_description'].str.contains(r'\\d', regex=True)\ntrain_matchings['undigit_caption'] = train_matchings['caption_reference_description'].str.replace(r'\\d', '', regex=True).replace(r'\\s{2,}', ' ', regex=True)\n\ntrain_matchings['page_title_contains_digit'] = train_matchings['page_title'].str.contains(r'\\d', regex=True)\ntrain_matchings['undigit_page_title'] = train_matchings['page_title'].str.replace(r'\\d', '', regex=True).replace(r'\\s{2,}', ' ', regex=True)\n\ntrain_matchings[matching_columns]","metadata":{"execution":{"iopub.status.busy":"2022-11-06T20:37:13.089689Z","iopub.execute_input":"2022-11-06T20:37:13.090297Z","iopub.status.idle":"2022-11-06T20:39:02.937766Z","shell.execute_reply.started":"2022-11-06T20:37:13.090262Z","shell.execute_reply":"2022-11-06T20:39:02.936173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_matchings[matching_columns].to_csv('matchings_part{}_between{},{}.csv'.format(PART, MIN, MAX), index=False)\nimages2train.to_csv('images_part{}_between{},{}.csv'.format(PART, MIN, MAX), index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-06T20:40:30.404428Z","iopub.execute_input":"2022-11-06T20:40:30.404808Z","iopub.status.idle":"2022-11-06T20:40:35.741662Z","shell.execute_reply.started":"2022-11-06T20:40:30.404781Z","shell.execute_reply":"2022-11-06T20:40:35.740964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0.info()\npd.set_option('max_colwidth',100)\ntrain_matchings.iloc[0:4,:]\n","metadata":{"execution":{"iopub.status.busy":"2022-11-06T21:34:33.144484Z","iopub.execute_input":"2022-11-06T21:34:33.144840Z","iopub.status.idle":"2022-11-06T21:34:33.182418Z","shell.execute_reply.started":"2022-11-06T21:34:33.144815Z","shell.execute_reply":"2022-11-06T21:34:33.180476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install deep-translator","metadata":{"execution":{"iopub.status.busy":"2022-11-06T23:43:41.927819Z","iopub.execute_input":"2022-11-06T23:43:41.929544Z","iopub.status.idle":"2022-11-06T23:43:56.156301Z","shell.execute_reply.started":"2022-11-06T23:43:41.929449Z","shell.execute_reply":"2022-11-06T23:43:56.154714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom deep_translator import GoogleTranslator\n\ntranslator = GoogleTranslator(source='auto', target='en')\nimages = pd.read_csv('./images_part0_between4,5.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-06T23:47:13.118536Z","iopub.execute_input":"2022-11-06T23:47:13.119106Z","iopub.status.idle":"2022-11-06T23:47:13.737032Z","shell.execute_reply.started":"2022-11-06T23:47:13.119043Z","shell.execute_reply":"2022-11-06T23:47:13.735798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nimages['undigit_filename_translation'] = np.nan\nindices = images[(images['filename_lang'] != 'en') & (~images['undigit_filename_empty'])].index\n\nprint(len(indices))\n\nl = indices[4 * len(indices) // 5 :]\n    \nimages.loc[l, 'undigit_filename_translation'] = translator.translate_batch(images.loc[l, 'undigit_filename'].str.slice(stop=4500).tolist())","metadata":{"execution":{"iopub.status.busy":"2022-11-06T23:49:08.238798Z","iopub.execute_input":"2022-11-06T23:49:08.239324Z"},"trusted":true},"execution_count":null,"outputs":[]}]}