{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# -*- coding: utf-8 -*-\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/class-descriptions.csv')\ndf.head()\n\ndef isEnglish(s):\n    try:\n        s.encode(encoding='utf-8').decode('ascii')\n    except UnicodeDecodeError:\n        return False\n    else:\n        return True\n\ndf['english'] = df['description'].apply(isEnglish)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cac811fd0a79ce5c0d9cd41823a4ca6df3a5e8bb"},"cell_type":"code","source":"xdf = df[df.english == False]\nprint('shape: ', xdf.shape)\nxdf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8de76fc2c9a7fec80fd513b5dd0d312b0e4b766f"},"cell_type":"code","source":"strange_label_list = xdf['label_code'].tolist()\nx = pd.read_csv('../input/classes-trainable.csv')\nfor l in list(x.label_code.unique()):\n    if l in strange_label_list:\n        print('Got {} in trainable labels'.format(l))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}