{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"\nimport numpy as np\nimport pandas as pd\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\npd.options.display.max_colwidth = 100\n\n# Todo: read data\ntrain=pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\")\ntrain.head()\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# Todo: show some basic information about the data\n# example: what is the size of the data, how is the distribution of targe look like\ntrain.shape\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.isnull().sum().sum()\n# This shows that there are no missing values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['target'].value_counts()\n# the distribution of target ??? ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Todo: try some basic text cleaning steps\n# example:\n# - remove numbers\ndef number_remove (string):\n    result = ''.join([i for i in string if not i.isdigit()])\n    return result\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"number_removed=train['question_text'].apply(number_remove)\nnumber_removed\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def space_remove(string):\n    return string.replace(\" \",\"\") \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# - remove multiple spaces\nnumber_space_removed=number_removed.apply(space_remove)\nnumber_space_removed","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def punc_remove(string):\n    punctuations = '''!()-[]{};:'\"\\,<>./?@#$%^&*_~'''\n    no_punct = \"\"\n    for char in string:\n        if char not in punctuations:\n            no_punct = no_punct + char\n    return no_punct\n\npunc_remove(\"aser2#$GF<sdf\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#cleaned text \nnumber_space_punct_removed=number_space_removed.apply(punc_remove)\nnumber_space_punct_removed","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['cleaned_text']=number_space_punct_removed\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"columns=['qid','cleaned_text','target']\ncleaned_df=train[columns]\ncleaned_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Todo: vectorizing text data using bag-of-words (ngram) and Tfidf\n\n\n# Todo: explore the arguments in CountVectorizer and TfidfVectorizer, compare the results\n\n\n# Todo: save the vectorized data use `pickle`\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}