{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport nltk","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = pd.read_csv(\"../input/quora-dataset/train.csv\")\ndata_test = pd.read_csv(\"../input/quora-dataset/test.csv\")\nsubmission = pd.read_csv(\"../input/quora-dataset/sample_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from nltk.corpus import stopwords\nfrom nltk.stem import WordNetLemmatizer\nlemmatizer = WordNetLemmatizer()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def clean(text):\n    wn = nltk.WordNetLemmatizer()\n    stopword = nltk.corpus.stopwords.words('english')\n    tokens = nltk.word_tokenize(text)\n    lower = [word.lower() for word in tokens]\n    no_stopwords = [word for word in lower if word not in stopword]\n    no_alpha = [word for word in no_stopwords if word.isalpha()]\n    lemm_text = [wn.lemmatize(word) for word in no_alpha]\n    clean_text = lemm_text\n    return clean_text","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['clean']=data['question_text'].map(clean)\ndata['clean_text']=data['clean'].apply(lambda x: \" \".join([str(word) for word in x]))\n\n#Test data\ndata_test['clean']=data_test['question_text'].map(clean)\ndata_test['clean_text']=data_test['clean'].apply(lambda x: \" \".join([str(word) for word in x]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#from sklearn.feature_extraction.text import CountVectorizer\n#cv = CountVectorizer()\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n#tfidf = TfidfVectorizer()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#cv_data = cv.fit_transform(data['clean_text'])\n#tfidf_data = tfidf.fit_transform(data['clean_text'])\n#tfidf_test = tfidf.fit_transform(data_test['clean_text'])\n\ntfidf = TfidfVectorizer(analyzer='word',ngram_range=(1, 2),min_df=0, stop_words='english')\na=tfidf.fit_transform(data['question_text'].values.tolist() + data_test['question_text'].values.tolist())\ntfidf_data = tfidf.transform(data['question_text'].values.tolist())\ntfidf_test = tfidf.transform(data_test['question_text'].values.tolist())\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_tfidf = tfidf.transform(data['question_text'].values.tolist())\ntest_tfidf = tfidf.transform(data_test['question_text'].values.tolist())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = train_tfidf\ny = data['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#from sklearn.model_selection import train_test_split\n#X_train,X_test,y_train,y_test=train_test_split(X,y,random_state=1,test_size=0.3,shuffle=False)\nX_train = X\ny_train = y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nregressor = LogisticRegression(max_iter=1400000).fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred = regressor.predict(test_tfidf)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission['prediction'] = y_pred\nsubmission.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}