{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pandas_profiling as pp\n\nimport re\nimport nltk.corpus\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom nltk.stem import WordNetLemmatizer\nfrom string import digits\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\nfrom sklearn.decomposition import NMF\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import mean_squared_error\nimport itertools\nfrom sklearn.cluster import KMeans\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import GradientBoostingClassifier\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-06T23:43:17.754941Z","iopub.execute_input":"2022-08-06T23:43:17.755339Z","iopub.status.idle":"2022-08-06T23:43:17.768705Z","shell.execute_reply.started":"2022-08-06T23:43:17.755310Z","shell.execute_reply":"2022-08-06T23:43:17.767648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part 1. EDA and Model Creation and Comparison with Supervised Learning\n\nIn this part I will do some simple EDA and feature creation to prepare the data for the models.\n\nI will then create two multiple models to try and get an accurate outcome.","metadata":{}},{"cell_type":"markdown","source":"# 1.1 EDA","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/learn-ai-bbc/BBC News Train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:18.089530Z","iopub.execute_input":"2022-08-06T23:43:18.090646Z","iopub.status.idle":"2022-08-06T23:43:18.126638Z","shell.execute_reply.started":"2022-08-06T23:43:18.090614Z","shell.execute_reply":"2022-08-06T23:43:18.125470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Starting the EDA with an investigation of the untokenized data.\nThe following observations are important:\n1. There is no missing data\n2. 96.6% of the articles are unique\n3. The article category prevelance is as follows - Sports {346}, Business {336}, Politics {274}, Entertainment {273}, Tech {261}","metadata":{}},{"cell_type":"code","source":"profile = pp.ProfileReport(train, title=\"Pandas Profiling Report\", explorative=True)\nprofile.to_notebook_iframe()\nprofile.to_file(\"first_profile.html\")","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:18.890529Z","iopub.execute_input":"2022-08-06T23:43:18.890873Z","iopub.status.idle":"2022-08-06T23:43:23.077760Z","shell.execute_reply.started":"2022-08-06T23:43:18.890846Z","shell.execute_reply":"2022-08-06T23:43:23.077168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stop_words = stopwords.words('english')\ntrain['Text'] = train['Text'].apply(lambda row: re.sub(r'[^\\w\\s]+', '', row))\ntrain['Text'] = train['Text'].apply(lambda x: ' '.join([word for word in x.split() if word not in (stop_words)]))\ntrain['Text'] = train['Text'].apply(lambda x: re.sub(' +', ' ', x))\n\nX_train, X_valid, y_train, y_valid = train_test_split(train['Text'], train['Category'], test_size=0.33, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:23.079256Z","iopub.execute_input":"2022-08-06T23:43:23.079633Z","iopub.status.idle":"2022-08-06T23:43:24.311982Z","shell.execute_reply.started":"2022-08-06T23:43:23.079610Z","shell.execute_reply":"2022-08-06T23:43:24.310938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next we run Pandas Profiling again on the dataset to discover any new insights from tokenizing the document.\n\nWe find that the average document has 220 words. The maximum amount of words is 1698 and the minimum is 48.","metadata":{}},{"cell_type":"code","source":"dataframe = pd.DataFrame()\ndataframe['tokenized'] = train.apply(lambda row: nltk.word_tokenize(row['Text']), axis=1)\ndataframe['num_words'] = dataframe['tokenized'].apply(lambda lst: len(lst))\n\nprofile = pp.ProfileReport(dataframe, title=\"Pandas Profiling Report\", explorative=True)\nprofile.to_notebook_iframe()\nprofile.to_file(\"second_profile.html\")","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:24.313333Z","iopub.execute_input":"2022-08-06T23:43:24.313601Z","iopub.status.idle":"2022-08-06T23:43:28.053217Z","shell.execute_reply.started":"2022-08-06T23:43:24.313575Z","shell.execute_reply":"2022-08-06T23:43:28.051727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next we create a vector of words and the number of times they appear in a text.\n\nWe transform this into an array and return a training and validation set set for supervised learning.","metadata":{}},{"cell_type":"code","source":"def count_vectorization(n_features):\n    tf_vectorizer = CountVectorizer(max_df=0.95, min_df=2, max_features=n_features, stop_words=\"english\", lowercase = True)\n    tf = tf_vectorizer.fit_transform(train['Text'])\n    count_tokens = tf_vectorizer.get_feature_names_out()\n    train_countvect = pd.DataFrame(data = tf.toarray(),columns = count_tokens)\n    return train_test_split(train_countvect, train['Category'], test_size=0.33, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:28.057230Z","iopub.execute_input":"2022-08-06T23:43:28.057648Z","iopub.status.idle":"2022-08-06T23:43:28.066416Z","shell.execute_reply.started":"2022-08-06T23:43:28.057613Z","shell.execute_reply":"2022-08-06T23:43:28.065016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next we calculate the tfidf weights for the corpus of documents\n\nWe transform this into an array and return a training and validation set for supervised learning.","metadata":{}},{"cell_type":"code","source":"def tfidf_vectorization(n_features):\n    tfidf_vectorizer = TfidfVectorizer(max_df=0.95, min_df=2, max_features=n_features,stop_words=\"english\", lowercase = True)\n    tf = tfidf_vectorizer.fit_transform(train['Text'])\n    tokens = tfidf_vectorizer.get_feature_names_out()\n    train_vect = pd.DataFrame(data = tf.toarray(),columns = tokens)\n    return train_test_split(train_vect, train['Category'], test_size=0.33, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:28.068156Z","iopub.execute_input":"2022-08-06T23:43:28.068579Z","iopub.status.idle":"2022-08-06T23:43:28.078491Z","shell.execute_reply.started":"2022-08-06T23:43:28.068553Z","shell.execute_reply":"2022-08-06T23:43:28.077614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The below code runs ETL on the vector of words. However, because of the number of words it is unweildy and can only be run for a few words at a time. \n\nThe inspiration was to test correlation to the category for each word and to see which words had high correlation to each other.","metadata":{}},{"cell_type":"code","source":"#n_features = 5\n#X_train_s, X_valid_s, y_train_s, y_valid_s = count_vectorization(n_features)\n#supervisedLearningData = pd.concat([X_train_s, y_train_s], axis=\"columns\")\n#profile = pp.ProfileReport(df_countvect, title=\"Pandas Profiling Report\", explorative=True)\n#profile.to_file(\"third_profile.html\")","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:28.079915Z","iopub.execute_input":"2022-08-06T23:43:28.080810Z","iopub.status.idle":"2022-08-06T23:43:28.092417Z","shell.execute_reply.started":"2022-08-06T23:43:28.080782Z","shell.execute_reply":"2022-08-06T23:43:28.091320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next we calculate the tfidf weights for the corpus of documents\n\nThis is the dataset we use for the nmf unsupervised learning model","metadata":{}},{"cell_type":"code","source":"def tfidf(X_train, X_valid):\n    tfidf_vectorizer = TfidfVectorizer(max_df=0.95, min_df=2, stop_words=\"english\", lowercase = True)\n    tfidf_vectorizer.fit(X_train)\n    tfidf_train = tfidf_vectorizer.transform(X_train)\n    tfidf_valid = tfidf_vectorizer.transform(X_valid)\n    tokens = tfidf_vectorizer.get_feature_names_out()\n    train_vect = pd.DataFrame(data = tfidf_train.toarray(),columns = tokens)\n    valid_vect = pd.DataFrame(data = tfidf_valid.toarray(),columns = tokens)\n    return tfidf_train, tfidf_valid, train_vect, valid_vect\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:28.093815Z","iopub.execute_input":"2022-08-06T23:43:28.094082Z","iopub.status.idle":"2022-08-06T23:43:28.104503Z","shell.execute_reply.started":"2022-08-06T23:43:28.094057Z","shell.execute_reply":"2022-08-06T23:43:28.103629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.2 MODEL","metadata":{}},{"cell_type":"markdown","source":"Here we initiate the model and its params","metadata":{}},{"cell_type":"code","source":"def model(mtype, stype):\n    nmf = NMF(\n        init = 'nndsvdar',\n        n_components=5,\n        solver = stype,\n        random_state=1,\n        beta_loss=mtype,\n        alpha_W=0.00005,\n        alpha_H=0.00005,\n        l1_ratio=.1,\n    )\n    return nmf","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:28.106038Z","iopub.execute_input":"2022-08-06T23:43:28.106340Z","iopub.status.idle":"2022-08-06T23:43:28.120952Z","shell.execute_reply.started":"2022-08-06T23:43:28.106307Z","shell.execute_reply":"2022-08-06T23:43:28.120051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here is a line of code to predict the outcome.","metadata":{}},{"cell_type":"code","source":"def predict(matrix):\n    sortedMatrix = np.argsort(matrix)\n    n_predictions, maxValue = sortedMatrix.shape\n    \n    predictions = [[sortedMatrix[i][maxValue - 1]] for i in range(n_predictions)]\n    topics = np.empty(n_predictions, dtype = np.int64)\n    \n    for i in range(n_predictions):\n        topics[i] = predictions[i][0]\n    return topics","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:28.122581Z","iopub.execute_input":"2022-08-06T23:43:28.122853Z","iopub.status.idle":"2022-08-06T23:43:28.134972Z","shell.execute_reply.started":"2022-08-06T23:43:28.122819Z","shell.execute_reply":"2022-08-06T23:43:28.133818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is a function to identify the accuracy based on the best permutation of categories. ","metadata":{}},{"cell_type":"code","source":"def label_permute(ytdf,yp,perm_list, n=5):\n    \"\"\"\n    ytdf: labels dataframe object\n    yp: clustering label prediction output\n    Returns permuted label order and accuracy. \n    Example output: (3, 4, 1, 2, 0), 0.74 \n    \"\"\"\n    # your code here\n    unique = np.unique(ytdf)\n    perm = itertools.permutations(perm_list)\n    accuracyMatrix = []\n    for i in list(perm):\n        j = 0\n        ytdf_guess = ytdf\n        for k in unique:\n            ytdf_guess = np.where(ytdf_guess == k, i[j], ytdf_guess)\n            j+=1\n        pred = ytdf_guess.tolist()\n\n        accuracy = accuracy_score(pred,yp)\n        accuracyMatrix.append((i,accuracy))\n    i = 0\n    maxAccuracy = 0\n    while i < len(accuracyMatrix):\n        if accuracyMatrix[i][1] > maxAccuracy:\n            maxtuple = accuracyMatrix[i]\n            maxAccuracy = accuracyMatrix[i][1]\n        i += 1\n    \n    print(maxtuple)\n    return maxtuple","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:28.137817Z","iopub.execute_input":"2022-08-06T23:43:28.138812Z","iopub.status.idle":"2022-08-06T23:43:28.152673Z","shell.execute_reply.started":"2022-08-06T23:43:28.138781Z","shell.execute_reply":"2022-08-06T23:43:28.151394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Belo we calculate the tfidf vector and array","metadata":{}},{"cell_type":"code","source":"tfidf_train, tfidf_valid, train_vect, valid_vect = tfidf(X_train, X_valid)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:28.154842Z","iopub.execute_input":"2022-08-06T23:43:28.155285Z","iopub.status.idle":"2022-08-06T23:43:28.724614Z","shell.execute_reply.started":"2022-08-06T23:43:28.155247Z","shell.execute_reply":"2022-08-06T23:43:28.723573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 4 models run in total. 2 for each loss function and one that predicts the validation set while the other predicts the training data. ","metadata":{}},{"cell_type":"code","source":"labels = [0,1,2,3,4]\nm = model(\"frobenius\", \"cd\").fit(tfidf_train)\nprint(f'trying with {m}')\nyhat = predict(m.transform(tfidf_valid))\nlabel_order, accuracy = label_permute(y_valid, yhat,labels)\n\nprint(label_order)\nprint(accuracy)\n\nm = model(\"kullback-leibler\", \"mu\").fit(tfidf_train)\nprint(f'trying with {m}')\n\nyhat_valid = predict(m.transform(tfidf_valid))\nlabel_order, accuracy = label_permute(y_valid, yhat, labels)\n\nprint(label_order)\nprint(accuracy)\n\nm = model(\"frobenius\", \"mu\").fit(tfidf_train)\nprint(f'trying with {m}')\nyhat = predict(m.transform(tfidf_train))\nlabel_order, accuracy = label_permute(y_train, yhat,labels)\n\nprint(label_order)\nprint(accuracy)\n\nm = model(\"kullback-leibler\", \"mu\").fit(tfidf_train)\nprint(f'trying with {m}')\nyhat = predict(m.transform(tfidf_train))\nlabel_order, accuracy = label_permute(y_train, yhat,labels)\nprint(label_order)\nprint(accuracy)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:28.725812Z","iopub.execute_input":"2022-08-06T23:43:28.726084Z","iopub.status.idle":"2022-08-06T23:43:35.073165Z","shell.execute_reply.started":"2022-08-06T23:43:28.726061Z","shell.execute_reply":"2022-08-06T23:43:35.072182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.3 COMPARISON","metadata":{}},{"cell_type":"markdown","source":"Next I predicted using a kmeans clustering algorihtm as I had the code already written from a previous project.","metadata":{}},{"cell_type":"code","source":"def kclass():\n    kmodel = KMeans(n_clusters=5)\n    kmodel.fit(tfidf_train)\n    return kmodel","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:35.075976Z","iopub.execute_input":"2022-08-06T23:43:35.077922Z","iopub.status.idle":"2022-08-06T23:43:35.084723Z","shell.execute_reply.started":"2022-08-06T23:43:35.077880Z","shell.execute_reply":"2022-08-06T23:43:35.082593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kmodel = kclass()\nyhat = kmodel.predict(tfidf_valid)\nlabelorder, acc = label_permute(y_valid, yhat,labels)\n#confusion(label, kmeans.labels_,labelorder)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:35.086496Z","iopub.execute_input":"2022-08-06T23:43:35.087091Z","iopub.status.idle":"2022-08-06T23:43:35.888919Z","shell.execute_reply.started":"2022-08-06T23:43:35.087052Z","shell.execute_reply":"2022-08-06T23:43:35.887457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below I run supervised training algorithm, a gradient boosting classifier. I use both types of array. The count as well as teh tfidf to see which one is more accurate dataset. ","metadata":{}},{"cell_type":"code","source":"n_features = 500\nX_train_s_c, X_valid_s_c, y_train_s_c, y_valid_s_c = count_vectorization(n_features)\nX_train_s, X_valid_s, y_train_s, y_valid_s = tfidf_vectorization(n_features)\nclf = GradientBoostingClassifier(n_estimators=100, learning_rate=0.1, max_depth=5, random_state=0)\nclf.fit(X_train_s, y_train_s)\nprint(clf.score(X_valid_s, y_valid_s))\nclf = GradientBoostingClassifier(n_estimators=500, learning_rate=1, max_depth=5, random_state=0)\nclf.fit(X_train_s_c, y_train_s_c)\nprint(clf.score(X_valid_s, y_valid_s))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:35.890022Z","iopub.execute_input":"2022-08-06T23:43:35.890407Z","iopub.status.idle":"2022-08-06T23:43:58.477374Z","shell.execute_reply.started":"2022-08-06T23:43:35.890382Z","shell.execute_reply":"2022-08-06T23:43:58.476298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part 2. Limitation(s) of sklearn’s non-negative matrix factorization library.\n\nIn this part I will use nnm to predict movie ratings and discuss the limitations of nnm.","metadata":{}},{"cell_type":"code","source":"movies = pd.read_csv('../input/moviesreal/movies.csv')\ntrain = pd.read_csv('../input/moviesreal/train.csv')\ntest = pd.read_csv('../input/moviesreal/test.csv')\nusers = pd.read_csv('../input/moviesreal/users.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:58.478698Z","iopub.execute_input":"2022-08-06T23:43:58.480002Z","iopub.status.idle":"2022-08-06T23:43:58.655870Z","shell.execute_reply.started":"2022-08-06T23:43:58.479944Z","shell.execute_reply":"2022-08-06T23:43:58.654807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.1 MODEL and RMSE","metadata":{}},{"cell_type":"markdown","source":"Below I create a training and validation set for the movie data and run the nmf models to predict the rating.\n\nI use RMSE to evaluate each prediction. The inuition behind RMSE is how far off from the real solution I am.\n\nFor example, the RMSE ranges from 2 to 3 across the models. This means I that I could be up to 3 rating levels away from the true rating.","metadata":{}},{"cell_type":"code","source":"data = pd.concat([train, movies], axis=\"columns\", join = 'inner')\ny = data.pop('rating')\ndata.drop('title', axis =1, inplace = True)\n\nX_train_movie, X_valid_movie, y_train_movie, y_valid_movie = train_test_split(data, y, test_size=0.33, random_state=42)\n\ncomponents = [1,2,3,4,5]\nprint(y_train_movie.unique())\n\nm = model(\"frobenius\", \"cd\").fit(X_train_movie)\nprint(f'trying with {m}')\nyhat = predict(m.transform(X_valid_movie))\nlabel_order, accuracy = label_permute(y_valid_movie, yhat,components)\nrmse = mean_squared_error(y_valid_movie, yhat, squared=False)\n\nprint(f'RMSE {rmse}')\n      \nm = model(\"kullback-leibler\", \"mu\").fit(X_train_movie)\nprint(f'trying with {m}')\nyhat_valid = predict(m.transform(X_valid_movie))\nlabel_order, accuracy = label_permute(y_valid_movie, yhat,components)\nrmse = mean_squared_error(y_valid_movie, yhat, squared=False)\n\nprint(f'RMSE {rmse}')\n\nm = model(\"frobenius\", \"mu\").fit(X_train_movie)\nprint(f'trying with {m}')\nyhat = predict(m.transform(X_train_movie))\nlabel_order, accuracy = label_permute(y_train_movie, yhat,components)\nrmse = mean_squared_error(y_train_movie, yhat, squared=False)\n\nprint(f'RMSE {rmse}')\n\nm = model(\"kullback-leibler\", \"mu\").fit(X_train_movie)\nprint(f'trying with {m}')\nyhat = predict(m.transform(X_train_movie))\nlabel_order, accuracy = label_permute(y_train_movie, yhat,components)\nrmse = mean_squared_error(y_train_movie, yhat, squared=False)\n\nprint(f'RMSE {rmse}')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:43:58.657002Z","iopub.execute_input":"2022-08-06T23:43:58.657428Z","iopub.status.idle":"2022-08-06T23:44:00.424262Z","shell.execute_reply.started":"2022-08-06T23:43:58.657402Z","shell.execute_reply":"2022-08-06T23:44:00.422612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.2 Explanation of why NMF didnt succeed. \n\nNMF does best is sparce matrices. The movie ratings did not have enough information in the feature to come up with a good solution.","metadata":{}},{"cell_type":"markdown","source":"Below I test supervised learning on the dataset to see a better outcome. ","metadata":{}},{"cell_type":"code","source":"clf = GradientBoostingClassifier(n_estimators=5000, learning_rate=0.01, max_depth=5, random_state=0)\nclf.fit(X_train_movie, y_train_movie)\nprint(clf.score(X_valid_movie, y_valid_movie))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T23:44:00.425505Z","iopub.execute_input":"2022-08-06T23:44:00.425927Z","iopub.status.idle":"2022-08-06T23:47:07.355832Z","shell.execute_reply.started":"2022-08-06T23:44:00.425901Z","shell.execute_reply":"2022-08-06T23:47:07.354813Z"},"trusted":true},"execution_count":null,"outputs":[]}]}