{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#                 ** EXPLORATORY DATA ANALYSIS **","metadata":{}},{"cell_type":"markdown","source":"## INDEX\n\n### 1. Dependencies \n### 2. Look & Clean Up of Data\n### 3. Detailed Analysis\n### 4. Text Data Processing\n### 5. Text Data Statistics \n### 6. N-GRAMS\n### 7. Word Cloud\n### 8. Insights","metadata":{}},{"cell_type":"markdown","source":"## 1. Dependencies ","metadata":{}},{"cell_type":"code","source":"# General Libraries \n\nimport os\nimport json\nimport numpy as np # \nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns # Seaborn is the only library we need to import for this example. By convention, it is imported with the shorthand sns\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:02:55.602335Z","iopub.execute_input":"2022-07-07T05:02:55.602799Z","iopub.status.idle":"2022-07-07T05:02:55.612712Z","shell.execute_reply.started":"2022-07-07T05:02:55.602759Z","shell.execute_reply":"2022-07-07T05:02:55.607632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Text Processing Libraries \n\nimport re # re is a RegularExpreseeion module, this offers a set of functions that allows us to search a string for a match\nimport string # This module will help us  quickly access some string constants\nimport nltk # NLTK is library for building programs to work with human language data. \n#It provides over 50 corpora and lexical resources such as WordNet, along with a suite of text processing libraries for classification, tokenization, stemming, tagging, parsing, and semantic reasoning, wrappers for industrial-strength NLP libraries\nfrom nltk.corpus import stopwords\nimport wordcloud # A word cloud is a collection, or cluster, of words depicted in different sizes\n\nfrom wordcloud import WordCloud, STOPWORDS\nstopwords = set(STOPWORDS)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:03:02.127676Z","iopub.execute_input":"2022-07-07T05:03:02.128055Z","iopub.status.idle":"2022-07-07T05:03:02.886262Z","shell.execute_reply.started":"2022-07-07T05:03:02.128024Z","shell.execute_reply":"2022-07-07T05:03:02.885062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SKLearn Libraries \n\nfrom sklearn import model_selection\nfrom sklearn.feature_extraction.text import CountVectorizer,TfidfVectorizer","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:03:05.947831Z","iopub.execute_input":"2022-07-07T05:03:05.948239Z","iopub.status.idle":"2022-07-07T05:03:05.953156Z","shell.execute_reply.started":"2022-07-07T05:03:05.948205Z","shell.execute_reply":"2022-07-07T05:03:05.952145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting up the theme for plot visualization\n\nsns.set(style=\"whitegrid\", palette=\"muted\")\nsns.set(rc={'figure.figsize':(11.7,8.27)})","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:03:07.548822Z","iopub.execute_input":"2022-07-07T05:03:07.549603Z","iopub.status.idle":"2022-07-07T05:03:07.558519Z","shell.execute_reply.started":"2022-07-07T05:03:07.549552Z","shell.execute_reply":"2022-07-07T05:03:07.557056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Thanks to Darien Schettler for creating the dataframe, that is used in this notebook. This can be added based on to this notebook \n# Loading & Reading the dataset \n\ntrain = pd.read_csv(\"../input/ai4code-train-dataframe/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:03:09.544719Z","iopub.execute_input":"2022-07-07T05:03:09.545454Z","iopub.status.idle":"2022-07-07T05:04:03.284313Z","shell.execute_reply.started":"2022-07-07T05:03:09.545412Z","shell.execute_reply":"2022-07-07T05:04:03.283112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Look & Clean Up of Data","metadata":{}},{"cell_type":"code","source":"# View the data for first 10 rows\ntrain.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:03.286365Z","iopub.execute_input":"2022-07-07T05:04:03.286704Z","iopub.status.idle":"2022-07-07T05:04:03.308029Z","shell.execute_reply.started":"2022-07-07T05:04:03.286673Z","shell.execute_reply":"2022-07-07T05:04:03.306748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View the shape and size of the data \ntrain.shape, train.size","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:03.309322Z","iopub.execute_input":"2022-07-07T05:04:03.309612Z","iopub.status.idle":"2022-07-07T05:04:03.316989Z","shell.execute_reply.started":"2022-07-07T05:04:03.309585Z","shell.execute_reply":"2022-07-07T05:04:03.315922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Information of the data in terms of the field lenght and the type of attributes. There are 4 columns with over 6 million data \ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:03.319688Z","iopub.execute_input":"2022-07-07T05:04:03.320285Z","iopub.status.idle":"2022-07-07T05:04:03.346744Z","shell.execute_reply.started":"2022-07-07T05:04:03.320251Z","shell.execute_reply":"2022-07-07T05:04:03.345706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can further analyse to look at the unique values in each of the coumns \ntrain.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T14:31:02.767720Z","iopub.execute_input":"2022-07-06T14:31:02.768154Z","iopub.status.idle":"2022-07-06T14:31:12.964179Z","shell.execute_reply.started":"2022-07-06T14:31:02.768119Z","shell.execute_reply":"2022-07-06T14:31:12.963064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This will give the number of notebooks present in training and testing data set \nprint(f\"Number of notebooks present in train set: \",len(os.listdir(\"../input/AI4Code/train\")))\nprint(f\"Number of notebooks present in test set: \",len(os.listdir(\"../input/AI4Code/test\")))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:03.347791Z","iopub.execute_input":"2022-07-07T05:04:03.348200Z","iopub.status.idle":"2022-07-07T05:04:05.609979Z","shell.execute_reply.started":"2022-07-07T05:04:03.348171Z","shell.execute_reply":"2022-07-07T05:04:05.609141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let us look at the missing values in each of the column \ntrain.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:05.611264Z","iopub.execute_input":"2022-07-07T05:04:05.612339Z","iopub.status.idle":"2022-07-07T05:04:07.110257Z","shell.execute_reply.started":"2022-07-07T05:04:05.612296Z","shell.execute_reply":"2022-07-07T05:04:07.109122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We will drop  as there are only 4 missing values \ntrain = train.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:07.111554Z","iopub.execute_input":"2022-07-07T05:04:07.111839Z","iopub.status.idle":"2022-07-07T05:04:09.189002Z","shell.execute_reply.started":"2022-07-07T05:04:07.111814Z","shell.execute_reply":"2022-07-07T05:04:09.187766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# There are no further na in our data set\ntrain.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:09.190888Z","iopub.execute_input":"2022-07-07T05:04:09.191340Z","iopub.status.idle":"2022-07-07T05:04:10.593294Z","shell.execute_reply.started":"2022-07-07T05:04:09.191294Z","shell.execute_reply":"2022-07-07T05:04:10.592490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Detailed Analysis","metadata":{}},{"cell_type":"code","source":"# Let us look at the data in each of the column and do further analysis\n\ntrain['cell_type'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:10.594287Z","iopub.execute_input":"2022-07-07T05:04:10.595091Z","iopub.status.idle":"2022-07-07T05:04:10.993655Z","shell.execute_reply.started":"2022-07-07T05:04:10.595043Z","shell.execute_reply":"2022-07-07T05:04:10.992748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,6))\nsns.countplot(train['cell_type'], order = sorted(train['cell_type'].unique()))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:10.997177Z","iopub.execute_input":"2022-07-07T05:04:10.998451Z","iopub.status.idle":"2022-07-07T05:04:14.302273Z","shell.execute_reply.started":"2022-07-07T05:04:10.998416Z","shell.execute_reply":"2022-07-07T05:04:14.301029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution of the data from the code and markdown \nmarkdown = (train['cell_type'].value_counts()['markdown']) / train.shape[0] * 100\ncode = (train['cell_type'].value_counts()['code']) / train.shape[0] * 100\nprint(\"-\"*32)\nprint('Code ({0:.2f}%)'.format(code), 'Markdown ({0:.2f}%)'.format(markdown))\nprint(\"-\"*32)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:14.303953Z","iopub.execute_input":"2022-07-07T05:04:14.304663Z","iopub.status.idle":"2022-07-07T05:04:15.093478Z","shell.execute_reply.started":"2022-07-07T05:04:14.304617Z","shell.execute_reply":"2022-07-07T05:04:15.092299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let us look data available in one of the celltypes \n\nprint(\"-\"*100)\nprint(\"Code :\\n\",train[train['cell_type']=='code']['source'].values[0])\nprint(\"-\"*100)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:15.094719Z","iopub.execute_input":"2022-07-07T05:04:15.095133Z","iopub.status.idle":"2022-07-07T05:04:15.937119Z","shell.execute_reply.started":"2022-07-07T05:04:15.095104Z","shell.execute_reply":"2022-07-07T05:04:15.935850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"-\"*100)\nprint(\"Code :\\n\",train[train['cell_type']=='markdown']['source'].values[10])\nprint(\"-\"*100)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:15.938765Z","iopub.execute_input":"2022-07-07T05:04:15.939599Z","iopub.status.idle":"2022-07-07T05:04:16.592688Z","shell.execute_reply.started":"2022-07-07T05:04:15.939561Z","shell.execute_reply":"2022-07-07T05:04:16.591464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Text Data Processing \n\nCleaning the content of the data in cells and markdown. \nCreate and define functions to do the following \n- Convert text to lowercase\n- Remove numbers, punctuations, hyperlinks, stopwords\n- Tokenize the text ","metadata":{}},{"cell_type":"code","source":"# Helper Functions: A helper function is a function that performs part of the computation of another function following the DRY (Don't repeat yourself) concept.\n\ndef clean_text(text):\n    '''Make text lowercase, remove text in square brackets,remove links,remove punctuation\n    and remove words containing numbers.'''\n    text = text.lower()\n    text = text.strip()\n    text = re.sub('\\[.*?\\]', '', text)\n    text = re.sub('https?://\\S+|www\\.\\S+', '', text)\n    text = re.sub('<.*?>+', '', text)\n    text = re.sub('[%s]' % re.escape(string.punctuation), '', text)\n    text = re.sub('\\n', '', text)\n    text = re.sub('\\w*\\d\\w*', '', text)\n    return text\n\ndef text_preprocessing(text):\n    \"\"\"\n    Cleaning and parsing the text.\n\n    \"\"\"\n    tokenizer = nltk.tokenize.RegexpTokenizer(r'\\w+')\n    nopunc = clean_text(text)\n    tokenized_text = tokenizer.tokenize(nopunc)\n    combined_text = ' '.join(tokenized_text)\n    return combined_text\n\ndef clean_code(text):\n    '''Make text lowercase, remove text in square brackets,remove links,remove punctuation\n    and remove words containing numbers.'''\n    text = text.replace('[', ' ').replace(']', ' ').replace('(', ' ').replace(')', ' ').replace('{', ' ').replace('}', ' ').replace('=', ' ').replace(',', ' ')\n    text = text.lower()\n    text = text.replace('_', '')\n    text = text.replace('\\n', ' ')\n    text = text.replace('.', ' ')\n    text = re.sub(r'\".*\"', ' ', text)\n    text = re.sub(r\"'.*'\", ' ', text)\n    text = re.sub(\"^\\d+\\s|\\s\\d+\\s|\\s\\d+$\", ' ', text)\n    text = re.sub(' +', ' ', text)\n    text = text.strip()\n    return text\n\ndef code_preprocessing(text):\n    \"\"\"\n    Cleaning and parsing the text.\n\n    \"\"\"\n    tokenizer = nltk.tokenize.RegexpTokenizer(r'\\w+')\n    nopunc = clean_code(text)\n    tokenized_text = tokenizer.tokenize(nopunc)\n    combined_text = ' '.join(tokenized_text)\n    return combined_text","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:16.594708Z","iopub.execute_input":"2022-07-07T05:04:16.595179Z","iopub.status.idle":"2022-07-07T05:04:16.614008Z","shell.execute_reply.started":"2022-07-07T05:04:16.595135Z","shell.execute_reply":"2022-07-07T05:04:16.612468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create 2 columns for markdown & code\nmarkdowns = train[train['cell_type'] == 'markdown']\ncodes = train[train['cell_type'] == 'code']","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:16.615837Z","iopub.execute_input":"2022-07-07T05:04:16.616312Z","iopub.status.idle":"2022-07-07T05:04:17.978211Z","shell.execute_reply.started":"2022-07-07T05:04:16.616236Z","shell.execute_reply":"2022-07-07T05:04:17.977029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Applying helper functions to the data in markdown & code by creating a new column source_clean \ncodes['source_clean'] = codes['source'].apply(str).apply(lambda x: code_preprocessing(x))\nmarkdowns['source_clean'] = markdowns['source'].apply(str).apply(lambda x: text_preprocessing(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:04:17.981162Z","iopub.execute_input":"2022-07-07T05:04:17.981519Z","iopub.status.idle":"2022-07-07T05:08:06.539029Z","shell.execute_reply.started":"2022-07-07T05:04:17.981487Z","shell.execute_reply":"2022-07-07T05:08:06.537839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatinate clean text into train data \ntrain = pd.concat([codes, markdowns], ignore_index = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:08:06.540347Z","iopub.execute_input":"2022-07-07T05:08:06.540640Z","iopub.status.idle":"2022-07-07T05:08:07.400511Z","shell.execute_reply.started":"2022-07-07T05:08:06.540614Z","shell.execute_reply":"2022-07-07T05:08:07.399340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View the data now with source and the cleaned text \ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:08:07.402370Z","iopub.execute_input":"2022-07-07T05:08:07.402745Z","iopub.status.idle":"2022-07-07T05:08:07.416437Z","shell.execute_reply.started":"2022-07-07T05:08:07.402713Z","shell.execute_reply":"2022-07-07T05:08:07.415318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Text Data Statistics \n\n### This is purely done on the count and lenth of text and words present the data ","metadata":{}},{"cell_type":"code","source":"# creating 2 new columns to get the text_length and word_count for the clean source \ntrain['text_len'] = train['source_clean'].astype(str).apply(len)\ntrain['text_word_count'] = train['source_clean'].apply(lambda x: len(str(x).split()))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:08:07.419248Z","iopub.execute_input":"2022-07-07T05:08:07.419639Z","iopub.status.idle":"2022-07-07T05:08:19.553939Z","shell.execute_reply.started":"2022-07-07T05:08:07.419586Z","shell.execute_reply":"2022-07-07T05:08:19.552958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View the data \ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:08:19.555213Z","iopub.execute_input":"2022-07-07T05:08:19.555631Z","iopub.status.idle":"2022-07-07T05:08:19.567462Z","shell.execute_reply.started":"2022-07-07T05:08:19.555594Z","shell.execute_reply":"2022-07-07T05:08:19.566676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot graph for text length & as you can see most of them prefer to write short text \n# KDE A kernel density estimate (KDE) plot is a method for visualizing the distribution of observations in a dataset, analagous to a histogram. \n#KDE represents the data using a continuous probability density curve in one or more dimensions.\nx = train['text_len'].value_counts()\ntrain_sample = train[train['text_len'].isin(x[x>400].index)]\nsns.kdeplot(data=train_sample, x=\"text_len\", hue = 'cell_type', shade = True, multiple=\"stack\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:08:19.568567Z","iopub.execute_input":"2022-07-07T05:08:19.569359Z","iopub.status.idle":"2022-07-07T05:08:52.870108Z","shell.execute_reply.started":"2022-07-07T05:08:19.569326Z","shell.execute_reply":"2022-07-07T05:08:52.868924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# box plot (or box-and-whisker plot) shows the distribution of quantitative data in a way that facilitates comparisons between variables or across levels of a categorical variable.\n# The box shows the quartiles of the dataset while the whiskers extend to show the rest of the distribution, except for points that are determined to be “outliers” using a method that is a function of the inter-quartile range.\n#Length of Markdowns are longer than codes and median length of markdown is slightly above that of code \nsns.boxplot(data=train,x = 'cell_type', y=\"text_len\",  showfliers = False)\n#sns.swarmplot(data=train,x = 'cell_type', y=\"text_len\", color=\".25\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:08:52.871418Z","iopub.execute_input":"2022-07-07T05:08:52.871738Z","iopub.status.idle":"2022-07-07T05:08:55.920872Z","shell.execute_reply.started":"2022-07-07T05:08:52.871710Z","shell.execute_reply":"2022-07-07T05:08:55.919615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot graph for word counts for markdown and code\nx = train['text_word_count'].value_counts()\ntrain_sample = train[train['text_word_count'].isin(x[x>400].index)]\n\nsns.kdeplot(data=train_sample, x=\"text_word_count\", hue = 'cell_type', shade = True)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:08:55.922425Z","iopub.execute_input":"2022-07-07T05:08:55.923475Z","iopub.status.idle":"2022-07-07T05:09:27.066274Z","shell.execute_reply.started":"2022-07-07T05:08:55.923438Z","shell.execute_reply":"2022-07-07T05:09:27.064966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count of Markdowns are longer than codes and median length of markdown is slightly above that of code \nsns.boxplot(data=train,x = 'cell_type', y=\"text_word_count\", showfliers = False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:09:27.068128Z","iopub.execute_input":"2022-07-07T05:09:27.068506Z","iopub.status.idle":"2022-07-07T05:09:30.144799Z","shell.execute_reply.started":"2022-07-07T05:09:27.068474Z","shell.execute_reply":"2022-07-07T05:09:30.143498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. N-GRAMS\n\n### N-grams are continuous sequences of words or symbols or tokens in a document. In technical terms, they can be defined as the neighbouring sequences of items in a document. They come into play when we deal with text data in NLP(Natural Language Processing) tasks.\n\n### n-grams are classified into the following types, depending on the value that ‘n’ takes.\n\n1 - unigram\n2 - bigram\n3 - trigram \nn - ngram \n\n","metadata":{}},{"cell_type":"markdown","source":"### Let us take a look at the top 30 Unigrams in Code and Markdown","metadata":{}},{"cell_type":"code","source":"# Define a helper function to get the top words \n\ndef get_top_n_words(corpus, n=None):\n    \"\"\"\n    List the top n words in a vocabulary according to occurrence in a text corpus.\n    \"\"\"\n    vec = CountVectorizer(stop_words = 'english').fit(corpus)\n    bag_of_words = vec.transform(corpus)\n    sum_words = bag_of_words.sum(axis=0) \n    words_freq = [(word, sum_words[0, idx]) for word, idx in vec.vocabulary_.items()]\n    words_freq =sorted(words_freq, key = lambda x: x[1], reverse=True)\n    return words_freq[:n]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:09:30.146155Z","iopub.execute_input":"2022-07-07T05:09:30.147441Z","iopub.status.idle":"2022-07-07T05:09:30.154974Z","shell.execute_reply.started":"2022-07-07T05:09:30.147406Z","shell.execute_reply":"2022-07-07T05:09:30.153503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"markdowns = train[train['cell_type'] == 'markdown']\ncodes = train[train['cell_type'] == 'code']","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:09:30.156663Z","iopub.execute_input":"2022-07-07T05:09:30.156978Z","iopub.status.idle":"2022-07-07T05:09:31.857846Z","shell.execute_reply.started":"2022-07-07T05:09:30.156951Z","shell.execute_reply":"2022-07-07T05:09:31.856891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"codes_unigrams = get_top_n_words(codes['source_clean'],30)\n\ndata = pd.DataFrame(codes_unigrams, columns = ['Text' , 'count'])\ndata = data.groupby('Text').sum()['count'].sort_values(ascending=False)\ndata = pd.DataFrame(data).reset_index()\nsns.barplot(y = 'Text', x = 'count', data = data)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:09:31.861450Z","iopub.execute_input":"2022-07-07T05:09:31.861770Z","iopub.status.idle":"2022-07-07T05:11:35.799157Z","shell.execute_reply.started":"2022-07-07T05:09:31.861742Z","shell.execute_reply":"2022-07-07T05:11:35.797194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"markdowns_unigrams = get_top_n_words(markdowns['source_clean'],30)\n\ndata = pd.DataFrame(markdowns_unigrams, columns = ['Text' , 'count'])\ndata = data.groupby('Text').sum()['count'].sort_values(ascending=False)\ndata = pd.DataFrame(data).reset_index()\nsns.barplot(y = 'Text', x = 'count', data = data)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:11:35.802808Z","iopub.execute_input":"2022-07-07T05:11:35.803374Z","iopub.status.idle":"2022-07-07T05:13:14.579023Z","shell.execute_reply.started":"2022-07-07T05:11:35.803329Z","shell.execute_reply":"2022-07-07T05:13:14.577821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let us take a look at the top 30 Bigrams in Code and Markdown","metadata":{}},{"cell_type":"code","source":"# Define a helper function to get the top words \n\ndef get_top_n_words(corpus, ngram_range, n=None):\n    \n    vec = CountVectorizer(ngram_range=ngram_range, stop_words = 'english').fit(corpus)\n    bag_of_words = vec.transform(corpus)\n    sum_words = bag_of_words.sum(axis=0) \n    words_freq = [(word, sum_words[0, idx]) for word, idx in vec.vocabulary_.items()]\n    words_freq =sorted(words_freq, key = lambda x: x[1], reverse=True)\n    return words_freq[:n]\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:13:14.580835Z","iopub.execute_input":"2022-07-07T05:13:14.581556Z","iopub.status.idle":"2022-07-07T05:13:14.589398Z","shell.execute_reply.started":"2022-07-07T05:13:14.581512Z","shell.execute_reply":"2022-07-07T05:13:14.588478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"codes_bigrams = get_top_n_words(codes['source_clean'],(2,2),30)\n\ndata = pd.DataFrame(codes_bigrams, columns = ['Text' , 'count'])\ndata = data.groupby('Text').sum()['count'].sort_values(ascending=False)\ndata = pd.DataFrame(data).reset_index()\nsns.barplot(y = 'Text', x = 'count', data = data)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:13:14.590531Z","iopub.execute_input":"2022-07-07T05:13:14.591386Z","iopub.status.idle":"2022-07-07T05:17:15.987517Z","shell.execute_reply.started":"2022-07-07T05:13:14.591305Z","shell.execute_reply":"2022-07-07T05:17:15.986147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"markdowns_bigrams = get_top_n_words(markdowns['source_clean'],(2,2),30)\n\ndata = pd.DataFrame(markdowns_bigrams, columns = ['Text' , 'count'])\ndata = data.groupby('Text').sum()['count'].sort_values(ascending=False)\ndata = pd.DataFrame(data).reset_index()\nsns.barplot(y = 'Text', x = 'count', data = data)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:17:15.989319Z","iopub.execute_input":"2022-07-07T05:17:15.989630Z","iopub.status.idle":"2022-07-07T05:20:43.840791Z","shell.execute_reply.started":"2022-07-07T05:17:15.989602Z","shell.execute_reply":"2022-07-07T05:20:43.839304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"codes_trigrams = get_top_n_words(codes['source_clean'],(3,3),30)\n\ndata = pd.DataFrame(codes_trigrams, columns = ['Text' , 'count'])\ndata = data.groupby('Text').sum()['count'].sort_values(ascending=False)\ndata = pd.DataFrame(data).reset_index()\nsns.barplot(y = 'Text', x = 'count', data = data)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:20:43.844531Z","iopub.execute_input":"2022-07-07T05:20:43.845253Z","iopub.status.idle":"2022-07-07T05:25:30.990834Z","shell.execute_reply.started":"2022-07-07T05:20:43.845205Z","shell.execute_reply":"2022-07-07T05:25:30.989569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"markdowns_trigrams = get_top_n_words(markdowns['source_clean'],(3,3),30)\n\ndata = pd.DataFrame(markdowns_trigrams, columns = ['Text' , 'count'])\ndata = data.groupby('Text').sum()['count'].sort_values(ascending=False)\ndata = pd.DataFrame(data).reset_index()\nsns.barplot(y = 'Text', x = 'count', data = data)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:25:30.992878Z","iopub.execute_input":"2022-07-07T05:25:30.993662Z","iopub.status.idle":"2022-07-07T05:29:55.819518Z","shell.execute_reply.started":"2022-07-07T05:25:30.993616Z","shell.execute_reply":"2022-07-07T05:29:55.818315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Word Cloud\n\n### A word cloud (also known as a tag cloud) is a visual representation of words. Cloud creators are used to highlight popular words and phrases based on frequency and relevance. They provide quick and simple visual insights that can lead to more in-depth analysis","metadata":{}},{"cell_type":"code","source":"def show_wordcloud(data, title = None):\n    wordcloud = WordCloud(\n        background_color='white',\n        stopwords=stopwords,\n        max_words=2000000,\n        max_font_size=40, \n        scale=2,\n        random_state=1).generate(str(data))\n\n    fig = plt.figure(1, figsize=(14, 14))\n    plt.axis('off')\n    if title: \n        fig.suptitle(title, fontsize=20)\n        fig.subplots_adjust(top=2.3)\n\n    plt.imshow(wordcloud)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:29:55.821340Z","iopub.execute_input":"2022-07-07T05:29:55.822020Z","iopub.status.idle":"2022-07-07T05:29:55.830540Z","shell.execute_reply.started":"2022-07-07T05:29:55.821974Z","shell.execute_reply":"2022-07-07T05:29:55.829542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Word Cloud for Codes\")\nshow_wordcloud(codes['source_clean'].values)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:29:55.831829Z","iopub.execute_input":"2022-07-07T05:29:55.832153Z","iopub.status.idle":"2022-07-07T05:29:56.280105Z","shell.execute_reply.started":"2022-07-07T05:29:55.832124Z","shell.execute_reply":"2022-07-07T05:29:56.278851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nWord Cloud for Markdowns\")\nshow_wordcloud(markdowns['source_clean'].values)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T05:29:56.281728Z","iopub.execute_input":"2022-07-07T05:29:56.282568Z","iopub.status.idle":"2022-07-07T05:29:56.566458Z","shell.execute_reply.started":"2022-07-07T05:29:56.282528Z","shell.execute_reply":"2022-07-07T05:29:56.565388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8. Insights ","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}