{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center><h1> Analytics of phrases skewness according to markdown position </h1></center>\n\n\n<center><img width=\"500px\" src=\"https://www.allaboutcircuits.com/uploads/articles/understanding-the-normal-distribution-parametric-tests-skewness-and-kurtosis-rk-aac-image2.jpg\"></center>\n\n\n**We will explore which phreses are more likely to occur in certain parts of the notebook**","metadata":{"execution":{"iopub.status.busy":"2022-07-22T04:37:58.935802Z","iopub.execute_input":"2022-07-22T04:37:58.936394Z","iopub.status.idle":"2022-07-22T04:37:58.946893Z","shell.execute_reply.started":"2022-07-22T04:37:58.936345Z","shell.execute_reply":"2022-07-22T04:37:58.945287Z"}}},{"cell_type":"code","source":"# Libs\nfrom nltk import ngrams\nimport re\nimport nltk\nimport seaborn as sns\nimport numpy as np\nfrom scipy import spatial\nimport matplotlib.pyplot as plt\nfrom sklearn.manifold import TSNE\nimport json\nimport inspect\nfrom pathlib import Path\nimport pandas as pd\nfrom tqdm.auto import tqdm\ntqdm.pandas()\nimport numpy as np\nfrom scipy import spatial\nimport matplotlib.pyplot as plt\nfrom sklearn.manifold import TSNE\nimport torch\nfrom scipy.stats import skew\nimport string\nplt.rcParams['figure.dpi'] = 300\nsns.despine()\nsns.set_style(\"white\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T04:41:47.548915Z","iopub.execute_input":"2022-07-22T04:41:47.549462Z","iopub.status.idle":"2022-07-22T04:41:47.562932Z","shell.execute_reply.started":"2022-07-22T04:41:47.549421Z","shell.execute_reply":"2022-07-22T04:41:47.561895Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_df_orders_and_ranks(df, data_dir):\n    # train orders\n    df_orders = pd.read_csv(\n      data_dir / 'train_orders.csv',\n      index_col='id',\n      squeeze=True,\n    ).str.split()  # cell_ids str -> list\n\n\n    df_orders_ = df_orders.to_frame().join(\n      # reset only one index out of many -> \"cell_id\"; make a list out of cells in train data\n      df.reset_index('cell_id').groupby('id')['cell_id'].apply(list),\n      how='right',\n    )\n\n    ranks = {}\n    for id_, cell_order, cell_id in df_orders_.itertuples():\n        ranks[id_] = {'cell_id': cell_id, 'rank': get_ranks(cell_order, cell_id)}\n\n    df_ranks = (\n      pd.DataFrame\n      .from_dict(ranks, orient='index')\n      .rename_axis('id')\n      .apply(pd.Series.explode)\n      .set_index('cell_id', append=True)\n    )\n    # now we have\n    # id cell_id rank\n    return df_orders, df_ranks\n\ndef get_ranks(base, derived):\n    return [base.index(d) for d in derived]\n\ndef read_train_data(data_dir, NUM_TRAIN = 10000):\n    def read_notebook(path):\n        return (\n            pd.read_json(\n                path,\n                dtype={'cell_type': 'category', 'source': 'str'})\n            .assign(id=path.stem)  # final path component\n            .rename_axis('cell_id')\n        )\n\n    paths_train = list((data_dir / 'train').glob('*.json'))[:NUM_TRAIN]\n    notebooks_train = [\n      read_notebook(path) for path in tqdm(paths_train, desc='Train NBs')\n    ]\n    df = (\n      pd.concat(notebooks_train)\n      .set_index('id', append=True)\n      .swaplevel()\n      .sort_index(level='id', sort_remaining=False)\n    )\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-22T04:01:36.366439Z","iopub.execute_input":"2022-07-22T04:01:36.367018Z","iopub.status.idle":"2022-07-22T04:01:36.384201Z","shell.execute_reply.started":"2022-07-22T04:01:36.366968Z","shell.execute_reply":"2022-07-22T04:01:36.383019Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = Path('../input/AI4Code')\ndf = read_train_data(data_dir, NUM_TRAIN=15000)\ndf_orders, df_ranks = get_df_orders_and_ranks(df, data_dir)\nprint(f\"Df shape is {df.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-22T04:44:40.142348Z","iopub.execute_input":"2022-07-22T04:44:40.142797Z","iopub.status.idle":"2022-07-22T04:47:43.460584Z","shell.execute_reply.started":"2022-07-22T04:44:40.142759Z","shell.execute_reply":"2022-07-22T04:47:43.459428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"def clean_text(text):\n    '''Make text lowercase, remove text in square brackets,remove links,remove punctuation'''\n    if not text:\n        return ''\n    text = text.lower()\n    text = text.strip()\n    text = ' '.join(text.split())\n    text = re.sub('<.*?>+', ' ', text)\n    text = text.replace('[' , ' ')\n    text = text.replace(']' , ' ')\n    text = re.sub('http.?://\\S+|www\\.\\S+', '[LINK]', text)\n    text = re.sub('[%s]' % re.escape(string.punctuation), ' ', text)\n    text = re.sub('\\n', '. ', text)\n    text = re.sub(' +', ' ', text)\n    text = text.strip()\n    return text\n\ndef text_preprocessing(text):\n    \"\"\"\n    Cleaning and parsing the text.\n\n    \"\"\"\n    tokenizer = nltk.tokenize.RegexpTokenizer(r'\\w+')\n    nopunc = clean_text(text)\n    tokenized_text = tokenizer.tokenize(nopunc)\n    combined_text = ' '.join(tokenized_text)\n    return combined_text","metadata":{"execution":{"iopub.status.busy":"2022-07-22T04:47:43.463372Z","iopub.execute_input":"2022-07-22T04:47:43.464393Z","iopub.status.idle":"2022-07-22T04:47:43.473930Z","shell.execute_reply.started":"2022-07-22T04:47:43.464342Z","shell.execute_reply":"2022-07-22T04:47:43.472780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['rank'] = df.reset_index().merge(df_ranks, on=[\"id\", \"cell_id\"])['rank'].values\ndf = df.reset_index()\ndf[\"pct_rank\"] = df[\"rank\"] / df.groupby(\"id\")[\"cell_id\"].transform(\"count\")","metadata":{"execution":{"iopub.status.busy":"2022-07-22T04:47:43.475250Z","iopub.execute_input":"2022-07-22T04:47:43.475602Z","iopub.status.idle":"2022-07-22T04:47:45.362118Z","shell.execute_reply.started":"2022-07-22T04:47:43.475562Z","shell.execute_reply":"2022-07-22T04:47:45.360985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets split to get only markdowns\nmarkdowns = df[df['cell_type'] == 'markdown'].reset_index()\nmarkdowns['source'] = markdowns.progress_apply(lambda x: text_preprocessing(x['source']), axis=1).values","metadata":{"execution":{"iopub.status.busy":"2022-07-22T04:47:45.364686Z","iopub.execute_input":"2022-07-22T04:47:45.365038Z","iopub.status.idle":"2022-07-22T04:48:01.086816Z","shell.execute_reply.started":"2022-07-22T04:47:45.365006Z","shell.execute_reply":"2022-07-22T04:48:01.085247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lets get all possible phrases combinations from text","metadata":{}},{"cell_type":"code","source":"values = markdowns['source'].values\ntop_starts = []\nn=3  # number of words in a combined sentence ngram\nfor value in tqdm(values):\n    for x in ngrams(value.lower().split(), n):\n        top_starts.append(' '.join(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T04:48:01.088606Z","iopub.execute_input":"2022-07-22T04:48:01.089821Z","iopub.status.idle":"2022-07-22T04:48:04.916213Z","shell.execute_reply.started":"2022-07-22T04:48:01.089779Z","shell.execute_reply":"2022-07-22T04:48:04.914909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# What are the top markdown starts?","metadata":{}},{"cell_type":"code","source":"counts = pd.Series(top_starts).value_counts().sort_values(ascending=False)\nsns.barplot(y=counts[:20].index, x=counts[:20])\nplt.xticks(rotation=45)\nplt.gcf().set_size_inches(7, 5)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T04:48:04.917367Z","iopub.execute_input":"2022-07-22T04:48:04.917678Z","iopub.status.idle":"2022-07-22T04:48:14.828269Z","shell.execute_reply.started":"2022-07-22T04:48:04.917649Z","shell.execute_reply":"2022-07-22T04:48:14.826849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# What phrases out of those are most skewed to the left or right?","metadata":{}},{"cell_type":"code","source":"skewness = {}\ndistributions = {}\nfor v in tqdm(counts[counts > 500].index):  # basic filtering\n    try:\n        \n        skewness[v] = skew(\n            markdowns[markdowns['source'].str.contains(v)]['pct_rank'].values\n        )\n        distributions[v] = markdowns[markdowns['source'].str.contains(v)]['pct_rank'].values\n    except:\n        # division by zero\n        pass\ntop_skewed_keys = {\n    k: v for k, v in reversed(sorted(skewness.items(), key=lambda item: abs(item[1])))\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-22T04:51:36.513402Z","iopub.execute_input":"2022-07-22T04:51:36.514656Z","iopub.status.idle":"2022-07-22T04:52:32.300995Z","shell.execute_reply.started":"2022-07-22T04:51:36.514613Z","shell.execute_reply":"2022-07-22T04:52:32.299621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins = np.array([x/100 for x in range(0, 100, 5)])\nfor key in list(top_skewed_keys.keys())[:5]:\n    inds = np.digitize(distributions[key], bins)\n    if skewness[key] < 0:\n        color='blue'\n    else:\n        color='green'\n    ax = sns.displot(inds, color=color, kind=\"kde\", fill=True, height=5, aspect=1.5)\n    ax.fig.set_dpi(120)\n    ax.fig.suptitle(f'«{key}» = {skewness[key]:.2f} skewness')\n    ax.set(xlabel='Percentile', ylabel='Density')\n    plt.xticks(np.arange(0, len(bins)+1, 2))\n    ax.set_xticklabels([str(5*x) + '%' for x in np.arange(0, len(bins)+1, 2)])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T04:58:11.556478Z","iopub.execute_input":"2022-07-22T04:58:11.557466Z","iopub.status.idle":"2022-07-22T04:58:13.383900Z","shell.execute_reply.started":"2022-07-22T04:58:11.557421Z","shell.execute_reply":"2022-07-22T04:58:13.382763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"If you like you know what to do ❤️","metadata":{}}]}