{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T14:04:16.544992Z","iopub.execute_input":"2022-07-31T14:04:16.545651Z","iopub.status.idle":"2022-07-31T14:04:16.564406Z","shell.execute_reply.started":"2022-07-31T14:04:16.545594Z","shell.execute_reply":"2022-07-31T14:04:16.563152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Exploratory data analysis for article classification\n\nThis project is a part of the unsupervised classification lab for the course Unsupervised Algorithms in Machine Learning. My approach is to the term frequency-inverse document frequency (TF-IDF) to classify the documents by topic. Then, with visual inspection of the top 20 words assign the order category for the eigen value matrix. To test the model, I separated the data for training and testing in a proportion of 2 to 1 (70%, 30%) respectively. Then, I played a little bit with the hyperparameter of number of features, solver and betaloss function to see if I could improve my results. My result im-proves a significant amount (5%) when I chose the mu solver and the kullback-leibler betaloss func-tion with respect of the default values. Finally, the first part of this notebook treats the explorative data analysis and the second part you will find the model fits and plots of the accuracy.\n\n### 1.1- Importing all libraries and datasets","metadata":{"execution":{"iopub.status.busy":"2022-07-31T12:53:40.203798Z","iopub.execute_input":"2022-07-31T12:53:40.204198Z","iopub.status.idle":"2022-07-31T12:53:40.215238Z","shell.execute_reply.started":"2022-07-31T12:53:40.204166Z","shell.execute_reply":"2022-07-31T12:53:40.213821Z"}}},{"cell_type":"code","source":"import pandas as pd\nimport sys, os\nfrom IPython.display import display, Markdown\nimport numpy as np\nimport csv\nimport string \nfrom wordcloud import WordCloud\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\nfrom sklearn.decomposition import NMF, LatentDirichletAllocation\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sn\n\nsys.path.append(os.getcwd())\nbasedir = \"/kaggle/input/learn-ai-bbc/\"\ndata_for_predict = pd.read_csv(basedir + \"BBC News Test.csv\")\ndata_train_whole = pd.read_csv( basedir + \"BBC News Train.csv\")\ndata_train, data_test = train_test_split(data_train_whole, test_size=0.3)\nsample_solution = pd.read_csv(basedir + \"BBC News Sample Solution.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:16.607686Z","iopub.execute_input":"2022-07-31T14:04:16.608993Z","iopub.status.idle":"2022-07-31T14:04:16.695815Z","shell.execute_reply.started":"2022-07-31T14:04:16.608952Z","shell.execute_reply":"2022-07-31T14:04:16.694564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.2- Getting summaries to check for null values and counts","metadata":{}},{"cell_type":"code","source":"display(Markdown('**Executive summary for the test dataset**'))\ndata_test.info(show_counts=True)\n\ndisplay(Markdown('<br> **Executive summary for the training dataset**'))\ndata_train.info(show_counts=True)\n### 1.3- Reviewing the head of the datasets","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:16.700612Z","iopub.execute_input":"2022-07-31T14:04:16.701313Z","iopub.status.idle":"2022-07-31T14:04:16.729257Z","shell.execute_reply.started":"2022-07-31T14:04:16.701265Z","shell.execute_reply":"2022-07-31T14:04:16.728007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.3- Reviewing the head of the datasets","metadata":{}},{"cell_type":"code","source":"display(Markdown('**head for the test dataset**'))\nprint(data_test.head(5))\n### 3 -- More verification for empty strings\ndisplay(Markdown('<br> **head for the training  dataset**'))\nprint(data_train.head(5))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:16.731364Z","iopub.execute_input":"2022-07-31T14:04:16.732389Z","iopub.status.idle":"2022-07-31T14:04:16.749395Z","shell.execute_reply.started":"2022-07-31T14:04:16.732345Z","shell.execute_reply":"2022-07-31T14:04:16.748131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3 -- More verification for empty strings","metadata":{}},{"cell_type":"code","source":"display(Markdown('**Further data verification: test dataset**'))\nprint(\"Checking for empty strings, number empty string: %s\" % len(np.where(data_train.applymap(lambda x: x == ''))[1]),\n      \"\\nChecking for unique values %s and total value are %s \"%  (len(pd.unique(data_test[\"ArticleId\"])), data_test.shape[0]))\n\n\ndisplay(Markdown('**Further data verification: training dataset**'))\nprint(\"Checking for empty strings, number empty string: %s\" % len(np.where(data_train.applymap(lambda x: x == ''))[1]),\n      \"\\nChecking for unique values %s and total value are %s \"%  (len(pd.unique(data_test[\"ArticleId\"])), data_train.shape[0]))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:16.763794Z","iopub.execute_input":"2022-07-31T14:04:16.764265Z","iopub.status.idle":"2022-07-31T14:04:16.784341Z","shell.execute_reply.started":"2022-07-31T14:04:16.764227Z","shell.execute_reply":"2022-07-31T14:04:16.783447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.4- formating the text for processing\n<ol> In this case, we will remove all punctuation and digits present in the data. Also, a word cloud is com-puted to see the most common words in the articles.","metadata":{}},{"cell_type":"code","source":"## converting all strings in lower case \ndef format_text(dataseries):\n    # removes puntuation and digits\n    #input:\n    #   dataseries = pandas dataseries \n    #output:\n    #   dataseries = pandas dataseries\n    dataseries = dataseries.str.lower()\n    dataseries = dataseries.replace(r'[{}]'.format(string.punctuation), '', regex=True)\n    dataseries = dataseries.replace(r'[{}]'.format(string.digits), '', regex=True)\n    #dataseries= dataseries.str.replace(r'\\b\\w\\b', '', regex=True).str.replace(r'\\s+', ' ', regex=True)\n    return dataseries\n\ndisplay(Markdown('**Checking an article with puntuation and digits**'))\nprint(data_train['Text'][5])\ndata_train['review'] = format_text(data_train['Text'])\nword_cloud = WordCloud(width=900,height=500, max_words=1500,relative_scaling=1,normalize_plurals=False).generate(''.join(data_train['review']))\ndisplay(Markdown('**Check an article without puntuation and digits**'))\nprint(data_train['review'][5])\nplt.imshow(word_cloud, interpolation='bilinear')\nplt.axis(\"off\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:16.811369Z","iopub.execute_input":"2022-07-31T14:04:16.812035Z","iopub.status.idle":"2022-07-31T14:04:23.021318Z","shell.execute_reply.started":"2022-07-31T14:04:16.811999Z","shell.execute_reply":"2022-07-31T14:04:23.019914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.5- checking article frequency by category\n<ol> Here, I want to find out whether a class is not balance using a histogram visualization","metadata":{}},{"cell_type":"code","source":"histogram_computation = data_train.groupby(\"Category\")[\"ArticleId\"].nunique()\nhistogram_computation.columns= [\"Category\", 'vals']\nplt.rcParams['font.sans-serif'] = ['Tahoma']\nc = ['#00bfff', '#0099cc', '#007399', '#004d66', '#002633']\nax = histogram_computation.plot.bar(x='Category', y='vals', rot=45, color=c,fontsize=12, ylabel=\"frequency\",)\nax.figure.suptitle('Histogram of articles/categories', fontsize=20)\nax.figure.patch.set_facecolor('silver')\nax.set_facecolor('silver')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:23.024487Z","iopub.execute_input":"2022-07-31T14:04:23.024957Z","iopub.status.idle":"2022-07-31T14:04:23.239519Z","shell.execute_reply.started":"2022-07-31T14:04:23.024905Z","shell.execute_reply":"2022-07-31T14:04:23.238465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model creation fitting and testing\n### 2.1 Model creation\n<ol> Here, I chose de TF-IDF method. First, I defined a class to set hyperparameter easily. This class con-tains functionality to train, predict and test the model. In this cell, I also train the model with the training data and predict the training values to have an idea of the accuracy.","metadata":{}},{"cell_type":"code","source":"class Text_classifier:\n    def __init__(self, \n                 data, \n                 number_samples = 2000, \n                 number_features = 1000, \n                 number_categories = 5, \n                 number_top_words = 20, \n                 solver='cd', \n                 betaloss ='frobenius'): \n        self.data = data \n        self.numb_features = number_features\n        self.numb_categories = number_categories\n        self.numb_topwords = number_top_words\n        self.solver = solver\n        self.betaloss = betaloss\n        self.max_df = 0.95\n        self.min_df = 2\n        self.init = \"nndsvda\"\n        self.stop_words = \"english\"\n        self.term_frequecy_idf = None # term frequency-inverse document frequency\n        self.non_negative_MF_model = None\n        self.transformed_data = None\n        self.title = \"Articles classification based on Frobenius norm\"\n        self.text_vectorizer = TfidfVectorizer(\n                                max_df= self.max_df, \n                                min_df= self.min_df, \n                                #strip_accents= 'unicode',\n                                max_features= self.numb_features, \n                                stop_words=\"english\"\n                                )\n        \n        \n    def transform_matrix(self, ):\n        self.term_frequecy_idf = self.text_vectorizer.fit_transform(self.data)\n    \n    def factorize_data(self, ):\n        self.non_negative_MF_model = NMF(\n                                n_components=self.numb_categories,\n                                random_state=1,\n                                solver = self.solver,\n                                init=self.init,\n                                beta_loss=self.betaloss,\n                                alpha_W=0.00005,\n                                alpha_H=0.00005,\n                                l1_ratio=1,\n                                ).fit(self.term_frequecy_idf)\n    \n    def get_feature_names(self, ): \n         self.feature_names = self.text_vectorizer.get_feature_names_out()\n            \n    def fit(self, ):\n        self.transform_matrix()\n        self.factorize_data()\n        self.get_feature_names()\n    \n    def predict(self, X = None):\n        if X is None:\n            transformed_data = self.non_negative_MF_model.transform(self.term_frequecy_idf)\n        else:\n            transformed_data = self.non_negative_MF_model.transform(self.text_vectorizer.transform(X))\n        \n        self.transformed_data = [np.where(x == np.amax(x))[0][0] for x in transformed_data]\n    \n    def compute_confusion_m(self, y_true):\n        cf_matrix = confusion_matrix(y_true, self.transformed_data)\n        overall_ac = sum(np.diagonal(cf_matrix))/len(y_true)\n        \n        return cf_matrix, overall_ac \n        \n    def plot_top_words(self, ):    \n        fig, axes = plt.subplots(1, 5, figsize=(30, 15), sharex=True)\n        axes = axes.flatten()\n        top_features_data = []\n        for topic_idx, topic in enumerate(self.non_negative_MF_model.components_):\n            top_features_ind = topic.argsort()[: -self.numb_topwords - 1 : -1]\n            top_features = [self.feature_names[i] for i in top_features_ind]\n            weights = topic[top_features_ind]\n            top_features_data.append(top_features)\n            ax = axes[topic_idx]\n            ax.barh(top_features, weights, height=0.7)\n            ax.set_title(f\"Topic {topic_idx +1}\", fontdict={\"fontsize\": 30})\n            ax.invert_yaxis()\n            ax.tick_params(axis=\"both\", which=\"major\", labelsize=20)\n            for i in \"top right left\".split():\n                ax.spines[i].set_visible(False)\n            fig.suptitle(self.title, fontsize=40)\n        plt.subplots_adjust(top=0.90, bottom=0.05, wspace=0.90, hspace=0.3)\n        plt.show()\n        return top_features_data\n\ndef print_acurracy_measures(topic_lut, data_series_category, model):\n    array_train_dataClass = [ topic_lut[x] for x in data_series_category]\n    cf_matrix, accuracy = model.compute_confusion_m(array_train_dataClass)\n\n    display(Markdown('**Accuracy**'))\n    print(\"accuracy(train_data): %s\" % round(accuracy*100,1))\n\n    display(Markdown('**presenting confusion matrix**'))\n    df_cf_matrix = pd.DataFrame(cf_matrix, index = [i for i in topic_lut.keys()],\n                      columns = [i for i in topic_lut.keys()])\n    plt.figure(figsize = (10,7))\n    sn.heatmap(df_cf_matrix, annot=True)\n\nunsupervised_text_class = Text_classifier(data_train['review'], number_features=1490)\n\nunsupervised_text_class.fit()\n\nunsupervised_text_class.predict()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:23.241663Z","iopub.execute_input":"2022-07-31T14:04:23.242118Z","iopub.status.idle":"2022-07-31T14:04:23.905121Z","shell.execute_reply.started":"2022-07-31T14:04:23.242075Z","shell.execute_reply":"2022-07-31T14:04:23.903729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.2 getting the order of the categories\n<ol> to find out which category order is set in the factorized matrix, we plot the top 20. With a quick look, we can see that the order is sport, politics, tech, entertainment, business. Then, we proceed to create a dictionary that keep these categories.","metadata":{}},{"cell_type":"code","source":"top_features_data = unsupervised_text_class.plot_top_words()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:23.913541Z","iopub.execute_input":"2022-07-31T14:04:23.917094Z","iopub.status.idle":"2022-07-31T14:04:25.241421Z","shell.execute_reply.started":"2022-07-31T14:04:23.917010Z","shell.execute_reply":"2022-07-31T14:04:25.240209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.3 Printing accuracy measures\nIn the first matrix visualization, I show the confusion matrix computed with the same data that the model was trained. Further down, I show the matrix accuracy for the test data.","metadata":{"execution":{"iopub.status.busy":"2022-07-31T13:08:30.984468Z","iopub.execute_input":"2022-07-31T13:08:30.984851Z","iopub.status.idle":"2022-07-31T13:08:30.993859Z","shell.execute_reply.started":"2022-07-31T13:08:30.984820Z","shell.execute_reply":"2022-07-31T13:08:30.992194Z"}}},{"cell_type":"code","source":"# This step is a trick for ordering the result of the matrix factorization because it give differents order of the components\n#everytime you run it.\n# This step is manual, so it requires that you see the top words and classified the groups quickly\n\ntop_words = {\"politics\" : ['mr', 'labour', 'blair', 'election', 'said', 'party', 'brown', 'government', 'howard', 'minister', 'prime', 'tory', 'tax', 'chancellor', 'campaign'],\n             \"sport\" : ['game', 'england', 'win', 'said', 'ireland', 'france', 'wales', 'players', 'play', 'cup', 'chelsea', 'team', 'season', 'coach', 'match'], \n             \"tech\" : ['music', 'people', 'mobile', 'said', 'phone', 'technology', 'users', 'digital', 'broadband', 'net', 'software', 'new', 'use', 'video', 'uk'],\n             \"entertainment\" : ['film', 'best', 'awards', 'actor', 'award', 'films', 'won', 'oscar', 'actress', 'star', 'comedy', 'director', 'aviator', 'movie', 'festival', 'oscars'],\n             \"business\": ['bn', 'said', 'growth', 'economy', 'year', 'oil', 'market', 'shares', 'firm', 'bank', 'sales', 'prices', 'company', 'economic', 'china', 'dollar', 'rise']\n             }\nindex = 0\ntopic_lut = {}\n\nfor top_features in top_features_data:\n    if (len(set(top_words[\"politics\"]) & set(top_features))) > 10:\n        topic_lut[\"politics\"] = index\n    elif (len(set(top_words[\"sport\"]) & set(top_features))) > 10:\n        topic_lut[\"sport\"] = index\n    elif (len(set(top_words[\"business\"]) & set(top_features))) > 10:\n        topic_lut[\"business\"] = index\n    elif (len(set(top_words[\"entertainment\"]) & set(top_features))) > 10:\n        topic_lut[\"entertainment\"] = index\n    elif (len(set(top_words[\"tech\"]) & set(top_features))) > 10:\n        topic_lut[\"tech\"] = index\n    index +=1\n            \n\nprint(topic_lut)\nprint_acurracy_measures(topic_lut, data_train[\"Category\"], unsupervised_text_class)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:25.243222Z","iopub.execute_input":"2022-07-31T14:04:25.243847Z","iopub.status.idle":"2022-07-31T14:04:25.612118Z","shell.execute_reply.started":"2022-07-31T14:04:25.243807Z","shell.execute_reply":"2022-07-31T14:04:25.611369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"######## working with test data #####\ndata_test['review'] = format_text(data_test['Text'])\nunsupervised_text_class.predict(data_test['review'])\nprint_acurracy_measures(topic_lut, data_test[\"Category\"], unsupervised_text_class)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:25.613311Z","iopub.execute_input":"2022-07-31T14:04:25.613772Z","iopub.status.idle":"2022-07-31T14:04:26.192098Z","shell.execute_reply.started":"2022-07-31T14:04:25.613743Z","shell.execute_reply":"2022-07-31T14:04:26.190859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.4 Hyper parameter tunning\n<ol> In this section, I do 3 tests: I change the number of word features, the matrix solver, and the betaloss function. As we could see, the accuracy drops after the number of word features is bigger than 1500. On the other hand, the accuracy increased slightly when changing the solver from \"cd\" to \"mu\". At last, with the kullback-leibler betaloss function the accuracy improve approximately 5%.","metadata":{}},{"cell_type":"code","source":"array_test_dataClass = [ topic_lut[x] for x in data_test[\"Category\"]]\naccuracy_l = []\n\nfor number_ftrs in range(500,5000,500):\n    unsupervised_text_class = Text_classifier(data_train['review'], number_features=number_ftrs)\n    unsupervised_text_class.fit()\n    unsupervised_text_class.predict(data_test['review'])\n    cf_matrix, accuracy = unsupervised_text_class.compute_confusion_m(array_test_dataClass)\n    accuracy_l.append([number_ftrs,accuracy])\n\npd.DataFrame(accuracy_l,columns = ['number_features', 'accuracy']).plot(x='number_features',y='accuracy')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:26.193529Z","iopub.execute_input":"2022-07-31T14:04:26.193860Z","iopub.status.idle":"2022-07-31T14:04:33.912142Z","shell.execute_reply.started":"2022-07-31T14:04:26.193829Z","shell.execute_reply":"2022-07-31T14:04:33.910727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"solver = [\"cd\", 'mu']\naccuracy_l = []\n\nfor s in solver:\n    unsupervised_text_class = Text_classifier(data_train['review'], number_features=1500, solver=s)\n    unsupervised_text_class.fit()\n    unsupervised_text_class.predict(data_test['review'])\n    cf_matrix, accuracy = unsupervised_text_class.compute_confusion_m(array_test_dataClass)\n    accuracy_l.append([s,accuracy])\n\npd.DataFrame(accuracy_l,columns = ['solver', 'accuracy']).plot(x='solver',y='accuracy', kind='bar', logy=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:33.913650Z","iopub.execute_input":"2022-07-31T14:04:33.914075Z","iopub.status.idle":"2022-07-31T14:04:36.178119Z","shell.execute_reply.started":"2022-07-31T14:04:33.914027Z","shell.execute_reply":"2022-07-31T14:04:36.177099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"betaloss = [\"frobenius\", 'kullback-leibler']\naccuracy_l = []\nfor fl in betaloss:\n    unsupervised_text_class = Text_classifier(data_train['review'], number_features=1500, solver='mu', betaloss=fl )\n    unsupervised_text_class.fit()\n    unsupervised_text_class.predict(data_test['review'])\n    cf_matrix, accuracy = unsupervised_text_class.compute_confusion_m(array_test_dataClass)\n    accuracy_l.append([fl,accuracy])\n\npd.DataFrame(accuracy_l,columns = ['solver', 'accuracy']).plot(x='solver',y='accuracy', kind='bar' , logy=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:36.179582Z","iopub.execute_input":"2022-07-31T14:04:36.180206Z","iopub.status.idle":"2022-07-31T14:04:39.614865Z","shell.execute_reply.started":"2022-07-31T14:04:36.180165Z","shell.execute_reply":"2022-07-31T14:04:39.614108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Here, I print the final solution","metadata":{}},{"cell_type":"code","source":"# Solution \n\ntopic_lut_invert = {v: k for k, v in topic_lut.items()}\ndata_train_whole['review'] = format_text(data_train_whole['Text'])\nunsupervised_text_class_solution = Text_classifier(data_train_whole['review'], number_features=1490)\nunsupervised_text_class_solution.fit()\nunsupervised_text_class_solution.predict(data_for_predict['Text'])\nBBC_news_solution = [[data_for_predict['ArticleId'][i], topic_lut_invert[unsupervised_text_class_solution.transformed_data[i]]] for i in range(len(data_for_predict))]\nwith open('BBC News Solution.csv','w' ,newline='') as csvfile:\n    writer = csv.writer(csvfile, delimiter=',',)\n    writer.writerow([\"ArticleId\",'Category'])\n    writer.writerows(BBC_news_solution)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T14:04:39.617465Z","iopub.execute_input":"2022-07-31T14:04:39.618321Z","iopub.status.idle":"2022-07-31T14:04:40.954984Z","shell.execute_reply.started":"2022-07-31T14:04:39.618284Z","shell.execute_reply":"2022-07-31T14:04:40.953822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}