{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":3043,"databundleVersionId":46668,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Iimporting the libraries","metadata":{}},{"cell_type":"code","source":"!pip install wordcloud\n!pip install nltk","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:11:28.126255Z","iopub.execute_input":"2024-03-10T15:11:28.126851Z","iopub.status.idle":"2024-03-10T15:11:58.441473Z","shell.execute_reply.started":"2024-03-10T15:11:28.12678Z","shell.execute_reply":"2024-03-10T15:11:58.440211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the libraries for visualization \nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom wordcloud import WordCloud\n\n# Importing the libraries for text processing\nimport nltk\nimport re\nimport string\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom nltk.stem import WordNetLemmatizer\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split \nfrom sklearn.feature_extraction.text import TfidfVectorizer \nfrom sklearn.metrics import confusion_matrix\n\nnltk.download('stopwords')\nnltk.download('punkt')\nnltk.download('wordnet')\n\n# Importing the libraries for model building\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom sklearn.metrics import classification_report\n\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:11:58.444765Z","iopub.execute_input":"2024-03-10T15:11:58.445605Z","iopub.status.idle":"2024-03-10T15:12:03.31782Z","shell.execute_reply.started":"2024-03-10T15:11:58.445564Z","shell.execute_reply":"2024-03-10T15:12:03.316653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Read Dataset","metadata":{}},{"cell_type":"code","source":"#reading the dataset\ndf = pd.read_csv(\"/kaggle/input/predict-closed-questions-on-stack-overflow/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:54:51.548366Z","iopub.execute_input":"2024-03-10T15:54:51.548899Z","iopub.status.idle":"2024-03-10T15:56:32.724959Z","shell.execute_reply.started":"2024-03-10T15:54:51.548864Z","shell.execute_reply":"2024-03-10T15:56:32.723521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:58:02.259129Z","iopub.execute_input":"2024-03-10T15:58:02.261934Z","iopub.status.idle":"2024-03-10T15:58:02.304499Z","shell.execute_reply.started":"2024-03-10T15:58:02.261873Z","shell.execute_reply":"2024-03-10T15:58:02.303158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.8541Z","iopub.status.idle":"2024-03-10T15:12:04.855005Z","shell.execute_reply.started":"2024-03-10T15:12:04.854674Z","shell.execute_reply":"2024-03-10T15:12:04.8547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Dropping the columns/features which are not gonna contribute to the results","metadata":{}},{"cell_type":"code","source":"\ndf.drop(['OwnerUserId'],axis=1,inplace=True)\ndf.drop(['OwnerUndeletedAnswerCountAtPostTime'],axis=1,inplace=True)\ndf.drop(['PostClosedDate'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.856595Z","iopub.status.idle":"2024-03-10T15:12:04.857451Z","shell.execute_reply.started":"2024-03-10T15:12:04.857175Z","shell.execute_reply":"2024-03-10T15:12:04.857199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.858989Z","iopub.status.idle":"2024-03-10T15:12:04.859846Z","shell.execute_reply.started":"2024-03-10T15:12:04.859537Z","shell.execute_reply":"2024-03-10T15:12:04.85956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.861662Z","iopub.status.idle":"2024-03-10T15:12:04.862501Z","shell.execute_reply.started":"2024-03-10T15:12:04.862222Z","shell.execute_reply":"2024-03-10T15:12:04.862245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Checking null values in the tag columns","metadata":{}},{"cell_type":"code","source":"#checking null values in the tag columns\ndf['Tag1'].isnull().sum()\n# df['Tag2'].isnull().sum()\n# df['Tag3'].isnull().sum()\n# df['Tag4'].isnull().sum()\n# df['Tag5'].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.863989Z","iopub.status.idle":"2024-03-10T15:12:04.864757Z","shell.execute_reply.started":"2024-03-10T15:12:04.864492Z","shell.execute_reply":"2024-03-10T15:12:04.864515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Elminate Null Values","metadata":{}},{"cell_type":"code","source":"#there are many nulll values in the tag columns, we need to elminate the null values\ndf['Tag1']=df['Tag1'].replace(np.nan,' ')\ndf['Tag2']=df['Tag2'].replace(np.nan,' ')\ndf['Tag3']=df['Tag3'].replace(np.nan,' ')\ndf['Tag4']=df['Tag4'].replace(np.nan,' ')\ndf['Tag5']=df['Tag5'].replace(np.nan,' ')","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.866236Z","iopub.status.idle":"2024-03-10T15:12:04.867024Z","shell.execute_reply.started":"2024-03-10T15:12:04.866764Z","shell.execute_reply":"2024-03-10T15:12:04.866812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Tags are just the tags related to the question, we can combine them all into one column","metadata":{}},{"cell_type":"code","source":"\ndf['Tags']=df['Tag1']+' '+df['Tag2']+' '+df['Tag3']+' '+df['Tag4']+' '+df['Tag5']","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.86811Z","iopub.status.idle":"2024-03-10T15:12:04.868671Z","shell.execute_reply.started":"2024-03-10T15:12:04.868484Z","shell.execute_reply":"2024-03-10T15:12:04.868501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Changing into lower case and removing the whitespaces","metadata":{}},{"cell_type":"code","source":"df['Tags']=df['Tags'].str.lower()\ndf['Tags']=df['Tags'].apply(lambda x:x.lstrip())\ndf['Tags']=df['Tags'].apply(lambda x:x.rstrip())","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.869739Z","iopub.status.idle":"2024-03-10T15:12:04.870329Z","shell.execute_reply.started":"2024-03-10T15:12:04.870146Z","shell.execute_reply":"2024-03-10T15:12:04.870162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.871372Z","iopub.status.idle":"2024-03-10T15:12:04.871777Z","shell.execute_reply.started":"2024-03-10T15:12:04.871584Z","shell.execute_reply":"2024-03-10T15:12:04.871602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visulization","metadata":{}},{"cell_type":"markdown","source":"#### Question Counts by Closing Reasons","metadata":{}},{"cell_type":"code","source":"\n\n# Sample data\nclosing_reasons = ['not constructive', 'not a real question', 'off topic', 'too localized']\n#count for each closing reasons in dataset\nquestion_counts = [df[df['OpenStatus'] == 'not constructive'].shape[0], df[df['OpenStatus'] == 'not a real question'].shape[0], df[df['OpenStatus'] == 'off topic'].shape[0], df[df['OpenStatus'] == 'too localized'].shape[0]]\n\n# Create a bar plot\nplt.figure(figsize=(10, 6))\nsns.barplot(y=closing_reasons, x=question_counts,  palette='viridis', orient='h',hue=closing_reasons)\nplt.xlabel('Question Count')\n\nplt.ylabel('Closing Reason')\nplt.title('Question Counts by Closing Reasons')\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.872955Z","iopub.status.idle":"2024-03-10T15:12:04.87337Z","shell.execute_reply.started":"2024-03-10T15:12:04.873156Z","shell.execute_reply":"2024-03-10T15:12:04.873172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Top 20 Tags","metadata":{}},{"cell_type":"code","source":"\n# Assuming 'df' is your original DataFrame\ntop_tags = df['Tags'].value_counts().head(20)\n\n# Plotting the top 20 tags\nplt.figure(figsize=(10, 5))\nsns.barplot(y=top_tags.index,x= top_tags.values, palette='viridis',hue=top_tags.index)\nplt.title('Top 20 Tags')\nplt.xlabel('Tags')\nplt.ylabel('Count')\nplt.xticks(rotation=45)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.875608Z","iopub.status.idle":"2024-03-10T15:12:04.876044Z","shell.execute_reply.started":"2024-03-10T15:12:04.875846Z","shell.execute_reply":"2024-03-10T15:12:04.875863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Top Tags for Closed Questions & Open Questions","metadata":{}},{"cell_type":"code","source":"\n#as there are 5 classes out of wich 4 are subclasses for the class 'closed', we'll be merging those 4 classes\nclosed_classes = ['not constructive', 'not a real question', 'off topic', 'too localized']\ndf['OpenStatus'] = df['OpenStatus'].replace(closed_classes, 'closed')\n\n# Splitting the dataset into closed and open questions\ndf_closed = df[df['OpenStatus'] == 'closed']\ndf_open = df[df['OpenStatus'] == 'open']\n\n# Top tags for closed questions\ntop_closed_tags = df_closed['Tags'].value_counts().head(10)\n\n# Top tags for open questions\ntop_open_tags = df_open['Tags'].value_counts().head(10)\n\n# Plotting the top tags for closed questions\nplt.figure(figsize=(10, 5))\nsns.barplot(x=top_closed_tags.index, y=top_closed_tags.values, palette='viridis',hue=top_closed_tags.values,legend=False)\nplt.title('Top Tags for Closed Questions')\nplt.xlabel('Tags')\nplt.ylabel('Count')\nplt.xticks(rotation=45)\nplt.show()\n\n# Plotting the top tags for open questions\nplt.figure(figsize=(10, 5))\nsns.barplot(x=top_open_tags.index, y=top_open_tags.values, palette='viridis',hue=top_open_tags.values,legend=False )\nplt.title('Top Tags for Open Questions')\nplt.xlabel('Tags')\nplt.ylabel('Count')\nplt.xticks(rotation=45)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.877472Z","iopub.status.idle":"2024-03-10T15:12:04.877916Z","shell.execute_reply.started":"2024-03-10T15:12:04.877672Z","shell.execute_reply":"2024-03-10T15:12:04.877688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Word Cloud of Question Titles ","metadata":{}},{"cell_type":"code","source":"\ntext = \" \".join(title for title in df['Title'][:1000])\n\nwordcloud = WordCloud(width=800, height=600, max_words=150).generate(text)\n\nplt.figure(figsize=(10, 6))\nplt.imshow(wordcloud, interpolation='bilinear')\nplt.axis('off')\nplt.title(\"Word Cloud of Question Titles \")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.880129Z","iopub.status.idle":"2024-03-10T15:12:04.880622Z","shell.execute_reply.started":"2024-03-10T15:12:04.88042Z","shell.execute_reply":"2024-03-10T15:12:04.880437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"df.drop(['PostId','OwnerCreationDate','Tag1','Tag2','Tag3','Tag4','Tag5'],axis=1,inplace=True)\ndf.drop(['ReputationAtPostCreation'],axis=1,inplace=True)\ndf.drop(['PostCreationDate'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.882719Z","iopub.status.idle":"2024-03-10T15:12:04.883855Z","shell.execute_reply.started":"2024-03-10T15:12:04.883596Z","shell.execute_reply":"2024-03-10T15:12:04.883621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.885328Z","iopub.status.idle":"2024-03-10T15:12:04.886415Z","shell.execute_reply.started":"2024-03-10T15:12:04.886205Z","shell.execute_reply":"2024-03-10T15:12:04.886224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.OpenStatus.unique()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.88775Z","iopub.status.idle":"2024-03-10T15:12:04.888205Z","shell.execute_reply.started":"2024-03-10T15:12:04.888003Z","shell.execute_reply":"2024-03-10T15:12:04.88802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.889641Z","iopub.status.idle":"2024-03-10T15:12:04.890113Z","shell.execute_reply.started":"2024-03-10T15:12:04.889886Z","shell.execute_reply":"2024-03-10T15:12:04.889911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#making a smaller dataset out of the big one\nsample_size = 50000  # Specify the size of the sample\nsample_df = df.sample(n=sample_size, random_state=42) ","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.891633Z","iopub.status.idle":"2024-03-10T15:12:04.892087Z","shell.execute_reply.started":"2024-03-10T15:12:04.891888Z","shell.execute_reply":"2024-03-10T15:12:04.891906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#text preprocessing\ndef preprocess_text(text):\n    #convert to lowercase\n    text = text.lower()\n    \n    #remove HTML tags\n    text = remove_html_tags(text)\n    \n    #remove URL tags\n    text = remove_urls(text)\n    \n    #remove special characters and numbers\n    text = re.sub(r'[^a-zA-Z]', ' ', text)\n    \n    #tokenize the text into individual words\n    tokens = word_tokenize(text)\n    \n    #remove stopwords\n    stop_words = set(stopwords.words('english'))\n    tokens = [word for word in tokens if word not in stop_words]\n    \n    #lemmatize the words\n    lemmatizer = WordNetLemmatizer()\n    tokens = [lemmatizer.lemmatize(word) for word in tokens]\n    \n    #join the tokens back into a single string\n    preprocessed_text = ' '.join(tokens)\n    \n    return preprocessed_text\n\n#function to remove HTML tags\ndef remove_html_tags(text):\n    pattern = re.compile('<.*?>')\n    return pattern.sub(r'', text)\n\n#function to remove URL links\ndef remove_urls(text):\n    pattern = re.compile(r'http\\S+|www\\S+')\n    return pattern.sub(r'', text)\n\n#apply preprocessing to the 'Title' and 'BodyMarkdown' columns\nsample_df['Title'] = sample_df['Title'].apply(preprocess_text)\nsample_df['BodyMarkdown'] = sample_df['BodyMarkdown'].apply(preprocess_text)\nsample_df['Tags'] = sample_df['Tags'].apply(preprocess_text)\n\n#display the preprocessed dataset\nsample_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.893408Z","iopub.status.idle":"2024-03-10T15:12:04.894018Z","shell.execute_reply.started":"2024-03-10T15:12:04.893746Z","shell.execute_reply":"2024-03-10T15:12:04.893768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.OpenStatus.unique()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.897151Z","iopub.status.idle":"2024-03-10T15:12:04.897591Z","shell.execute_reply.started":"2024-03-10T15:12:04.897393Z","shell.execute_reply":"2024-03-10T15:12:04.897411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#encoding the OpenStatus so that categorical classes will be transformed into numeric\n\nencoder = LabelEncoder()\nencoder.fit(sample_df['OpenStatus'])\nsample_df['OpenStatus'] = encoder.transform(sample_df['OpenStatus'])","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.899108Z","iopub.status.idle":"2024-03-10T15:12:04.899526Z","shell.execute_reply.started":"2024-03-10T15:12:04.899332Z","shell.execute_reply":"2024-03-10T15:12:04.89935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.901022Z","iopub.status.idle":"2024-03-10T15:12:04.901477Z","shell.execute_reply.started":"2024-03-10T15:12:04.901278Z","shell.execute_reply":"2024-03-10T15:12:04.901295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.OpenStatus.unique()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.903437Z","iopub.status.idle":"2024-03-10T15:12:04.903895Z","shell.execute_reply.started":"2024-03-10T15:12:04.903657Z","shell.execute_reply":"2024-03-10T15:12:04.903674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#splitting the dataset into train and test set\nx = sample_df['Title'] + ' '+ sample_df['BodyMarkdown']+' '+sample_df['Tags']\ny = sample_df['OpenStatus']\nxtrain, xtest, ytrain, ytest = train_test_split(x, y, test_size = 0.3, random_state = 203)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.905653Z","iopub.status.idle":"2024-03-10T15:12:04.90608Z","shell.execute_reply.started":"2024-03-10T15:12:04.905883Z","shell.execute_reply":"2024-03-10T15:12:04.9059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.911777Z","iopub.status.idle":"2024-03-10T15:12:04.912319Z","shell.execute_reply.started":"2024-03-10T15:12:04.9121Z","shell.execute_reply":"2024-03-10T15:12:04.912121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtest.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.914052Z","iopub.status.idle":"2024-03-10T15:12:04.91452Z","shell.execute_reply.started":"2024-03-10T15:12:04.914307Z","shell.execute_reply":"2024-03-10T15:12:04.914325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#converting words to numbers using TF-IDF ie performing vectorization\nvectorizer = TfidfVectorizer(max_features = 10000)\nxtrain_tfidf = vectorizer.fit_transform(xtrain).toarray()  # converting words to numbers for train data \nxtest_tfidf = vectorizer.transform(xtest).toarray()        # converting words to numbers for test data ","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.916472Z","iopub.status.idle":"2024-03-10T15:12:04.916997Z","shell.execute_reply.started":"2024-03-10T15:12:04.916738Z","shell.execute_reply":"2024-03-10T15:12:04.916758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain_tfidf.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.918413Z","iopub.status.idle":"2024-03-10T15:12:04.918968Z","shell.execute_reply.started":"2024-03-10T15:12:04.918692Z","shell.execute_reply":"2024-03-10T15:12:04.918711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.921357Z","iopub.status.idle":"2024-03-10T15:12:04.921844Z","shell.execute_reply.started":"2024-03-10T15:12:04.921604Z","shell.execute_reply":"2024-03-10T15:12:04.921623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Modeling","metadata":{}},{"cell_type":"code","source":"\nmodel = Sequential()\nmodel.add(Dense(16, input_dim=xtrain_tfidf.shape[1], activation='relu'))  \nmodel.add(Dense(4, activation='relu'))  \nmodel.add(Dense(1, activation='sigmoid'))\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\n# Training the model\nmodel.fit(xtrain_tfidf, ytrain, epochs=10, batch_size=32, validation_data=(xtest_tfidf,ytest))\nloss, accuracy = model.evaluate(xtest_tfidf, ytest)\nprint('Test Accuracy:', accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.9241Z","iopub.status.idle":"2024-03-10T15:12:04.92456Z","shell.execute_reply.started":"2024-03-10T15:12:04.924352Z","shell.execute_reply":"2024-03-10T15:12:04.924371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(xtest_tfidf)\ny_pred","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.926872Z","iopub.status.idle":"2024-03-10T15:12:04.927356Z","shell.execute_reply.started":"2024-03-10T15:12:04.927122Z","shell.execute_reply":"2024-03-10T15:12:04.92714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Convert continuous predictions to binary values using a threshold\ny_pred_binary = (y_pred > 0.5).astype(int)\n\nprint(classification_report(ytest, y_pred_binary))\n\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.929021Z","iopub.status.idle":"2024-03-10T15:12:04.929469Z","shell.execute_reply.started":"2024-03-10T15:12:04.929256Z","shell.execute_reply":"2024-03-10T15:12:04.929274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nconf_matrix = confusion_matrix(ytest, y_pred_binary)\nconf_matrix\n\n\n# Plotting the confusion matrix\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, cmap='viridis', fmt='g')\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T15:12:04.931042Z","iopub.status.idle":"2024-03-10T15:12:04.931481Z","shell.execute_reply.started":"2024-03-10T15:12:04.931277Z","shell.execute_reply":"2024-03-10T15:12:04.931294Z"},"trusted":true},"execution_count":null,"outputs":[]}]}