{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Download libraries and data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\n\n# Download data\ntrain = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ntest = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2024-07-25T21:44:38.346396Z","iopub.execute_input":"2024-07-25T21:44:38.347324Z","iopub.status.idle":"2024-07-25T21:44:42.696057Z","shell.execute_reply.started":"2024-07-25T21:44:38.347271Z","shell.execute_reply":"2024-07-25T21:44:42.694667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Exploration and processing of data","metadata":{}},{"cell_type":"code","source":"# عرض بعض العينات من البيانات\nprint(train.head())\n\n# تنظيف البيانات (يمكن تحسين هذا الجزء لاحقًا)\ntrain['question_text'] = train['question_text'].str.replace('[^\\w\\s]', '').str.lower()\n\n# تقسيم البيانات\nX_train, X_val, y_train, y_val = train_test_split(train['question_text'], train['target'], test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-25T21:44:42.698369Z","iopub.execute_input":"2024-07-25T21:44:42.698766Z","iopub.status.idle":"2024-07-25T21:44:44.170945Z","shell.execute_reply.started":"2024-07-25T21:44:42.698734Z","shell.execute_reply":"2024-07-25T21:44:44.169816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Feature extraction using TF-IDF","metadata":{}},{"cell_type":"code","source":"# استخدام TF-IDF لاستخراج الميزات\nvectorizer = TfidfVectorizer(max_features=10000)\nX_train_tfidf = vectorizer.fit_transform(X_train)\nX_val_tfidf = vectorizer.transform(X_val)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-25T21:44:44.172374Z","iopub.execute_input":"2024-07-25T21:44:44.172819Z","iopub.status.idle":"2024-07-25T21:45:13.701775Z","shell.execute_reply.started":"2024-07-25T21:44:44.172779Z","shell.execute_reply":"2024-07-25T21:45:13.700449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Build the model with increasing number of iterations","metadata":{}},{"cell_type":"code","source":"# زيادة عدد التكرارات\nmodel = LogisticRegression(max_iter=1000)\nmodel.fit(X_train_tfidf, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-25T21:45:13.704659Z","iopub.execute_input":"2024-07-25T21:45:13.705051Z","iopub.status.idle":"2024-07-25T21:46:00.844486Z","shell.execute_reply.started":"2024-07-25T21:45:13.705018Z","shell.execute_reply":"2024-07-25T21:46:00.842534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Model evaluation","metadata":{}},{"cell_type":"code","source":"# التنبؤ على بيانات التحقق\ny_val_pred = model.predict(X_val_tfidf)\n\n# حساب الدقة\naccuracy = accuracy_score(y_val, y_val_pred)\nprint(f'Accuracy: {accuracy}')\n","metadata":{"execution":{"iopub.status.busy":"2024-07-25T21:46:00.847220Z","iopub.execute_input":"2024-07-25T21:46:00.848744Z","iopub.status.idle":"2024-07-25T21:46:00.920580Z","shell.execute_reply.started":"2024-07-25T21:46:00.848566Z","shell.execute_reply":"2024-07-25T21:46:00.919048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Processing and making predictions on test data","metadata":{}},{"cell_type":"code","source":"# معالجة بيانات الاختبار\ntest['question_text'] = test['question_text'].str.replace('[^\\w\\s]', '').str.lower()\nX_test_tfidf = vectorizer.transform(test['question_text'])\n\n# التنبؤ على بيانات الاختبار\ntest_predictions = model.predict(X_test_tfidf)\n\n# إنشاء ملف للرفع على Kaggle\nsubmission = pd.DataFrame({'qid': test['qid'], 'prediction': test_predictions})\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-25T21:46:00.923248Z","iopub.execute_input":"2024-07-25T21:46:00.924432Z","iopub.status.idle":"2024-07-25T21:46:10.877751Z","shell.execute_reply.started":"2024-07-25T21:46:00.924371Z","shell.execute_reply":"2024-07-25T21:46:10.876363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Optimization using StandardScaler in Pipeline","metadata":{}},{"cell_type":"code","source":"# إنشاء أنبوب يحتوي على محول المقياس والنموذج\npipeline = Pipeline([\n    ('scaler', StandardScaler(with_mean=False)),  # with_mean=False لأن البيانات المصفوفة النادرة\n    ('logreg', LogisticRegression(max_iter=1000))\n])\n\npipeline.fit(X_train_tfidf, y_train)\n\n# التنبؤ على بيانات التحقق باستخدام الأنبوب\ny_val_pred = pipeline.predict(X_val_tfidf)\naccuracy = accuracy_score(y_val, y_val_pred)\nprint(f'Accuracy with StandardScaler: {accuracy}')\n\n# التنبؤ على بيانات الاختبار باستخدام الأنبوب\ntest_predictions = pipeline.predict(X_test_tfidf)\nsubmission = pd.DataFrame({'qid': test['qid'], 'prediction': test_predictions})\nsubmission.to_csv('submission_pipeline.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-25T21:46:10.879121Z","iopub.execute_input":"2024-07-25T21:46:10.879492Z","iopub.status.idle":"2024-07-25T21:47:09.116512Z","shell.execute_reply.started":"2024-07-25T21:46:10.879461Z","shell.execute_reply":"2024-07-25T21:47:09.115146Z"},"trusted":true},"execution_count":null,"outputs":[]}]}