{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":17777,"databundleVersionId":869809,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nfrom transformers import DistilBertTokenizer, DistilBertModel\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, f1_score\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nfrom tqdm import tqdm\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2024-06-24T12:13:57.570209Z","iopub.execute_input":"2024-06-24T12:13:57.570689Z","iopub.status.idle":"2024-06-24T12:13:57.589806Z","shell.execute_reply.started":"2024-06-24T12:13:57.570651Z","shell.execute_reply":"2024-06-24T12:13:57.588731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.manual_seed(42)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T12:13:57.592592Z","iopub.execute_input":"2024-06-24T12:13:57.592897Z","iopub.status.idle":"2024-06-24T12:13:57.599592Z","shell.execute_reply.started":"2024-06-24T12:13:57.592870Z","shell.execute_reply":"2024-06-24T12:13:57.598699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Завантаження даних\ntrain_data = pd.read_csv('/kaggle/input/nlp-getting-started/train.csv', encoding='utf-8')\ntest_data = pd.read_csv('/kaggle/input/nlp-getting-started/test.csv', encoding='utf-8')\n\n# Перевірка структури даних\nprint(train_data.head())\nprint(test_data.head())","metadata":{"execution":{"iopub.status.busy":"2024-06-24T12:13:57.600570Z","iopub.execute_input":"2024-06-24T12:13:57.600888Z","iopub.status.idle":"2024-06-24T12:13:57.651808Z","shell.execute_reply.started":"2024-06-24T12:13:57.600863Z","shell.execute_reply":"2024-06-24T12:13:57.650798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Завантаження токенізатора та моделі\ntokenizer = DistilBertTokenizer.from_pretrained('distilbert-base-multilingual-cased')\nmodel = DistilBertModel.from_pretrained('distilbert-base-multilingual-cased')","metadata":{"execution":{"iopub.status.busy":"2024-06-24T12:13:57.654212Z","iopub.execute_input":"2024-06-24T12:13:57.654886Z","iopub.status.idle":"2024-06-24T12:13:59.110395Z","shell.execute_reply.started":"2024-06-24T12:13:57.654847Z","shell.execute_reply":"2024-06-24T12:13:59.109469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tokenize_text(text):\n    return tokenizer(text, padding=True, truncation=True, return_tensors='pt')\n\n# Приклад токенізації\nexample_text = train_data['text'].iloc[0]\ntokens = tokenize_text(example_text)\nprint(tokens)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T12:13:59.112041Z","iopub.execute_input":"2024-06-24T12:13:59.112330Z","iopub.status.idle":"2024-06-24T12:13:59.119355Z","shell.execute_reply.started":"2024-06-24T12:13:59.112305Z","shell.execute_reply":"2024-06-24T12:13:59.118519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_embeddings(text_list, batch_size=32):\n    embeddings = []\n    for i in tqdm(range(0, len(text_list), batch_size)):\n        batch_texts = text_list[i:i+batch_size]\n        tokens = tokenizer(batch_texts, padding=True, truncation=True, return_tensors='pt')\n        with torch.no_grad():\n            outputs = model(**tokens)\n        batch_embeddings = outputs.last_hidden_state.mean(dim=1).numpy()\n        embeddings.append(batch_embeddings)\n    return np.vstack(embeddings)\n\n# Отримання ембеддингів для навчального датасету\nX_train = get_embeddings(train_data['text'].tolist())\ny_train = train_data['target'].values\n\n# Отримання ембеддингів для тестового датасету\nX_test = get_embeddings(test_data['text'].tolist())\n\n# Масштабування даних\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T12:13:59.120738Z","iopub.execute_input":"2024-06-24T12:13:59.121165Z","iopub.status.idle":"2024-06-24T12:18:48.988324Z","shell.execute_reply.started":"2024-06-24T12:13:59.121131Z","shell.execute_reply":"2024-06-24T12:18:48.987380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Навчання моделі логістичної регресії з збільшеною кількістю ітерацій\nclassifier = LogisticRegression(max_iter=1000)  # збільшення кількості ітерацій до 1000\nclassifier.fit(X_train_scaled, y_train)\n\n# Передбачення та оцінка точності\ny_pred = classifier.predict(X_test_scaled)\n# Виведення результатів моделі\nprint(y_pred)  # для відправки результатів на Kaggle","metadata":{"execution":{"iopub.status.busy":"2024-06-24T12:18:48.989385Z","iopub.execute_input":"2024-06-24T12:18:48.989680Z","iopub.status.idle":"2024-06-24T12:18:50.737036Z","shell.execute_reply.started":"2024-06-24T12:18:48.989654Z","shell.execute_reply":"2024-06-24T12:18:50.735639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Розділення train даних на навчальні та валідаційні для оцінки моделі\nX_train_split, X_val_split, y_train_split, y_val_split = train_test_split(X_train_scaled, y_train, test_size=0.4, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T12:18:50.739221Z","iopub.execute_input":"2024-06-24T12:18:50.740208Z","iopub.status.idle":"2024-06-24T12:18:50.754771Z","shell.execute_reply.started":"2024-06-24T12:18:50.740154Z","shell.execute_reply":"2024-06-24T12:18:50.753429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Навчання моделі на спліт-даних\nclassifier_split = LogisticRegression(max_iter=1000)\nclassifier_split.fit(X_train_split, y_train_split)\n\n# Передбачення та оцінка точності на валідаційному датасеті\ny_val_pred = classifier_split.predict(X_val_split)\naccuracy = accuracy_score(y_val_split, y_val_pred)\nf1 = f1_score(y_val_split, y_val_pred, average='weighted')\nprint(f'Validation Accuracy: {accuracy}')\nprint(f'Validation F1 Score: {f1}')","metadata":{"execution":{"iopub.status.busy":"2024-06-24T12:18:50.759504Z","iopub.execute_input":"2024-06-24T12:18:50.760369Z","iopub.status.idle":"2024-06-24T12:18:51.721516Z","shell.execute_reply.started":"2024-06-24T12:18:50.760320Z","shell.execute_reply":"2024-06-24T12:18:51.720317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Візуалізація матриці плутанини\ncm = confusion_matrix(y_val_split, y_val_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm)\ndisp.plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-24T12:18:51.723570Z","iopub.execute_input":"2024-06-24T12:18:51.724416Z","iopub.status.idle":"2024-06-24T12:18:51.961547Z","shell.execute_reply.started":"2024-06-24T12:18:51.724365Z","shell.execute_reply":"2024-06-24T12:18:51.960681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Візуалізація точності та F1-міри при різних розмірах тестових даних\ntest_sizes = [0.1, 0.2, 0.3, 0.4]\naccuracies = []\nf1_scores = []\n\nfor size in test_sizes:\n    X_train_part, X_val_part, y_train_part, y_val_part = train_test_split(X_train_scaled, y_train, test_size=size, random_state=42)\n    classifier.fit(X_train_part, y_train_part)\n    y_val_part_pred = classifier.predict(X_val_part)\n    accuracy = accuracy_score(y_val_part, y_val_part_pred)\n    f1 = f1_score(y_val_part, y_val_part_pred, average='weighted')\n    accuracies.append(accuracy)\n    f1_scores.append(f1)\n\nplt.plot(test_sizes, accuracies, marker='o', label='Accuracy')\nplt.plot(test_sizes, f1_scores, marker='x', label='F1 Score')\nplt.xlabel('Test Size')\nplt.ylabel('Score')\nplt.title('Accuracy and F1 Score vs Test Size')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-24T12:18:51.962871Z","iopub.execute_input":"2024-06-24T12:18:51.963638Z","iopub.status.idle":"2024-06-24T12:18:57.214652Z","shell.execute_reply.started":"2024-06-24T12:18:51.963592Z","shell.execute_reply":"2024-06-24T12:18:57.213799Z"},"trusted":true},"execution_count":null,"outputs":[]}]}