{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"},{"sourceId":848739,"sourceType":"datasetVersion","datasetId":251095},{"sourceId":180546233,"sourceType":"kernelVersion"},{"sourceId":180548762,"sourceType":"kernelVersion"},{"sourceId":180552797,"sourceType":"kernelVersion"},{"sourceId":180559085,"sourceType":"kernelVersion"}],"dockerImageVersionId":30699,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torchvision import datasets, transforms\nfrom torch.utils.data import DataLoader\n#from efficientnet_pytorch import EfficientNet\nfrom sklearn.model_selection import train_test_split\nfrom PIL import Image\nimport pandas as pd\nfrom torch.utils.data import Dataset\nimport os\nimport torch.nn.functional as F\n\nimport numpy as np \nimport pandas as pd \nimport os","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:04:01.37685Z","iopub.execute_input":"2024-05-30T05:04:01.378006Z","iopub.status.idle":"2024-05-30T05:04:01.384632Z","shell.execute_reply.started":"2024-05-30T05:04:01.377961Z","shell.execute_reply":"2024-05-30T05:04:01.383547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomImageDataset(Dataset):\n    def __init__(self, csv_file, root_dir, transform=None):\n        # csv_file: ファイルパス\n        # root_dir: 画像データが格納されているディレクトリ\n        # transform: 画像データの前処理\n        self.annotations = pd.read_csv(csv_file)\n        self.root_dir = root_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.annotations)\n\n    def __getitem__(self, idx):\n        # 画像データの読み込み\n        img_path = os.path.join(self.root_dir, self.annotations.iloc[idx, 0])\n        image = Image.open(img_path).convert(\"RGB\")\n        label = int(self.annotations.iloc[idx, 1])\n\n        if self.transform:\n            image = self.transform(image)\n\n        return image, label","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:04:02.27325Z","iopub.execute_input":"2024-05-30T05:04:02.273935Z","iopub.status.idle":"2024-05-30T05:04:02.281995Z","shell.execute_reply.started":"2024-05-30T05:04:02.273902Z","shell.execute_reply":"2024-05-30T05:04:02.28103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# インターネットアクセスしないため\npackage_path = \"../input/efficientnet-pytorch/EfficientNet-PyTorch/EfficientNet-PyTorch-master/\"\nimport sys \nsys.path.append(package_path)\nfrom efficientnet_pytorch import EfficientNet","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:04:04.563785Z","iopub.execute_input":"2024-05-30T05:04:04.564161Z","iopub.status.idle":"2024-05-30T05:04:04.60317Z","shell.execute_reply.started":"2024-05-30T05:04:04.56413Z","shell.execute_reply":"2024-05-30T05:04:04.602216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# ハイパーパラメータ\nnum_epochs = 20\nbatch_size = 32\nlearning_rate = 0.00005 #　調整済み\nnum_classes = 5  # クラス数はデータセットに応じて変更\nseed = 42\n\n# データの前処理\ntransform = transforms.Compose([\n    transforms.Resize([224,224]),\n    transforms.ToTensor(),\n    #transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\n\nbase_dir = \"/kaggle/input/cassava-leaf-disease-classification\"\n\n# データの読み込みと分割\n#annotations = pd.read_csv(f'{base_dir}/train.csv')\n#annotations = annotations.sample(2000,random_state=seed)\n\n# サンプリングに用いたデータをわかるようにしておく\n#annotations.to_csv(\"used_data.csv\",index=False)\n\n\n# データの読み込みと分割\nused_dataset = pd.read_csv(\"/kaggle/input/exp-106-dual-8000/used_data.csv\")\ntarget_dataset = pd.read_csv(\"/kaggle/input/exp-106-dual-8000/next_sampling_train.csv\")\nannotations = pd.concat([used_dataset,target_dataset],axis=0)\nprint(annotations.shape)\n\n# サンプリングに用いたデータをわかるようにしておく\nannotations.to_csv(\"used_data.csv\",index=False)\n\n\ntrain_annotations, valid_annotations = train_test_split(annotations, test_size=0.2, stratify=annotations['label'], random_state=seed)\n\n# 一時的なCSVファイルを作成\ntrain_annotations.to_csv('train_split.csv', index=False)\nvalid_annotations.to_csv('valid_split.csv', index=False)\n# カスタムデータセットのインスタンス作成\ntrain_dataset = CustomImageDataset(csv_file='train_split.csv', root_dir=f'{base_dir}/train_images', transform=transform)\nvalid_dataset = CustomImageDataset(csv_file='valid_split.csv', root_dir=f'{base_dir}/train_images', transform=transform)\n\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\nvalid_loader = DataLoader(valid_dataset, batch_size=batch_size, shuffle=False)\n\n\n# モデルの準備\n#model = EfficientNet.from_pretrained('efficientnet-b0')\nclass MyEfficientNet(nn.Module):\n    def __init__(self, num_classes):\n        super(MyEfficientNet, self).__init__()\n        self.net = EfficientNet.from_name(\"efficientnet-b0\")\n        checkpoint = torch.load(\"../input/efficientnet-pytorch/efficientnet-b0-08094119.pth\")\n        self.net.load_state_dict(checkpoint)\n        num_ftrs = self.net._fc.in_features\n        self.net._fc = nn.Linear(num_ftrs, num_classes)\n    \n        # フック用の辞書を初期化\n        self.hidden_activations = {}\n    \n    # フォワードフックを登録する関数\n    def register_hooks(self):\n        def hook(module, input, output):\n            self.hidden_activations[module] = output\n        \n        # 最後のbn層にフックを登録\n        self.net._bn1.register_forward_hook(hook)\n    \n    def forward(self, x):\n        return self.net(x)\n\nmodel = MyEfficientNet(num_classes=num_classes)\nmodel.register_hooks()\n\n# GPUが使用可能ならモデルをGPUに転送\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = model.to(device)\n\n# 損失関数とオプティマイザ\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=learning_rate)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:04:05.1001Z","iopub.execute_input":"2024-05-30T05:04:05.100747Z","iopub.status.idle":"2024-05-30T05:04:05.802027Z","shell.execute_reply.started":"2024-05-30T05:04:05.100709Z","shell.execute_reply":"2024-05-30T05:04:05.800999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 訓練関数\ndef train(model, train_loader, criterion, optimizer, device):\n    model.train()\n    running_loss = 0.0\n    for inputs, labels in train_loader:\n        inputs, labels = inputs.to(device), labels.to(device)\n\n        optimizer.zero_grad()\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item()\n    return running_loss / len(train_loader)\n\n# 評価関数\ndef evaluate(model, valid_loader, criterion, device):\n    model.eval()\n    running_loss = 0.0\n    correct = 0\n    total = 0\n    with torch.no_grad():\n        for inputs, labels in valid_loader:\n            inputs, labels = inputs.to(device), labels.to(device)\n            outputs = model(inputs)\n            loss = criterion(outputs, labels)\n            running_loss += loss.item()\n\n            _, predicted = torch.max(outputs, 1)\n            total += labels.size(0)\n            correct += (predicted == labels).sum().item()\n    accuracy = correct / total\n    return running_loss / len(valid_loader), accuracy\n","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:04:07.046057Z","iopub.execute_input":"2024-05-30T05:04:07.046449Z","iopub.status.idle":"2024-05-30T05:04:07.056563Z","shell.execute_reply.started":"2024-05-30T05:04:07.046417Z","shell.execute_reply":"2024-05-30T05:04:07.055411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# トレーニングループ\nbest_accuracy = 0.0\nbest_model_weights = None\n\n# 各エポックごとの損失と精度を保存するリスト\ntrain_losses = []\ntest_losses = []\ntest_accuracies = []\n\nfor epoch in range(num_epochs):\n    train_loss = train(model, train_loader, criterion, optimizer, device)\n    test_loss, test_accuracy = evaluate(model, valid_loader, criterion, device)\n    \n    # 各エポックの結果をリストに追加\n    train_losses.append(train_loss)\n    test_losses.append(test_loss)\n    test_accuracies.append(test_accuracy)\n\n    print(f'Epoch {epoch+1}/{num_epochs}, Train Loss: {train_loss:.4f}, Test Loss: {test_loss:.4f}, Test Accuracy: {test_accuracy:.4f}')\n    \n    # バリデーション精度が向上した場合、モデルの重みを保存\n    if test_accuracy > best_accuracy:\n        best_accuracy = test_accuracy\n        best_model_weights = model.state_dict()\n\nprint('Training complete')\n\n# 最良モデルの重みを保存\nif best_model_weights is not None:\n    torch.save(best_model_weights, 'best_model.pth')\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:04:07.336638Z","iopub.execute_input":"2024-05-30T05:04:07.336991Z","iopub.status.idle":"2024-05-30T05:05:35.123714Z","shell.execute_reply.started":"2024-05-30T05:04:07.336963Z","shell.execute_reply":"2024-05-30T05:05:35.122734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 損失と精度をグラフで表示\nepochs = range(1, num_epochs + 1)\n\nplt.figure(figsize=(12, 4))\n\n# トレーニング損失をプロット\nplt.subplot(1, 2, 1)\nplt.plot(epochs, train_losses, 'bo-', label='Train Loss')\nplt.title('Train Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\n\n# テスト損失をプロット\nplt.subplot(1, 2, 1)\nplt.plot(epochs, test_losses, 'ro-', label='Test Loss')\nplt.title('Test Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\n\n# テスト精度をプロット\nplt.subplot(1, 2, 2)\nplt.plot(epochs, test_accuracies, 'go-', label='Test Accuracy')\nplt.title('Test Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:05:35.125731Z","iopub.execute_input":"2024-05-30T05:05:35.126079Z","iopub.status.idle":"2024-05-30T05:05:35.771954Z","shell.execute_reply.started":"2024-05-30T05:05:35.126045Z","shell.execute_reply":"2024-05-30T05:05:35.770905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class InferenceDataset(Dataset):\n    def __init__(self, annotations_file, root_dir, transform=None):\n        self.annotations = pd.read_csv(annotations_file)\n        self.root_dir = root_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.annotations)\n\n    def __getitem__(self, idx):\n        img_path = os.path.join(self.root_dir, self.annotations.iloc[idx, 0])\n        image = Image.open(img_path).convert(\"RGB\")\n\n        if self.transform:\n            image = self.transform(image)\n\n        return image\n\n# トレーニング時に使用したクラス数と同じ値\nmodel = MyEfficientNet(num_classes=num_classes)\nmodel.register_hooks()\n# 学習済みモデルの重みをロード\nmodel.load_state_dict(torch.load('best_model.pth'))\n\n# GPUが使用可能ならモデルをGPUに転送\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = model.to(device)\n\n# 推論関数\ndef inference(model, data_loader, device):\n    model.eval()\n    predictions = []\n    least_confs = []\n    hidden_vectors = []\n    with torch.no_grad():\n        for images in data_loader:\n            images = images.to(device)\n            outputs = model(images)\n            #print(outputs)\n            \n            # 各サンプルごとのhidden_vectorをリストに追加\n            batch_hidden_vectors = []\n            for layer, activation in model.hidden_activations.items():\n                activation_numpy = activation.detach().cpu().numpy() \n                hidden_vectors_per_sample = np.mean(activation_numpy, axis=(2, 3), keepdims=True)\n                \n                # サンプルごとにhidden_vectorsをフラットにしてリストに追加\n                for hidden_vector in hidden_vectors_per_sample:\n                    #print(hidden_vector.shape)\n                    batch_hidden_vectors.append(hidden_vector.flatten())\n            \n            # 各サンプルごとのhidden_vectorを全体のリストに追加\n            hidden_vectors.extend(batch_hidden_vectors)\n            # 最小確信度\n            softmax_tensor = F.softmax(outputs, dim=1)\n            least_conf, _ = torch.min(1-softmax_tensor, dim=1)\n            #print(least_confidence(outputs))\n            _, predicted = torch.max(outputs, 1)\n            predictions.extend(predicted.cpu().numpy())\n            least_confs.extend(least_conf.cpu().numpy())\n    return predictions,least_confs,hidden_vectors\n    \n# 推論用データセットとデータローダーの準備\ninference_dataset = InferenceDataset(annotations_file=f'{base_dir}/train.csv', root_dir=f'{base_dir}/train_images', transform=transform)\ninference_loader = DataLoader(inference_dataset, batch_size=32, shuffle=False)\n\n# 推論の実行\npredicted_classes,least_confs,hidden_vectors = inference(model, inference_loader, device)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:05:35.773145Z","iopub.execute_input":"2024-05-30T05:05:35.773485Z","iopub.status.idle":"2024-05-30T05:11:05.393025Z","shell.execute_reply.started":"2024-05-30T05:05:35.773436Z","shell.execute_reply":"2024-05-30T05:11:05.392216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(least_confs))\nprint(len(hidden_vectors))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:11:05.394839Z","iopub.execute_input":"2024-05-30T05:11:05.395134Z","iopub.status.idle":"2024-05-30T05:11:05.399656Z","shell.execute_reply.started":"2024-05-30T05:11:05.395109Z","shell.execute_reply":"2024-05-30T05:11:05.398653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 不確実性サンプリングの結果を保存(次サイクルで使用)\nuncertainty_df = pd.read_csv(f\"{base_dir}/train.csv\")\nuncertainty_df[\"least_conf\"] = least_confs\nused_dataset = annotations.copy()\n#used_dataset = pd.read_csv(\"/kaggle/input/sampling-2000/sampling_s2000.csv\")\nuncertainty_df = uncertainty_df[~uncertainty_df[\"image_id\"].isin(used_dataset[\"image_id\"])]\nuncertainty_df.sort_values(by=\"least_conf\",ascending=False).to_csv(\"uncertain_result.csv\")\nuncertainty_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:11:06.30728Z","iopub.execute_input":"2024-05-30T05:11:06.307612Z","iopub.status.idle":"2024-05-30T05:11:06.413588Z","shell.execute_reply.started":"2024-05-30T05:11:06.307585Z","shell.execute_reply":"2024-05-30T05:11:06.412654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#不確実性サンプリングの上位50%のデータを対象に、クラスタリングを行い2000件を抽出\nnp.save(\"hidden_vectors.npy\",np.array(hidden_vectors))\n\nused_data = annotations.copy()\n#used_data = used_dataset.copy()\ntrain_data = pd.read_csv(f'{base_dir}/train.csv')\ntarget_train_data = train_data[~train_data[\"image_id\"].isin(used_data[\"image_id\"])]\ntarget_train_data = train_data[train_data[\"image_id\"].isin(uncertainty_df.sort_values(by=\"least_conf\",ascending=False).head(uncertainty_df.shape[0]//2)[\"image_id\"])]\ntarget_train_data = target_train_data.reset_index(drop=True)\nremain_index = target_train_data.index\nX = np.array(hidden_vectors)[remain_index]\nX.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:11:06.414706Z","iopub.execute_input":"2024-05-30T05:11:06.414987Z","iopub.status.idle":"2024-05-30T05:11:06.629422Z","shell.execute_reply.started":"2024-05-30T05:11:06.414962Z","shell.execute_reply":"2024-05-30T05:11:06.628502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[\"label\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:15:17.483386Z","iopub.execute_input":"2024-05-30T05:15:17.484055Z","iopub.status.idle":"2024-05-30T05:15:17.491599Z","shell.execute_reply.started":"2024-05-30T05:15:17.484023Z","shell.execute_reply":"2024-05-30T05:15:17.490603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_train_data[\"label\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:15:00.964524Z","iopub.execute_input":"2024-05-30T05:15:00.965528Z","iopub.status.idle":"2024-05-30T05:15:00.975765Z","shell.execute_reply.started":"2024-05-30T05:15:00.965485Z","shell.execute_reply":"2024-05-30T05:15:00.974683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics.pairwise import cosine_similarity\n\nclass CosineKMeans:\n    def __init__(self, n_clusters=3, max_iter=100, tol=1e-4):\n        self.n_clusters = n_clusters\n        self.max_iter = max_iter\n        self.tol = tol\n\n    def fit(self, X):\n        # ランダムにクラスタ中心を初期化\n        self.initial_indices = np.random.choice(X.shape[0], self.n_clusters, replace=False)\n        self.centers = X[self.initial_indices]\n        \n        for i in range(self.max_iter):\n            old_centers = np.copy(self.centers)\n            \n            # 各データ点を最も類似度の高いクラスタに割り当てる\n            similarities = cosine_similarity(X, self.centers)\n            self.labels_ = np.argmax(similarities, axis=1)\n            \n            # 新しいクラスタ中心を計算\n            for j in range(self.n_clusters):\n                if np.any(self.labels_ == j):\n                    self.centers[j] = X[self.labels_ == j].mean(axis=0)\n            \n            # クラスタ中心の移動が収束したかどうかをチェック\n            if np.all(np.abs(self.centers - old_centers) < self.tol):\n                break\n\n    def predict(self, X):\n        similarities = cosine_similarity(X, self.centers)\n        return np.argmax(similarities, axis=1)\n    \n    def get_centroids(self):\n        return self.centers\n\n    def get_centroid_indices(self, X):\n        centroid_indices = []\n        for j in range(self.n_clusters):\n            cluster_points = X[self.labels_ == j]\n            cluster_indices = np.where(self.labels_ == j)[0]\n            if len(cluster_points) == 0:\n                centroid_indices.append(None)\n                continue\n            similarities = cosine_similarity(cluster_points, [self.centers[j]])\n            centroid_idx = np.argmax(similarities)\n            centroid_indices.append(cluster_indices[centroid_idx])\n        return centroid_indices\n\n    def get_outliers(self, X):\n        outliers = []\n        outlier_indices = []\n        for j in range(self.n_clusters):\n            cluster_points = X[self.labels_ == j]\n            cluster_indices = np.where(self.labels_ == j)[0]\n            if len(cluster_points) == 0:\n                outliers.append(None)\n                outlier_indices.append(None)\n                continue\n            similarities = cosine_similarity(cluster_points, [self.centers[j]])\n            outlier_idx = np.argmin(similarities)\n            outliers.append(cluster_points[outlier_idx])\n            outlier_indices.append(cluster_indices[outlier_idx])\n        return outliers, outlier_indices\n\n    def random_sample_from_clusters(self, X, num_samples_per_cluster):\n        samples = []\n        sample_indices = []\n        for j in range(self.n_clusters):\n            cluster_points = X[self.labels_ == j]\n            cluster_indices = np.where(self.labels_ == j)[0]\n            if len(cluster_points) == 0:\n                samples.append([])\n                sample_indices.append([])\n                continue\n            similarities = cosine_similarity(cluster_points, [self.centers[j]])\n            centroid_idx = np.argmax(similarities)\n            outlier_idx = np.argmin(similarities)\n            valid_indices = np.delete(cluster_indices, [centroid_idx, outlier_idx])\n            valid_points = X[valid_indices]\n            \n            if len(valid_points) == 0:\n                samples.append([])\n                sample_indices.append([])\n                continue\n            \n            if len(valid_points) > num_samples_per_cluster:\n                selected_indices = np.random.choice(len(valid_points), num_samples_per_cluster, replace=False)\n                sampled_points = valid_points[selected_indices]\n                sampled_indices = valid_indices[selected_indices]\n            else:\n                sampled_points = valid_points\n                sampled_indices = valid_indices\n            \n            samples.append(sampled_points)\n            sample_indices.append(sampled_indices)\n        \n        return samples, sample_indices","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:15:59.49264Z","iopub.execute_input":"2024-05-30T05:15:59.492993Z","iopub.status.idle":"2024-05-30T05:15:59.514513Z","shell.execute_reply.started":"2024-05-30T05:15:59.492965Z","shell.execute_reply":"2024-05-30T05:15:59.513632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cosine_kmeans = CosineKMeans(n_clusters=400)\ncosine_kmeans.fit(X)\n    \ncentroid_indices = cosine_kmeans.get_centroid_indices(X)\n_, outlier_indices =cosine_kmeans.get_outliers(X)\n_, sample_indices = cosine_kmeans.random_sample_from_clusters(X, 3)\n\ntmp = []\nfor s in sample_indices:\n    for i in s:\n        tmp.append(i)\n        \nindices = [np.array(centroid_indices),np.array(outlier_indices),np.array(tmp)]\ncombined_indices = np.concatenate(indices)\n# クラスタサイズが1のとき重複が発生するため重複削除\ncombined_indices = np.unique(combined_indices)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:17:41.794493Z","iopub.execute_input":"2024-05-30T05:17:41.795312Z","iopub.status.idle":"2024-05-30T05:17:46.693623Z","shell.execute_reply.started":"2024-05-30T05:17:41.795279Z","shell.execute_reply":"2024-05-30T05:17:46.692425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(combined_indices)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:17:50.070651Z","iopub.execute_input":"2024-05-30T05:17:50.071312Z","iopub.status.idle":"2024-05-30T05:17:50.077116Z","shell.execute_reply.started":"2024-05-30T05:17:50.071282Z","shell.execute_reply":"2024-05-30T05:17:50.075834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if len(combined_indices) < 2000:\n    nonselected_indices = np.setdiff1d(np.arange(len(target_train_data)), combined_indices)\n    nonselected_rows = target_train_data.iloc[nonselected_indices]\n    add_indices = nonselected_rows.sample(2000-len(combined_indices)).index\n    #combined_indices.extend(add_indices)\n    combined_indices = np.concatenate((combined_indices, add_indices))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:18:09.32888Z","iopub.execute_input":"2024-05-30T05:18:09.32924Z","iopub.status.idle":"2024-05-30T05:18:09.339724Z","shell.execute_reply.started":"2024-05-30T05:18:09.329211Z","shell.execute_reply":"2024-05-30T05:18:09.338697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(combined_indices).nunique()","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:18:28.543742Z","iopub.execute_input":"2024-05-30T05:18:28.544426Z","iopub.status.idle":"2024-05-30T05:18:28.552316Z","shell.execute_reply.started":"2024-05-30T05:18:28.544393Z","shell.execute_reply":"2024-05-30T05:18:28.551395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_train_data.iloc[combined_indices].to_csv(\"next_sampling_train.csv\",index=False)\nnp.save(\"next_annotation_indices.npy\",combined_indices)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:18:37.565109Z","iopub.execute_input":"2024-05-30T05:18:37.565472Z","iopub.status.idle":"2024-05-30T05:18:37.575314Z","shell.execute_reply.started":"2024-05-30T05:18:37.565434Z","shell.execute_reply":"2024-05-30T05:18:37.574463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.cluster import KMeans\nfrom sklearn.decomposition import PCA\nimport matplotlib.pyplot as plt\n#可視化確認\n# PCAによる次元削減\npca = PCA(n_components=2)\ndata_2d = pca.fit_transform(X)\ncentroids_2d = pca.transform(X[combined_indices])\n\n# 可視化\nplt.figure(figsize=(14, 8))\nplt.scatter(data_2d[:, 0], data_2d[:, 1], s=1, alpha=0.5, label='Data Points')\nplt.scatter(centroids_2d[:, 0], centroids_2d[:, 1], c='red', marker='x', s=5, label='Centroids')\n#plt.scatter(data_2d[closest_indices, 0], data_2d[closest_indices, 1], c='green', marker='o', s=50, label='Closest Points')\n\nplt.legend()\nplt.title('PCA Visualization of Data Points, Centroids, and Closest Points')\nplt.xlabel('PCA Component 1')\nplt.ylabel('PCA Component 2')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-30T05:18:52.325257Z","iopub.execute_input":"2024-05-30T05:18:52.325624Z","iopub.status.idle":"2024-05-30T05:18:53.436217Z","shell.execute_reply.started":"2024-05-30T05:18:52.325594Z","shell.execute_reply":"2024-05-30T05:18:53.43532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# このサイクルでの提出結果\nclass InferenceDataset(Dataset):\n    def __init__(self, annotations_file, root_dir, transform=None):\n        self.annotations = pd.read_csv(annotations_file)\n        self.root_dir = root_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.annotations)\n\n    def __getitem__(self, idx):\n        img_path = os.path.join(self.root_dir, self.annotations.iloc[idx, 0])\n        image = Image.open(img_path).convert(\"RGB\")\n\n        if self.transform:\n            image = self.transform(image)\n\n        return image\n\n# トレーニング時に使用したクラス数と同じ値\nmodel = MyEfficientNet(num_classes=num_classes)\n# 学習済みモデルの重みをロード\nmodel.load_state_dict(torch.load('best_model.pth'))\n\n# GPUが使用可能ならモデルをGPUに転送\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = model.to(device)\n\n# 推論関数\ndef inference(model, data_loader, device):\n    model.eval()\n    predictions = []\n    least_confs = []\n    with torch.no_grad():\n        for images in data_loader:\n            images = images.to(device)\n            outputs = model(images)\n            _, predicted = torch.max(outputs, 1)\n            predictions.extend(predicted.cpu().numpy())\n           \n    return predictions\n    \n# 推論用データセットとデータローダーの準備\ninference_dataset = InferenceDataset(annotations_file=f'{base_dir}/sample_submission.csv', root_dir=f'{base_dir}/test_images', transform=transform)\n#inference_dataset = InferenceDataset(annotations_file=f'{base_dir}/train.csv', root_dir=f'{base_dir}/train_images', transform=transform)\ninference_loader = DataLoader(inference_dataset, batch_size=32, shuffle=False)\n\n# 推論の実行\npredicted_classes = inference(model, inference_loader, device)\n\n# 結果の表示\n#print(f'Predicted classes: {predicted_classes}')\n\n\nsample_submission = pd.read_csv(f\"{base_dir}/sample_submission.csv\")\nsample_submission.loc[:,\"label\"] = predicted_classes\nsample_submission.to_csv(\"submission.csv\",index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}