{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":50160,"databundleVersionId":7921029,"sourceType":"competition"},{"sourceId":175128601,"sourceType":"kernelVersion"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pickle\nimport numpy as np\nimport pandas as pd\n\n\n\ndata_path = '/kaggle/input/df-train/df_train672_pl.pkl'\nwith open(data_path, 'rb') as f:\n    df_train= pickle.load(f)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-02T15:31:41.224569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nhownul=df_train.isnull()\n\ndef predict_kmeans_batches(kmeans_model, data, batch_size=10000):\n    \"\"\"\n    使用已训练的 K-Means 模型对数据进行逐批预测，并将预测结果保存在一个与目标数据长度一致的列中。\n\n    参数:\n    kmeans_model: 已训练的 K-Means 模型\n    data: 要预测的数据，可以是 DataFrame 或数组格式\n    batch_size: 每个批次的大小，默认为10000\n\n    返回:\n    predictions: 预测结果，与目标数据长度一致的数组\n    \"\"\"\n    # 初始化预测结果列表\n    predictions = []\n\n    # 计算数据的总长度\n    total_length = len(data)\n\n    # 计算批次数量\n    num_batches = int(np.ceil(total_length / batch_size))\n\n    # 逐批预测数据\n    for i in range(num_batches):\n        print(i)\n        # 计算当前批次的起始索引和结束索引\n        start_idx = i * batch_size\n        end_idx = min((i + 1) * batch_size, total_length)\n\n        # 获取当前批次的数据\n        batch_data = data.iloc[start_idx:end_idx] if isinstance(data, pd.DataFrame) else data[start_idx:end_idx]\n\n        # 使用 K-Means 模型预测当前批次的类别\n        batch_predictions = kmeans_model.predict(batch_data)\n\n        # 将预测结果添加到预测结果列表中\n        predictions.extend(batch_predictions)\n\n    return predictions\n\n\n#for n_clusters in [3,5,7,9,11]:\n #   nulclas = KMeans(n_clusters=n_clusters, random_state=0).fit(hownul)\n  #  clas_pred = nulclas.labels_\n   # print(silhouette_score(hownul,clas_pred))\nsample_size = 0.5  # 例如，抽取50%的数据\n\n# 进行均匀随机抽样\nsampled = hownul.sample(frac=sample_size, random_state=42) \n    \n\nn_clast=6#自定义\nnulclas = KMeans(n_clusters=n_clast, random_state=44).fit(sampled)\n\ndel sampled\nimport gc\n\n\n\n\n\nclas= predict_kmeans_batches(nulclas, hownul)\n\nlen(clas)\n\n#clas_pred = nulclas.labels_\n#clas=nulclas.predict(hownul)\n#gc.collectc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(20, 9))\ncolors = clas  # 根据target值确定颜色\nsizes = df_train['target']*2+5  # 设置大小\nalphas = 0.4  # 设置透明度\nplt.scatter(df_train[\"date_decision\"], df_train['missing_values_count'], marker='o', c=colors, s=sizes, alpha=alphas)\nplt.xlabel('Week Number')\nplt.ylabel('Missing Values Count')\nplt.title('Relationship between Weeknum and Missing Values')\nplt.grid(True)\nplt.show()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-02T16:04:55.693463Z","iopub.execute_input":"2024-05-02T16:04:55.694592Z","iopub.status.idle":"2024-05-02T16:05:17.363819Z","shell.execute_reply.started":"2024-05-02T16:04:55.694545Z","shell.execute_reply":"2024-05-02T16:05:17.362483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clas=pd.DataFrame(clas)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T16:05:39.351009Z","iopub.execute_input":"2024-05-02T16:05:39.352539Z","iopub.status.idle":"2024-05-02T16:05:39.358992Z","shell.execute_reply.started":"2024-05-02T16:05:39.352483Z","shell.execute_reply":"2024-05-02T16:05:39.357504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n\n\nclas_path = 'class.pkl'\n\n\nwith open(clas_path, 'wb') as f:\n    pickle.dump(clas, f)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T16:05:37.132282Z","iopub.execute_input":"2024-05-02T16:05:37.133458Z","iopub.status.idle":"2024-05-02T16:05:37.170303Z","shell.execute_reply.started":"2024-05-02T16:05:37.133412Z","shell.execute_reply":"2024-05-02T16:05:37.169121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 根据类别绘制不同的图\nfig, axes = plt.subplots(2, 3, figsize=(40, 20))\n\n# 根据类别绘制不同的图\nfor  clss in range(6):\n    # 选择当前类别的数据\n    #subset = df[df['class'] == clss]\n    \n    # 计算子图的位置\n    row = clss // 3\n    col = clss % 3\n    alphas = 0.6*(clas==clss)+0.007\n    # 绘制当前类别的数据\n    axes[row, col].scatter(df_train[\"date_decision\"], df_train['missing_values_count'], marker='o',c=colors,s = 4,alpha=alphas)\n    \n    # 设置子图标题和标签\n    axes[row, col].set_title(f\"Class {clss}\")\n    axes[row, col].set_xlabel(\"Time\")\n    axes[row, col].set_ylabel(\"Count\")\n    \n    # 显示图例\n    axes[row, col].legend()\n\n# 调整子图之间的间距\nplt.tight_layout()\n\n# 显示大图\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T16:05:42.487064Z","iopub.execute_input":"2024-05-02T16:05:42.487548Z","iopub.status.idle":"2024-05-02T16:09:59.095757Z","shell.execute_reply.started":"2024-05-02T16:05:42.487500Z","shell.execute_reply":"2024-05-02T16:09:59.093432Z"},"trusted":true},"execution_count":null,"outputs":[]}]}