{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Ink Detection Random Forest","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib import animation, rc\nrc('animation', html='jshtml')\nfrom tqdm import tqdm\nimport seaborn as sns\nimport random\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import classification_report, accuracy_score, confusion_matrix\nfrom sklearn.metrics import plot_confusion_matrix\nfrom sklearn.model_selection import KFold, train_test_split\nfrom sklearn.model_selection import ShuffleSplit\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.ensemble import RandomForestRegressor","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:05:25.582212Z","iopub.execute_input":"2023-06-07T11:05:25.582608Z","iopub.status.idle":"2023-06-07T11:05:28.906598Z","shell.execute_reply.started":"2023-06-07T11:05:25.582575Z","shell.execute_reply":"2023-06-07T11:05:28.905287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train #1 image","metadata":{}},{"cell_type":"code","source":"paths=[]\ntpaths=[]\nfor dirname, _, filenames in os.walk('/kaggle/input/vesuvius-challenge-ink-detection/train/1'):\n    for filename in filenames:\n        if filename[-4:]=='.png':\n            paths+=[(os.path.join(dirname, filename))]\n        if filename[-4:]=='.tif':\n            tpaths+=[(os.path.join(dirname, filename))]\nprint(paths)\ntpaths=sorted(tpaths)\nprint(tpaths[0:5])","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:05:28.909185Z","iopub.execute_input":"2023-06-07T11:05:28.909684Z","iopub.status.idle":"2023-06-07T11:05:28.931827Z","shell.execute_reply.started":"2023-06-07T11:05:28.909634Z","shell.execute_reply":"2023-06-07T11:05:28.930291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"factor=0.04","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:05:28.933870Z","iopub.execute_input":"2023-06-07T11:05:28.934875Z","iopub.status.idle":"2023-06-07T11:05:28.941189Z","shell.execute_reply.started":"2023-06-07T11:05:28.934818Z","shell.execute_reply":"2023-06-07T11:05:28.939714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs=[]\nfor path in paths:\n    img=cv2.imread(path)\n    img=cv2.resize(img,dsize=None,fx=factor,fy=factor)\n    imgs+=[img]\n    print(img.dtype)\n    plt.imshow(img)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:05:28.943429Z","iopub.execute_input":"2023-06-07T11:05:28.944621Z","iopub.status.idle":"2023-06-07T11:05:31.288143Z","shell.execute_reply.started":"2023-06-07T11:05:28.944568Z","shell.execute_reply":"2023-06-07T11:05:31.286691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(imgs))\nprint(imgs[0].shape)\nprint(imgs[1].shape)\nprint(len(tpaths))","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:05:31.292034Z","iopub.execute_input":"2023-06-07T11:05:31.292471Z","iopub.status.idle":"2023-06-07T11:05:31.302293Z","shell.execute_reply.started":"2023-06-07T11:05:31.292432Z","shell.execute_reply":"2023-06-07T11:05:31.300509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gray=cv2.cvtColor(imgs[0],cv2.COLOR_BGR2GRAY)\n#gray=cv2.resize(gray,dsize=None,fx=0.1,fy=0.1)\nprint(gray.shape)\nYar=gray.flatten()\ndataY=pd.DataFrame(columns=['gray'],data=Yar)\nprint(dataY.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:05:31.304552Z","iopub.execute_input":"2023-06-07T11:05:31.305635Z","iopub.status.idle":"2023-06-07T11:05:31.324076Z","shell.execute_reply.started":"2023-06-07T11:05:31.305587Z","shell.execute_reply":"2023-06-07T11:05:31.322520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs=[]\nfor path in tpaths[0:3]:\n    img=cv2.imread(path)\n    img=cv2.cvtColor(img,cv2.COLOR_BGR2GRAY)\n    img=cv2.resize(img,dsize=None,fx=factor,fy=factor)\n    imgs+=[img]\n    plt.imshow(img)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:05:31.325952Z","iopub.execute_input":"2023-06-07T11:05:31.326387Z","iopub.status.idle":"2023-06-07T11:05:37.564522Z","shell.execute_reply.started":"2023-06-07T11:05:31.326350Z","shell.execute_reply":"2023-06-07T11:05:37.563107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(imgs[0].shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:05:37.566225Z","iopub.execute_input":"2023-06-07T11:05:37.566586Z","iopub.status.idle":"2023-06-07T11:05:37.573558Z","shell.execute_reply.started":"2023-06-07T11:05:37.566554Z","shell.execute_reply":"2023-06-07T11:05:37.571991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img=cv2.imread(tpaths[0])\nimg=cv2.cvtColor(img,cv2.COLOR_BGR2GRAY)\nimg=cv2.resize(img,dsize=None,fx=factor,fy=factor)\nprint(img.shape)\nxi0=img.flatten().reshape(-1,1)\nprint(xi0.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:05:37.575995Z","iopub.execute_input":"2023-06-07T11:05:37.576537Z","iopub.status.idle":"2023-06-07T11:05:38.164878Z","shell.execute_reply.started":"2023-06-07T11:05:37.576497Z","shell.execute_reply":"2023-06-07T11:05:38.163800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Xar=xi0\nfor i,path in enumerate(tpaths[1:30]):\n    img=cv2.imread(path)\n    img=cv2.cvtColor(img,cv2.COLOR_BGR2GRAY)\n    img=cv2.resize(img,dsize=None,fx=factor,fy=factor)\n    xi=img.flatten().reshape(-1,1)\n    Xar=np.concatenate([Xar,xi],axis=1)\nprint(Xar.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:05:38.167172Z","iopub.execute_input":"2023-06-07T11:05:38.168031Z","iopub.status.idle":"2023-06-07T11:06:27.082409Z","shell.execute_reply.started":"2023-06-07T11:05:38.167991Z","shell.execute_reply":"2023-06-07T11:06:27.080922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataX=pd.DataFrame(data=Xar)\ncols=[]\nfor i in range(30):\n    cols+=[str(i).zfill(3)]\ndataX.columns=cols\ndisplay(dataX)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:06:27.084108Z","iopub.execute_input":"2023-06-07T11:06:27.084624Z","iopub.status.idle":"2023-06-07T11:06:27.123384Z","shell.execute_reply.started":"2023-06-07T11:06:27.084573Z","shell.execute_reply":"2023-06-07T11:06:27.121927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrainX, testX, trainY, testY = train_test_split(dataX, dataY, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:06:27.124901Z","iopub.execute_input":"2023-06-07T11:06:27.125318Z","iopub.status.idle":"2023-06-07T11:06:27.139337Z","shell.execute_reply.started":"2023-06-07T11:06:27.125284Z","shell.execute_reply":"2023-06-07T11:06:27.137912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(trainX.shape)\nprint(testX.shape)\nprint(trainY.shape)\nprint(testY.shape)\ntrainX=np.array(trainX)\ntrainY=np.array(trainY)\ntestX=np.array(testX)\ntestY=np.array(testY)\ntarget=['gray']\ncolumns=dataX.columns.tolist()\nprint(columns)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T11:06:27.141081Z","iopub.execute_input":"2023-06-07T11:06:27.141590Z","iopub.status.idle":"2023-06-07T11:06:27.152450Z","shell.execute_reply.started":"2023-06-07T11:06:27.141540Z","shell.execute_reply":"2023-06-07T11:06:27.150651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Forest Regressor","metadata":{}},{"cell_type":"code","source":"clf = RandomForestRegressor()\nss = ShuffleSplit(n_splits=5,train_size=0.7,test_size =0.3,random_state=25) \n\nX=trainX\ny=trainY\n\nfor train_index, test_index in ss.split(X):\n    trainx, testx = X[train_index], X[test_index]\n    trainy, testy = y[train_index], y[test_index]\n    clf.fit(trainx,trainy) \n    print(clf.score(testx, testy))\n    \ntest_pred=clf.predict(testX).astype(int)\ntest_true=np.array(testY)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nwith open('clf.pkl', 'wb') as f:\n    pickle.dump(clf, f)    ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.scatter(test_true, test_pred)\nplt.xlim(100,255)\nplt.ylim(100,255)\nplt.xlabel('test_true')\nplt.ylabel('test_pred')\nplt.title('Ink Detection')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}