{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![](https://i.imgur.com/kwgnmDy.png)","metadata":{}},{"cell_type":"markdown","source":"> The oil and gas industry has several important features that necessitate the search for innovative solutions - this is the continuity and high complexity of the technological chain, which begins with geological exploration and ends with the delivery of oil and gas to consumers.\n> \n> The exploration and production sector includes the search for commercial oil and gas deposits onshore and offshore, the drilling of exploratory wells and the operation of wells producing oil, gas and liquid products of deposits or a mixture thereof.\n> \n> Machine learning algorithms can be useful for solving various problems in the oil and gas industry, in particular, for forecasting the development of new profitable fields.\n> \n> You are invited to implement a machine learning algorithm that will allow you to determine the location of oil and gas deposits by various parameters: on land, at sea.\n> \n> P.s.\n> All tasks were designed and created for the Samsung Innovation Campus Bootcamp: Classical Machine Learning.\n> The data was collected both from open sources and generated independently.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport matplotlib.pyplot as plt\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2022-07-30T15:43:35.636686Z","iopub.execute_input":"2022-07-30T15:43:35.637271Z","iopub.status.idle":"2022-07-30T15:43:35.647349Z","shell.execute_reply.started":"2022-07-30T15:43:35.637233Z","shell.execute_reply":"2022-07-30T15:43:35.646584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1=pd.read_csv(\"../input/oilgas-field-prediction/oil_test.csv\");df1  #test data","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:43:35.648720Z","iopub.execute_input":"2022-07-30T15:43:35.649882Z","iopub.status.idle":"2022-07-30T15:43:35.706977Z","shell.execute_reply.started":"2022-07-30T15:43:35.649846Z","shell.execute_reply":"2022-07-30T15:43:35.706033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:43:35.708905Z","iopub.execute_input":"2022-07-30T15:43:35.709188Z","iopub.status.idle":"2022-07-30T15:43:35.715569Z","shell.execute_reply.started":"2022-07-30T15:43:35.709153Z","shell.execute_reply":"2022-07-30T15:43:35.714615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install geopy","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:43:35.717382Z","iopub.execute_input":"2022-07-30T15:43:35.718109Z","iopub.status.idle":"2022-07-30T15:43:42.469161Z","shell.execute_reply.started":"2022-07-30T15:43:35.718074Z","shell.execute_reply":"2022-07-30T15:43:42.468306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2=pd.read_csv(\"../input/oilgas-field-prediction/train_oil.csv\");df2 #train data","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:43:42.471898Z","iopub.execute_input":"2022-07-30T15:43:42.472196Z","iopub.status.idle":"2022-07-30T15:43:42.520066Z","shell.execute_reply.started":"2022-07-30T15:43:42.472164Z","shell.execute_reply":"2022-07-30T15:43:42.519309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# frames = [df1[df1.isna().any(axis=1)], df2[df2.isna().any(axis=1)]]\n  \n# result = pd.concat(frames)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:43:42.521066Z","iopub.execute_input":"2022-07-30T15:43:42.521367Z","iopub.status.idle":"2022-07-30T15:43:42.528263Z","shell.execute_reply.started":"2022-07-30T15:43:42.521328Z","shell.execute_reply":"2022-07-30T15:43:42.527511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1[df1.isna().any(axis=1)]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:43:42.530648Z","iopub.execute_input":"2022-07-30T15:43:42.531226Z","iopub.status.idle":"2022-07-30T15:43:42.534713Z","shell.execute_reply.started":"2022-07-30T15:43:42.531190Z","shell.execute_reply":"2022-07-30T15:43:42.534012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from geopy.geocoders import Nominatim\n\n# Initialize Nominatim API\ngeolocator = Nominatim(user_agent=\"MyApp\")\n\n# location = geolocator.geocode(\"BELAYIM MARINE\")\n\n# print(\"The latitude of the location is: \", location.latitude)\n# print(\"The longitude of the location is: \", location.longitude)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:43:42.536052Z","iopub.execute_input":"2022-07-30T15:43:42.536304Z","iopub.status.idle":"2022-07-30T15:43:42.545546Z","shell.execute_reply.started":"2022-07-30T15:43:42.536273Z","shell.execute_reply":"2022-07-30T15:43:42.544866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GULF OF SUEZ\n\ndf1.loc[df1['Field name'] == 'JULY', 'Longitude'] = geolocator.geocode(\"GULF OF SUEZ\").longitude\ndf1.loc[df1['Field name'] == 'JULY', 'Latitude'] = geolocator.geocode(\"GULF OF SUEZ\").latitude\n\n# DJEITUN\n\ndf1.loc[df1['Field name'] == 'DJEITUN', 'Longitude'] = geolocator.geocode(\"RED SERIES\").longitude\n\ndf1.loc[df1['Field name'] == 'ARUN', 'Longitude'] = geolocator.geocode(\"ARUN\").longitude\ndf1.loc[df1['Field name'] == 'ARUN', 'Latitude'] = geolocator.geocode(\"ARUN\").latitude\n\n# CHALYBEAT SPRINGS\t\n\ndf1.loc[df1['Field name'] == 'BALOL', 'Longitude'] = geolocator.geocode(\"KALOL\").longitude\ndf1.loc[df1['Field name'] == 'BALOL', 'Latitude'] = geolocator.geocode(\"KALOL\").latitude\n\ndf1.loc[df1['Field name'] == 'CHALYBEAT SPRINGS', 'Longitude'] = geolocator.geocode(\"SMACKOVER\").longitude\ndf1.loc[df1['Field name'] == 'CHALYBEAT SPRINGS', 'Latitude'] = geolocator.geocode(\"SMACKOVER\").latitude\n\n# GORGON\n\ndf1.loc[df1['Field name'] == 'GORGON', 'Longitude'] = geolocator.geocode(\"MUNGAROO\").longitude\ndf1.loc[df1['Field name'] == 'GORGON', 'Latitude'] = geolocator.geocode(\"MUNGAROO\").latitude\n\ndf1.loc[df1['Field name'] == 'KG', 'Longitude'] = geolocator.geocode(\"KG\").longitude\ndf1.loc[df1['Field name'] == 'KG', 'Latitude'] = geolocator.geocode(\"KG\").latitude\n\n# KHALDA\ndf1.loc[df1['Field name'] == 'KHALDA', 'Longitude'] = geolocator.geocode(\"KHALDA\").longitude\ndf1.loc[df1['Field name'] == 'KHALDA', 'Latitude'] = geolocator.geocode(\"KHALDA\").latitude\n\ndf1.loc[df1['Field name'] == 'LIUHUA 11-1', 'Longitude'] = geolocator.geocode(\"ZHUJIANG\").longitude\n\ndf1.loc[df1['Field name'] == 'MAYDAN MAHZAM', 'Longitude'] = geolocator.geocode(\"Qatar\").longitude\ndf1.loc[df1['Field name'] == 'MAYDAN MAHZAM', 'Latitude'] = geolocator.geocode(\"Qatar\").latitude\n\ndf1.loc[df1['Field name'] == 'WEIYUAN', 'Longitude'] = geolocator.geocode(\"WEIYUAN\").longitude\ndf1.loc[df1['Field name'] == 'WEIYUAN', 'Latitude'] = geolocator.geocode(\"WEIYUAN\").latitude\n\ndf1.loc[df1['Field name'] == 'VERMEJO-MOORE HOOPER', 'Longitude'] = geolocator.geocode(\"FUSSELMAN\").longitude\ndf1.loc[df1['Field name'] == 'VERMEJO-MOORE HOOPER', 'Latitude'] = geolocator.geocode(\"FUSSELMAN\").latitude\n\ndf1.loc[df1['Field name'] == 'RAMA', 'Longitude'] = geolocator.geocode(\"BATURAJA\").longitude\ndf1.loc[df1['Field name'] == 'RAMA', 'Latitude'] = geolocator.geocode(\"BATURAJA\").latitude\n\ndf1.loc[df1['Field name'] == 'PRIRAZLOM', 'Longitude'] = geolocator.geocode(\"TIMAN-PECHORA\").longitude\ndf1.loc[df1['Field name'] == 'PRIRAZLOM', 'Latitude'] = geolocator.geocode(\"TIMAN-PECHORA\").latitude\n\ndf1.loc[df1['Field name'] == 'OCTOBER', 'Longitude'] = geolocator.geocode(\"GULF OF SUEZ\").longitude\ndf1.loc[df1['Field name'] == 'OCTOBER', 'Latitude'] = geolocator.geocode(\"GULF OF SUEZ\").latitude\n\ndf2.loc[df2['Field name'] == 'BADR EL DIN-2', 'Longitude'] = geolocator.geocode(\"BAHARIYA\").longitude\ndf2.loc[df2['Field name'] == 'BADR EL DIN-2', 'Latitude'] = geolocator.geocode(\"BAHARIYA\").latitude\n\ndf2.loc[df2['Field name'] == 'ZAKUM', 'Longitude'] = geolocator.geocode(\"ZAKUM\").longitude\ndf2.loc[df2['Field name'] == 'ZAKUM', 'Latitude'] = geolocator.geocode(\"ZAKUM\").latitude\n\ndf2.loc[df2['Field name'] == 'UZEN', 'Longitude'] = geolocator.geocode(\"UZEN\").longitude\ndf2.loc[df2['Field name'] == 'UZEN', 'Latitude'] = geolocator.geocode(\"UZEN\").latitude\n\ndf2.loc[df2['Field name'] == 'SCOTT', 'Longitude'] = geolocator.geocode(\"SCOTT\").longitude\ndf2.loc[df2['Field name'] == 'SCOTT', 'Latitude'] = geolocator.geocode(\"SCOTT\").latitude\n\ndf2.loc[df2['Field name'] == 'BRIDGER LAKE', 'Longitude'] = geolocator.geocode(\"BRIDGER LAKE\").longitude\ndf2.loc[df2['Field name'] == 'BRIDGER LAKE', 'Latitude'] = geolocator.geocode(\"BRIDGER LAKE\").latitude\n\ndf2.loc[df2['Field name'] == 'YAKIN', 'Longitude'] = geolocator.geocode(\"YAKIN\").longitude\ndf2.loc[df2['Field name'] == 'YAKIN', 'Latitude'] = geolocator.geocode(\"YAKIN\").latitude\n\ndf2.loc[df2['Field name'] == 'BARQUE', 'Longitude'] = geolocator.geocode(\"BARQUE\").longitude\n\ndf2.loc[df2['Field name'] == 'ORENBURG', 'Longitude'] = geolocator.geocode(\"ORENBURG\").longitude\ndf2.loc[df2['Field name'] == 'ORENBURG', 'Latitude'] = geolocator.geocode(\"ORENBURG\").latitude\n\ndf2.loc[df2['Field name'] == 'CASHIRIARI', 'Longitude'] = geolocator.geocode(\"CASHIRIARI\").longitude\ndf2.loc[df2['Field name'] == 'CASHIRIARI', 'Latitude'] = geolocator.geocode(\"CASHIRIARI\").latitude\n\ndf2.loc[df2['Field name'] == 'ROURKE GAP', 'Longitude'] = geolocator.geocode(\"MINNELUSA\").longitude\ndf2.loc[df2['Field name'] == 'ROURKE GAP', 'Latitude'] = geolocator.geocode(\"MINNELUSA\").latitude\n\ndf2.loc[df2['Field name'] == 'ANDREW', 'Longitude'] = geolocator.geocode(\"ANDREW SANDSTONE\").longitude\ndf2.loc[df2['Field name'] == 'ANDREW', 'Latitude'] = geolocator.geocode(\"ANDREW SANDSTONE\").latitude\n\ndf2.loc[df2['Field name'] == 'QARUN', 'Longitude'] = geolocator.geocode(\"BAHARIYA\").longitude\n\ndf2.loc[df2['Field name'] == 'TALCO', 'Longitude'] = geolocator.geocode(\"PALUXY\").longitude\ndf2.loc[df2['Field name'] == 'TALCO', 'Latitude'] = geolocator.geocode(\"PALUXY\").latitude\n\ndf2.loc[df2['Field name'] == 'BEAVER LODGE', 'Longitude'] = geolocator.geocode(\"BEAVER LODGE\").longitude\ndf2.loc[df2['Field name'] == 'BEAVER LODGE', 'Latitude'] = geolocator.geocode(\"BEAVER LODGE\").latitude\n\ndf2.loc[df2['Field name'] == 'ALPINE', 'Longitude'] = geolocator.geocode(\"ALPINE\").longitude\ndf2.loc[df2['Field name'] == 'ALPINE', 'Latitude'] = geolocator.geocode(\"ALPINE\").latitude\n\ndf2.loc[df2['Field name'] == 'RHOURDE EL BAGUEL', 'Longitude'] = geolocator.geocode(\"GHADAMES\").longitude\n\ndf2.loc[df2['Field name'] == 'NORTH ROBERTSON', 'Longitude'] = geolocator.geocode(\"NORTH ROBERTSON\").longitude\ndf2.loc[df2['Field name'] == 'NORTH ROBERTSON', 'Latitude'] = geolocator.geocode(\"NORTH ROBERTSON\").latitude\n\ndf2.loc[df2['Field name'] == 'CAROLINE', 'Longitude'] = geolocator.geocode(\"SWAN HILLS\").longitude\ndf2.loc[df2['Field name'] == 'CAROLINE', 'Latitude'] = geolocator.geocode(\"SWAN HILLS\").latitude\n\ndf2.loc[df2['Field name'] == 'TABER NORTH', 'Longitude'] = geolocator.geocode(\"TABER NORTH\").longitude\ndf2.loc[df2['Field name'] == 'TABER NORTH', 'Latitude'] = geolocator.geocode(\"TABER NORTH\").latitude\n\ndf2.loc[df2['Field name'] == 'GASIKULE', 'Longitude'] = geolocator.geocode(\"GASIKULE\").longitude\ndf2.loc[df2['Field name'] == 'GASIKULE', 'Latitude'] = geolocator.geocode(\"GASIKULE\").latitude\n\ndf2.loc[df2['Field name'] == 'EMPIRE ABO', 'Longitude'] = geolocator.geocode(\"ABO\").longitude\ndf2.loc[df2['Field name'] == 'EMPIRE ABO', 'Latitude'] = geolocator.geocode(\"ABO\").latitude\n\n# TIA JUANA\t\n\ndf2.loc[df2['Field name'] == 'TIA JUANA', 'Longitude'] = geolocator.geocode(\"TIA JUANA\").longitude\ndf2.loc[df2['Field name'] == 'TIA JUANA', 'Latitude'] = geolocator.geocode(\"TIA JUANA\").latitude\n\n# TIA JUANA\t\n\ndf2.loc[df2['Field name'] == 'HARMATTAN-ELKTON', 'Longitude'] = geolocator.geocode(\"TURNER VALLEY\").longitude\ndf2.loc[df2['Field name'] == 'HARMATTAN-ELKTON', 'Latitude'] = geolocator.geocode(\"TURNER VALLEY\").latitude\n\n# HARMATTAN-ELKTON\n\ndf2.loc[df2['Field name'] == 'GLENBURN', 'Longitude'] = geolocator.geocode(\"GLENBURN\").longitude\ndf2.loc[df2['Field name'] == 'GLENBURN', 'Latitude'] = geolocator.geocode(\"GLENBURN\").latitude\n\n# OROCUAL\n\ndf2.loc[df2['Field name'] == 'OROCUAL', 'Longitude'] = geolocator.geocode(\"OROCUAL\").longitude\ndf2.loc[df2['Field name'] == 'OROCUAL', 'Latitude'] = geolocator.geocode(\"OROCUAL\").latitude\n\n# INDEFATIGABLE\t\n\ndf2.loc[df2['Field name'] == 'WUBAITI', 'Longitude'] = geolocator.geocode(\"HUANGLONG\").longitude\ndf2.loc[df2['Field name'] == 'WUBAITI', 'Latitude'] = geolocator.geocode(\"HUANGLONG\").latitude\n\ndf2.loc[df2['Field name'] == 'PALM VALLEY', 'Longitude'] = geolocator.geocode(\"PALM VALLEY\").longitude\ndf2.loc[df2['Field name'] == 'PALM VALLEY', 'Latitude'] = geolocator.geocode(\"PALM VALLEY\").latitude\n\ndf2.loc[df2['Field name'] == 'YIBAL', 'Longitude'] = geolocator.geocode(\"YIBAL\").longitude\ndf2.loc[df2['Field name'] == 'YIBAL', 'Latitude'] = geolocator.geocode(\"YIBAL\").latitude\n\ndf2.loc[df2['Field name'] == 'ULA', 'Longitude'] = geolocator.geocode(\"ULA\").longitude\ndf2.loc[df2['Field name'] == 'ULA', 'Latitude'] = geolocator.geocode(\"ULA\").latitude\n\ndf2.loc[df2['Field name'] == 'WEST SEMINOLE', 'Longitude'] = geolocator.geocode(\"WEST SEMINOLE\").longitude\ndf2.loc[df2['Field name'] == 'WEST SEMINOLE', 'Latitude'] = geolocator.geocode(\"WEST SEMINOLE\").latitude\n\ndf1.loc[df1['Field name'] == 'BELAYIM MARINE', 'Longitude'] = geolocator.geocode(\"GULF OF SUEZ\").longitude\ndf1.loc[df1['Field name'] == 'BELAYIM MARINE', 'Latitude'] = geolocator.geocode(\"GULF OF SUEZ\").latitude","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:43:42.549115Z","iopub.execute_input":"2022-07-30T15:43:42.549492Z","iopub.status.idle":"2022-07-30T15:44:25.896573Z","shell.execute_reply.started":"2022-07-30T15:43:42.549454Z","shell.execute_reply":"2022-07-30T15:44:25.895642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1_new=pd.get_dummies(df1[['Country']], prefix=['Country'])\n# df1_new[\"Country_AFGHANISTAN\"]=0\n# df1_new[\"Country_ALGERIA\"]=0\n# df1_new[\"Country_COLOMBIA\"]=0\n# df1_new[\"Country_EQUATORIAL GUINEA\"]=0\n# df1_new[\"Country_GABON\"]=0\n# df1_new[\"Country_IRAQ\"]=0\n# df1_new[\"Country_LIBYA\"]=0\n# df1_new[\"Country_NIGERIA\"]=0\n# df1_new['Country_NORWAY /UK']=0\n# df1_new[\"Country_OMAN\"]=0\n# df1_new['Country_PAPUA NEW GUINEA']=0\n# df1_new['Country_SOUTH AFRICA']=0\n# df1_new[\"Country_THAILAND\"]=0\n# df1_new['Country_TRINIDAD AND TOBAGO']=0\n# df1_new['Country_TUNISIA /ALGERIA']=0\n# df1_new['Country_TURKEY']=0\n# df1_new['Country_UAE']=0\n# df1_new['Country_UK /NORWAY']=0","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:25.898784Z","iopub.execute_input":"2022-07-30T15:44:25.899192Z","iopub.status.idle":"2022-07-30T15:44:25.902976Z","shell.execute_reply.started":"2022-07-30T15:44:25.899156Z","shell.execute_reply":"2022-07-30T15:44:25.902231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df2_new=pd.get_dummies(df2[['Country']], prefix=['Country'])\n# df2_new[\"Country_AUSTRIA\"]=0\n# df2_new[\"Country_BAHRAIN\"]=0\n# df2_new[\"Country_ECUADOR\"]=0\n# df2_new[\"Country_KUWAIT\"]=0\n# df2_new[\"Country_MAURITANIA\"]=0\n# df2_new['Country_NETHERLANDS /GERMANY']=0","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:25.906405Z","iopub.execute_input":"2022-07-30T15:44:25.906639Z","iopub.status.idle":"2022-07-30T15:44:25.914167Z","shell.execute_reply.started":"2022-07-30T15:44:25.906598Z","shell.execute_reply":"2022-07-30T15:44:25.913432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1=df1.drop([\"Country\",\"Region\",\"Field name\",\"Reservoir unit\",\"Basin name\",\"Operator company\"],axis=1)\n# df2=df2.drop([\"Country\",\"Region\",\"Field name\",\"Reservoir unit\",\"Basin name\",\"Operator company\"],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:25.915449Z","iopub.execute_input":"2022-07-30T15:44:25.915906Z","iopub.status.idle":"2022-07-30T15:44:25.924352Z","shell.execute_reply.started":"2022-07-30T15:44:25.915873Z","shell.execute_reply":"2022-07-30T15:44:25.923709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1[\"Latitude\"]=[i+360 if i<0 else i for i in df1[\"Latitude\"]]\n# df2[\"Latitude\"]=[i+360 if i<0 else i for i in df2[\"Latitude\"]]\n# df1[\"Longitude\"]=[i+360 if i<0 else i for i in df1[\"Longitude\"]]\n# df2[\"Longitude\"]=[i+360 if i<0 else i for i in df2[\"Longitude\"]]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:25.925624Z","iopub.execute_input":"2022-07-30T15:44:25.926090Z","iopub.status.idle":"2022-07-30T15:44:25.933391Z","shell.execute_reply.started":"2022-07-30T15:44:25.926057Z","shell.execute_reply":"2022-07-30T15:44:25.932586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list_lat,list_lon=[],[]\n# for i in df[\"Reservoir unit\"]:\n#     lat=geolocator.geocode(i).latitude\n#     lon=geolocator.geocode(i).longtitude\n#     try:\n#         list_lat.append(lat)\n#         list_lon.append(lon)\n#     except AttributeError:\n#         list_lat.append(0)\n#         list_lona.append(0)\n        \n#     finally:\n#         list_lat.append(0)\n#         list_lona.append(0)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:25.934723Z","iopub.execute_input":"2022-07-30T15:44:25.935186Z","iopub.status.idle":"2022-07-30T15:44:25.946975Z","shell.execute_reply.started":"2022-07-30T15:44:25.935153Z","shell.execute_reply":"2022-07-30T15:44:25.946170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pip install basemap","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:25.949498Z","iopub.execute_input":"2022-07-30T15:44:25.949965Z","iopub.status.idle":"2022-07-30T15:44:25.955544Z","shell.execute_reply.started":"2022-07-30T15:44:25.949938Z","shell.execute_reply":"2022-07-30T15:44:25.954807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lat = df2['Latitude'].values\n# lon = df2['Longitude'].values\n# depth = df2['Depth']*0.001\n# Permeability= df2['Permeability'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:25.956981Z","iopub.execute_input":"2022-07-30T15:44:25.957234Z","iopub.status.idle":"2022-07-30T15:44:25.964834Z","shell.execute_reply.started":"2022-07-30T15:44:25.957204Z","shell.execute_reply":"2022-07-30T15:44:25.964143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"float_columns= df2.select_dtypes(include=['float'])\nobj_columns= df2.select_dtypes(include=['object'])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:25.966137Z","iopub.execute_input":"2022-07-30T15:44:25.966393Z","iopub.status.idle":"2022-07-30T15:44:25.977193Z","shell.execute_reply.started":"2022-07-30T15:44:25.966362Z","shell.execute_reply":"2022-07-30T15:44:25.976472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle=LabelEncoder()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:25.978548Z","iopub.execute_input":"2022-07-30T15:44:25.979028Z","iopub.status.idle":"2022-07-30T15:44:25.987071Z","shell.execute_reply.started":"2022-07-30T15:44:25.978994Z","shell.execute_reply":"2022-07-30T15:44:25.986391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def drop_col(df):\n    df=df.drop([\"Field name\",\"Hydrocarbon type\",\"Reservoir unit\",\"Country\",\n                \"Region\",\"Reservoir period\",\"Basin name\",\"Operator company\",\"Tectonic regime\",\"Lithology\",\"Structural setting\"\n               ,\"Reservoir status\"],axis=1)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:25.988432Z","iopub.execute_input":"2022-07-30T15:44:25.989199Z","iopub.status.idle":"2022-07-30T15:44:25.996196Z","shell.execute_reply.started":"2022-07-30T15:44:25.989166Z","shell.execute_reply":"2022-07-30T15:44:25.995535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2=drop_col(df2)\ndf1=drop_col(df1)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:25.997355Z","iopub.execute_input":"2022-07-30T15:44:25.998065Z","iopub.status.idle":"2022-07-30T15:44:26.008460Z","shell.execute_reply.started":"2022-07-30T15:44:25.998003Z","shell.execute_reply":"2022-07-30T15:44:26.007864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1=pd.concat([df1, df1_new], axis=1)\n# df2=pd.concat([df2, df2_new], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.011499Z","iopub.execute_input":"2022-07-30T15:44:26.011699Z","iopub.status.idle":"2022-07-30T15:44:26.017526Z","shell.execute_reply.started":"2022-07-30T15:44:26.011677Z","shell.execute_reply":"2022-07-30T15:44:26.016692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reservoir status\tStructural setting\tDepth\tReservoir period\tLithology","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.018821Z","iopub.execute_input":"2022-07-30T15:44:26.019590Z","iopub.status.idle":"2022-07-30T15:44:26.026137Z","shell.execute_reply.started":"2022-07-30T15:44:26.019556Z","shell.execute_reply":"2022-07-30T15:44:26.025274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2[\"Onshore/Offshore\"]=le.fit_transform(df2[\"Onshore/Offshore\"])\n# df2[\"Hydrocarbon type\"]=le.fit_transform(df2[\"Hydrocarbon type\"])\n# df2[\"Reservoir status\"]=le.fit_transform(df2[\"Reservoir status\"])\n# df2[\"Structural setting\"]=le.fit_transform(df2[\"Structural setting\"])\n# df2[\"Reservoir period\"]=le.fit_transform(df2[\"Reservoir period\"])\n# df2[\"Lithology\"]=le.fit_transform(df2[\"Lithology\"])\n# df2[\"Reservoir unit\"]=le.fit_transform(df2[\"Reservoir unit\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.027639Z","iopub.execute_input":"2022-07-30T15:44:26.028365Z","iopub.status.idle":"2022-07-30T15:44:26.035468Z","shell.execute_reply.started":"2022-07-30T15:44:26.028331Z","shell.execute_reply":"2022-07-30T15:44:26.034857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr_matrix = np.corrcoef(df2).round(decimals=2)\ncorr_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.036576Z","iopub.execute_input":"2022-07-30T15:44:26.037302Z","iopub.status.idle":"2022-07-30T15:44:26.055041Z","shell.execute_reply.started":"2022-07-30T15:44:26.037265Z","shell.execute_reply":"2022-07-30T15:44:26.054388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(24,12))\nsns.heatmap(df2.corr(),annot=True,cmap=\"YlGnBu_r\")","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.056150Z","iopub.execute_input":"2022-07-30T15:44:26.056589Z","iopub.status.idle":"2022-07-30T15:44:26.813352Z","shell.execute_reply.started":"2022-07-30T15:44:26.056551Z","shell.execute_reply":"2022-07-30T15:44:26.812686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=df2.pop(\"Onshore/Offshore\")\nX=df2","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.814594Z","iopub.execute_input":"2022-07-30T15:44:26.815579Z","iopub.status.idle":"2022-07-30T15:44:26.820695Z","shell.execute_reply.started":"2022-07-30T15:44:26.815544Z","shell.execute_reply":"2022-07-30T15:44:26.819904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1[\"Hydrocarbon type\"]=le.fit_transform(df1[\"Hydrocarbon type\"])\n# df1[\"Reservoir status\"]=le.fit_transform(df1[\"Reservoir status\"])\n# df1[\"Structural setting\"]=le.fit_transform(df1[\"Structural setting\"])\n# df1[\"Reservoir period\"]=le.fit_transform(df1[\"Reservoir period\"])\n# df1[\"Lithology\"]=le.fit_transform(df1[\"Lithology\"])\n# df1[\"Reservoir unit\"]=le.fit_transform(df1[\"Reservoir unit\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.822239Z","iopub.execute_input":"2022-07-30T15:44:26.822747Z","iopub.status.idle":"2022-07-30T15:44:26.829618Z","shell.execute_reply.started":"2022-07-30T15:44:26.822710Z","shell.execute_reply":"2022-07-30T15:44:26.828980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2[\"Latitude\"]=df2[\"Latitude\"].fillna(0)\ndf2[\"Longitude\"]=df2[\"Longitude\"].fillna(0)\n# df2=df2.fillna(0)\n# df1=df1.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.830933Z","iopub.execute_input":"2022-07-30T15:44:26.831390Z","iopub.status.idle":"2022-07-30T15:44:26.839921Z","shell.execute_reply.started":"2022-07-30T15:44:26.831358Z","shell.execute_reply":"2022-07-30T15:44:26.839343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1[\"Latitude\"].isnull().sum() #13 \ndf1[\"Longitude\"].isnull().sum() #16\n# df1[\"Hydrocarbon type\"].isnull().sum() #0\n# df1[\"Reservoir status\"].isnull().sum() #0\n# df1[\"Structural setting\"].isnull().sum() #0\n# df1[\"Reservoir period\"].isnull().sum() #0\n# df1[\"Depth\"].isnull().sum() #0\n# df1[\"Lithology\"].isnull().sum() #0\n# df1[\"Thickness (gross average ft)\"].isnull().sum() #0","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.844534Z","iopub.execute_input":"2022-07-30T15:44:26.845057Z","iopub.status.idle":"2022-07-30T15:44:26.852995Z","shell.execute_reply.started":"2022-07-30T15:44:26.845023Z","shell.execute_reply":"2022-07-30T15:44:26.852198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import RobustScaler,StandardScaler\n\nrob=StandardScaler()\ndf1=rob.fit_transform(df1)\nX=rob.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.854461Z","iopub.execute_input":"2022-07-30T15:44:26.855858Z","iopub.status.idle":"2022-07-30T15:44:26.870582Z","shell.execute_reply.started":"2022-07-30T15:44:26.855828Z","shell.execute_reply":"2022-07-30T15:44:26.869973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.871843Z","iopub.execute_input":"2022-07-30T15:44:26.872273Z","iopub.status.idle":"2022-07-30T15:44:26.878935Z","shell.execute_reply.started":"2022-07-30T15:44:26.872240Z","shell.execute_reply":"2022-07-30T15:44:26.878280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n# clf = RandomForestClassifier(max_depth=20, random_state=0)\n# clf.fit(X_train, y_train)\n# y_pred=clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.880278Z","iopub.execute_input":"2022-07-30T15:44:26.880661Z","iopub.status.idle":"2022-07-30T15:44:26.886138Z","shell.execute_reply.started":"2022-07-30T15:44:26.880630Z","shell.execute_reply":"2022-07-30T15:44:26.885238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\n\n# dTree_clf = DecisionTreeClassifier()\n\n# dTree_clf.fit(X_train, y_train)\n\n# y_pred=dTree_clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.889033Z","iopub.execute_input":"2022-07-30T15:44:26.889461Z","iopub.status.idle":"2022-07-30T15:44:26.895174Z","shell.execute_reply.started":"2022-07-30T15:44:26.889432Z","shell.execute_reply":"2022-07-30T15:44:26.894519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn import svm\n\n# clf = svm.SVC()\n# clf.fit(X_train, y_train)\n# y_pred=clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.896078Z","iopub.execute_input":"2022-07-30T15:44:26.896725Z","iopub.status.idle":"2022-07-30T15:44:26.905319Z","shell.execute_reply.started":"2022-07-30T15:44:26.896689Z","shell.execute_reply":"2022-07-30T15:44:26.904555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install lightgbm","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:26.906717Z","iopub.execute_input":"2022-07-30T15:44:26.907222Z","iopub.status.idle":"2022-07-30T15:44:33.681045Z","shell.execute_reply.started":"2022-07-30T15:44:26.907185Z","shell.execute_reply":"2022-07-30T15:44:33.680214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\n# clf = lgb.LGBMClassifier()\n# clf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:33.682913Z","iopub.execute_input":"2022-07-30T15:44:33.683259Z","iopub.status.idle":"2022-07-30T15:44:33.688344Z","shell.execute_reply.started":"2022-07-30T15:44:33.683199Z","shell.execute_reply":"2022-07-30T15:44:33.687627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pip install xgboost","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:33.689934Z","iopub.execute_input":"2022-07-30T15:44:33.690343Z","iopub.status.idle":"2022-07-30T15:44:33.700127Z","shell.execute_reply.started":"2022-07-30T15:44:33.690310Z","shell.execute_reply":"2022-07-30T15:44:33.699345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:33.701604Z","iopub.execute_input":"2022-07-30T15:44:33.702186Z","iopub.status.idle":"2022-07-30T15:44:33.709915Z","shell.execute_reply.started":"2022-07-30T15:44:33.702151Z","shell.execute_reply":"2022-07-30T15:44:33.709157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import GridSearchCV\n# from sklearn.model_selection import KFold","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:33.711332Z","iopub.execute_input":"2022-07-30T15:44:33.711920Z","iopub.status.idle":"2022-07-30T15:44:33.718983Z","shell.execute_reply.started":"2022-07-30T15:44:33.711888Z","shell.execute_reply":"2022-07-30T15:44:33.718253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # XGBoost\n# n_jobs=-1 to allow run it on all cores\n# params = {\n#     \"base_score\":[0.2,0.5,1],\n#     \"booster\":\n#     'n_estimators': [100, 200, 500],\n#     'learning_rate': [0.01,0.05,0.1],\n#     'booster': ['gbtree', 'gblinear'],\n#     'gamma': [0, 0.5, 1],\n#     'reg_alpha': [0, 0.5, 1],\n#     'reg_lambda': [0.5, 1, 5],\n#     'base_score': [0.2, 0.5, 1]\n# }\n\n# gs2 = GridSearchCV(XGBClassifier(n_jobs=-1), params, n_jobs=-1, cv=KFold(n_splits=3), scoring='roc_auc')\n# gs2.fit(X_train, y_train)\n\n# print('Best score:', gs2.best_score_)\n# print('Best score:', gs2.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:33.720374Z","iopub.execute_input":"2022-07-30T15:44:33.720940Z","iopub.status.idle":"2022-07-30T15:44:33.732774Z","shell.execute_reply.started":"2022-07-30T15:44:33.720907Z","shell.execute_reply":"2022-07-30T15:44:33.731970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# XGB = XGBClassifier(random_state=200) #base_score = 0.2, booster= 'gbtree', learning_rate = 0.01, n_estimators = 100)\n# XGB.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:33.733856Z","iopub.execute_input":"2022-07-30T15:44:33.734899Z","iopub.status.idle":"2022-07-30T15:44:33.741878Z","shell.execute_reply.started":"2022-07-30T15:44:33.734870Z","shell.execute_reply":"2022-07-30T15:44:33.741022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from catboost import CatBoostClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:33.743262Z","iopub.execute_input":"2022-07-30T15:44:33.743534Z","iopub.status.idle":"2022-07-30T15:44:33.751365Z","shell.execute_reply.started":"2022-07-30T15:44:33.743501Z","shell.execute_reply":"2022-07-30T15:44:33.750510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier\nestimator = []\nestimator.append(('XGB', XGBClassifier(random_state=200,seed=100)))\n# estimator.append(('dtc', DecisionTreeClassifier()))  \n# Voting Classifier with hard voting\nclf = VotingClassifier(estimators = estimator, voting ='soft')\nclf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:33.754664Z","iopub.execute_input":"2022-07-30T15:44:33.754935Z","iopub.status.idle":"2022-07-30T15:44:35.602699Z","shell.execute_reply.started":"2022-07-30T15:44:33.754906Z","shell.execute_reply":"2022-07-30T15:44:35.601997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:35.603960Z","iopub.execute_input":"2022-07-30T15:44:35.604901Z","iopub.status.idle":"2022-07-30T15:44:35.611763Z","shell.execute_reply.started":"2022-07-30T15:44:35.604863Z","shell.execute_reply":"2022-07-30T15:44:35.611210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\ny_true = list(y_test)\ny_pred = y_pred\nconfusion_matrix(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:35.612968Z","iopub.execute_input":"2022-07-30T15:44:35.613364Z","iopub.status.idle":"2022-07-30T15:44:35.626980Z","shell.execute_reply.started":"2022-07-30T15:44:35.613333Z","shell.execute_reply":"2022-07-30T15:44:35.625832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=clf.predict(df1)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:35.628191Z","iopub.execute_input":"2022-07-30T15:44:35.629080Z","iopub.status.idle":"2022-07-30T15:44:35.637431Z","shell.execute_reply.started":"2022-07-30T15:44:35.629041Z","shell.execute_reply":"2022-07-30T15:44:35.636822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:35.640411Z","iopub.execute_input":"2022-07-30T15:44:35.641917Z","iopub.status.idle":"2022-07-30T15:44:35.648914Z","shell.execute_reply.started":"2022-07-30T15:44:35.641887Z","shell.execute_reply":"2022-07-30T15:44:35.648351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.DataFrame({\"Onshore/Offshore\":y_pred});df.index.name=\"index\"","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:35.650033Z","iopub.execute_input":"2022-07-30T15:44:35.651033Z","iopub.status.idle":"2022-07-30T15:44:35.657530Z","shell.execute_reply.started":"2022-07-30T15:44:35.651001Z","shell.execute_reply":"2022-07-30T15:44:35.656898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"Onshore/Offshore\"]=df[\"Onshore/Offshore\"].replace({0:\"OFFSHORE\",1:\"ONSHORE\",2:\"ONSHORE/OFFSHORE\"});df","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:35.659277Z","iopub.execute_input":"2022-07-30T15:44:35.659476Z","iopub.status.idle":"2022-07-30T15:44:35.675096Z","shell.execute_reply.started":"2022-07-30T15:44:35.659455Z","shell.execute_reply":"2022-07-30T15:44:35.674398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Improving the Score from 94% to 100%.So,Please wait and stay Tuned FOR THE PROGRESS............................","metadata":{}},{"cell_type":"code","source":"df.to_csv(\"subm.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-30T15:44:35.676723Z","iopub.execute_input":"2022-07-30T15:44:35.677306Z","iopub.status.idle":"2022-07-30T15:44:35.684957Z","shell.execute_reply.started":"2022-07-30T15:44:35.677271Z","shell.execute_reply":"2022-07-30T15:44:35.684220Z"},"trusted":true},"execution_count":null,"outputs":[]}]}