{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-19T10:22:07.534820Z","iopub.execute_input":"2022-07-19T10:22:07.535314Z","iopub.status.idle":"2022-07-19T10:22:07.544490Z","shell.execute_reply.started":"2022-07-19T10:22:07.535282Z","shell.execute_reply":"2022-07-19T10:22:07.543420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.linear_model import LinearRegression","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:07.553176Z","iopub.execute_input":"2022-07-19T10:22:07.554095Z","iopub.status.idle":"2022-07-19T10:22:08.382427Z","shell.execute_reply.started":"2022-07-19T10:22:07.554021Z","shell.execute_reply":"2022-07-19T10:22:08.381121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nimport xgboost as xgb ","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:08.384719Z","iopub.execute_input":"2022-07-19T10:22:08.385301Z","iopub.status.idle":"2022-07-19T10:22:09.521060Z","shell.execute_reply.started":"2022-07-19T10:22:08.385254Z","shell.execute_reply":"2022-07-19T10:22:09.519917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/aviachipta-narxini-bashorat-qilish/train_data.csv')\ndf_train.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.522483Z","iopub.execute_input":"2022-07-19T10:22:09.523170Z","iopub.status.idle":"2022-07-19T10:22:09.644807Z","shell.execute_reply.started":"2022-07-19T10:22:09.523127Z","shell.execute_reply":"2022-07-19T10:22:09.644037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('../input/aviachipta-narxini-bashorat-qilish/test_data.csv')\ndf_test.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.646702Z","iopub.execute_input":"2022-07-19T10:22:09.647250Z","iopub.status.idle":"2022-07-19T10:22:09.688504Z","shell.execute_reply.started":"2022-07-19T10:22:09.647215Z","shell.execute_reply":"2022-07-19T10:22:09.687543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_result = pd.read_csv('../input/aviachipta-narxini-bashorat-qilish/sample_solution.csv')\ndf_result.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.690071Z","iopub.execute_input":"2022-07-19T10:22:09.690710Z","iopub.status.idle":"2022-07-19T10:22:09.709901Z","shell.execute_reply.started":"2022-07-19T10:22:09.690665Z","shell.execute_reply":"2022-07-19T10:22:09.708789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.711413Z","iopub.execute_input":"2022-07-19T10:22:09.711884Z","iopub.status.idle":"2022-07-19T10:22:09.761866Z","shell.execute_reply.started":"2022-07-19T10:22:09.711831Z","shell.execute_reply":"2022-07-19T10:22:09.761048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.762999Z","iopub.execute_input":"2022-07-19T10:22:09.763452Z","iopub.status.idle":"2022-07-19T10:22:09.781006Z","shell.execute_reply.started":"2022-07-19T10:22:09.763421Z","shell.execute_reply":"2022-07-19T10:22:09.780106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.782053Z","iopub.execute_input":"2022-07-19T10:22:09.782483Z","iopub.status.idle":"2022-07-19T10:22:09.807895Z","shell.execute_reply.started":"2022-07-19T10:22:09.782454Z","shell.execute_reply":"2022-07-19T10:22:09.807240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#time_di = {\"Early_Morning\":0,\"Morning\":1,\"Afternoon\":2,\"Evening\":3,\"Night\":4,\"Late_Night\":5}\nstops_di = {\"zero\":0,\"one\":1,\"two_or_more\":2}\nclass_di = {\"Economy\":0,\"Business\":1}","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.808955Z","iopub.execute_input":"2022-07-19T10:22:09.809362Z","iopub.status.idle":"2022-07-19T10:22:09.813529Z","shell.execute_reply.started":"2022-07-19T10:22:09.809335Z","shell.execute_reply":"2022-07-19T10:22:09.812899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_processing(df):\n    df.drop('flight',axis=1,inplace=True)\n    df['class'].replace(class_di,inplace=True)\n    df['stops'].replace(stops_di,inplace=True)\n   # df['departure_time'].replace(time_di,inplace=True)\n   # df['arrival_time'].replace(time_di,inplace=True)\n    return df.drop('id',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.816185Z","iopub.execute_input":"2022-07-19T10:22:09.816954Z","iopub.status.idle":"2022-07-19T10:22:09.837170Z","shell.execute_reply.started":"2022-07-19T10:22:09.816918Z","shell.execute_reply":"2022-07-19T10:22:09.835905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Test = data_processing(df_test)\nTrain = data_processing(df_train)\nX_train = Train.drop('price',axis=1)\nY_train = Train.price","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.838973Z","iopub.execute_input":"2022-07-19T10:22:09.839542Z","iopub.status.idle":"2022-07-19T10:22:09.905441Z","shell.execute_reply.started":"2022-07-19T10:22:09.839493Z","shell.execute_reply":"2022-07-19T10:22:09.904663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#num_attribs = ['class','stops','departure_time','arrival_time','duration','days_left']\n#cat_attribs = ['airline','source_city','destination_city']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.906372Z","iopub.execute_input":"2022-07-19T10:22:09.907015Z","iopub.status.idle":"2022-07-19T10:22:09.910744Z","shell.execute_reply.started":"2022-07-19T10:22:09.906979Z","shell.execute_reply":"2022-07-19T10:22:09.909954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_attribs = ['class','stops','duration','days_left']\ncat_attribs = ['airline','source_city','departure_time','arrival_time','destination_city']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.911806Z","iopub.execute_input":"2022-07-19T10:22:09.912229Z","iopub.status.idle":"2022-07-19T10:22:09.922663Z","shell.execute_reply.started":"2022-07-19T10:22:09.912201Z","shell.execute_reply":"2022-07-19T10:22:09.921130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#num_attribs = ['duration','days_left']\n#cat_attribs = ['class','stops','airline','source_city','departure_time','arrival_time','destination_city']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.923884Z","iopub.execute_input":"2022-07-19T10:22:09.924351Z","iopub.status.idle":"2022-07-19T10:22:09.934870Z","shell.execute_reply.started":"2022-07-19T10:22:09.924320Z","shell.execute_reply":"2022-07-19T10:22:09.934108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_pipeline = Pipeline([\n          ('std_scaler', StandardScaler())             \n])\nfull_pipeline = ColumnTransformer([\n    ('num', num_pipeline, num_attribs),\n    ('cat', OneHotEncoder(), cat_attribs)\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.936087Z","iopub.execute_input":"2022-07-19T10:22:09.936539Z","iopub.status.idle":"2022-07-19T10:22:09.948923Z","shell.execute_reply.started":"2022-07-19T10:22:09.936508Z","shell.execute_reply":"2022-07-19T10:22:09.947876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_prepared = full_pipeline.fit_transform(X_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:09.951692Z","iopub.execute_input":"2022-07-19T10:22:09.952053Z","iopub.status.idle":"2022-07-19T10:22:10.019900Z","shell.execute_reply.started":"2022-07-19T10:22:09.952021Z","shell.execute_reply":"2022-07-19T10:22:10.018831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#RF_model = RandomForestRegressor()\n#RF_model.fit(X_prepared, Y_train)\n#3517.74453\n#3447.50385\n#3434.53205","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:10.024566Z","iopub.execute_input":"2022-07-19T10:22:10.025488Z","iopub.status.idle":"2022-07-19T10:22:10.030871Z","shell.execute_reply.started":"2022-07-19T10:22:10.025445Z","shell.execute_reply":"2022-07-19T10:22:10.029697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\ny_train = le.fit_transform(Y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:40:03.972879Z","iopub.execute_input":"2022-07-19T10:40:03.973785Z","iopub.status.idle":"2022-07-19T10:40:03.983782Z","shell.execute_reply.started":"2022-07-19T10:40:03.973748Z","shell.execute_reply":"2022-07-19T10:40:03.982891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = xgb.XGBClassifier()\nmodel.fit(X_prepared, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:40:06.649917Z","iopub.execute_input":"2022-07-19T10:40:06.650368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Avvaliga ba'zi qiymatlarni qo'lda songa aylandirdim so'ng hammasini OneHotencoder bilan o'zgartirdim. Onehotencoderda natija yaxshiroq chiqdi","metadata":{}},{"cell_type":"code","source":"#LR_model = LogisticRegression()\n#LR_model.fit(X_prepared, Y_train)\n#5567.22847","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:10.142009Z","iopub.status.idle":"2022-07-19T10:22:10.143056Z","shell.execute_reply.started":"2022-07-19T10:22:10.142842Z","shell.execute_reply":"2022-07-19T10:22:10.142867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#DT_model = DecisionTreeRegressor()\n#DT_model.fit(X_prepared, Y_train)\n#5105.1453","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:10.144218Z","iopub.status.idle":"2022-07-19T10:22:10.144810Z","shell.execute_reply.started":"2022-07-19T10:22:10.144618Z","shell.execute_reply":"2022-07-19T10:22:10.144642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#SVM_model = SVC()\n#SVM_model.fit(X_prepared, Y_train)\n#5119.42967","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:10.145886Z","iopub.status.idle":"2022-07-19T10:22:10.146445Z","shell.execute_reply.started":"2022-07-19T10:22:10.146259Z","shell.execute_reply":"2022-07-19T10:22:10.146281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#LineR_model = LinearRegression()\n#LineR_model.fit(X_prepared, Y_train)\n#6500","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:10.147524Z","iopub.status.idle":"2022-07-19T10:22:10.148137Z","shell.execute_reply.started":"2022-07-19T10:22:10.147942Z","shell.execute_reply":"2022-07-19T10:22:10.147964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Test_prepared = full_pipeline.fit_transform(Test)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:32:22.687174Z","iopub.execute_input":"2022-07-19T10:32:22.687598Z","iopub.status.idle":"2022-07-19T10:32:22.714006Z","shell.execute_reply.started":"2022-07-19T10:32:22.687550Z","shell.execute_reply":"2022-07-19T10:32:22.713143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_predicted = model.predict(Test_prepared)\ny_predicted","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:32:25.313172Z","iopub.execute_input":"2022-07-19T10:32:25.314384Z","iopub.status.idle":"2022-07-19T10:32:57.567353Z","shell.execute_reply.started":"2022-07-19T10:32:25.314334Z","shell.execute_reply":"2022-07-19T10:32:57.566344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(y_predicted)\nsubmission['id'] = df_test['id']\nsubmission.rename({0:'price'},axis =1,inplace=True)\nsubmission = submission[['id','price']]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:10.152460Z","iopub.status.idle":"2022-07-19T10:22:10.153041Z","shell.execute_reply.started":"2022-07-19T10:22:10.152859Z","shell.execute_reply":"2022-07-19T10:22:10.152880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:22:10.154050Z","iopub.status.idle":"2022-07-19T10:22:10.154421Z","shell.execute_reply.started":"2022-07-19T10:22:10.154250Z","shell.execute_reply":"2022-07-19T10:22:10.154269Z"},"trusted":true},"execution_count":null,"outputs":[]}]}