# This Python 3 environment comes with many helpful analytics libraries installed

# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python

# For example, here's several helpful packages to load in 



import numpy as np # linear algebra

import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)



# Input data files are available in the "../input/" directory.

# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory



import os

print(os.listdir("../input"))



# Any results you write to the current directory are saved as output.



#Importing the dataset

train_csv = pd.read_csv('../input/train.csv',nrows = 1000)

test_csv = pd.read_csv('../input/test.csv', nrows = 1000)



#drop nan rows



train_csv.columns



#Choose my predictors - pay attention to test for nan values before selecting the columns

cat_features = ["region", "city","parent_category_name","category_name","user_type"]

#cat_features = ["category_name"]

num_features = []

target = ["deal_probability"]



X_train = train_csv.loc[:, cat_features+num_features]

X_test = test_csv.loc[:,cat_features+num_features]



#Define target variable

y_train = train_csv.loc[:, target]



#Firstly I will build the entire features vectors so that when I fit my encoder I make sure I fit it through all the possible values in the cateorical data

frames = [X_train,X_test]

X = pd.concat(frames,keys=['training','test'])



from sklearn.preprocessing import LabelEncoder, OneHotEncoder

labelencoder_X = LabelEncoder()

X[cat_features] = X[cat_features].apply(labelencoder_X.fit_transform)



X_train = X.loc['training']

X_test = X.loc['test']



if X.isnull().values.any()=='True':

    sys.exit

    

'''onehotencoder = OneHotEncoder(categorical_features = [0])

X[cat_features] = onehotencoder.fit_transform(X[cat_features].values).toarray()

X[cat_features] = X[cat_features].apply(OneHotEncoder().fit_transform)

X = onehotencoder.fit_transform(X).toarray()'''



'''X_train[cat_features] = X_train[cat_features].apply(labelencoder_X.fit(X[cat_features]).transform)



X_train[cat_features] = fitter.transform(X_train[cat_features])



X_test[cat_features] = X_test[cat_features].apply(LabelEncoder().transform)



X_train = onehotencoder.transform(X_train).toarray()

X_test = onehotencoder.transform(X_test).toarray()'''



from sklearn.linear_model import LinearRegression

regressor = LinearRegression()

regressor.fit(X_train, y_train)



# Predicting the Test set result

y_pred = regressor.predict(X_test)

y_pred = np.concatenate( y_pred, axis=0 )



my_submission = pd.DataFrame({'item_id': test_csv.item_id, 'deal_probability': y_pred})

# you could use any filename. We choose submission here

my_submission.to_csv('submission.csv', index=False)