import numpy as np
import pandas as pd
from numpy import genfromtxt
from sklearn.model_selection import cross_val_score
from sklearn.ensemble import BaggingClassifier
from sklearn.ensemble import RandomForestClassifier

# Read in train/test sets
train = pd.read_csv('../input/train.csv', header=0,sep=',')
test = pd.read_csv('../input/test.csv', header=0,sep=',')


##### Fields to copy since we need them later but they must be dropped from train/test
# Copy yTrain
yTrain = train['label'].copy()
#####

# Drop the label column
train.drop('label',axis=1,inplace=True)  


train.info()
test.info()

randforest = RandomForestClassifier(n_estimators=100,max_features='sqrt')
clf = BaggingClassifier(randforest, n_estimators=10)
clf.fit(train, yTrain)

#score = cross_val_score(clf, train, yTrain).mean()

ypred = clf.predict(test)

#print(score)

ids = test.index + 1

test = pd.DataFrame( { 'ImageId': ids, 'Label': ypred } )
test.shape
test.head()
test.to_csv( 'digits_prediction.csv' , index = False )

print("done")