In [9]:
#NAIVE BAYES
import sys
import matplotlib.pyplot as plt
import numpy as np
%matplotlib inline
from time import time
sys.path.append("../tools")
from email_preprocess import preprocess
### features_train and features_test are the features for the training
### and testing datasets, respectively
### labels_train and labels_test are the corresponding item labels
features_train, features_test, labels_train, labels_test = preprocess()
#########################################################
### your code goes here ###
from sklearn.naive_bayes import GaussianNB
from sklearn.metrics import accuracy_score
clf = GaussianNB()
t0 = time()
clf.fit(features_train, labels_train)
arr = []
traintime = "training time: " + str(round(time()-t0, 3))
arr.append(traintime)
print traintime, "s"
t1 = time()
pred = clf.predict(features_test)
predtime = "predicting time: " + str(round(time()-t1, 3))
arr.append(predtime)
print predtime, "s"
accuracy = accuracy_score(labels_test, pred)
arr.append(accuracy)
np.save('/tmp/123', arr)
print accuracy
# rng = np.random.RandomState(10) # deterministic random data
# a = np.hstack((rng.normal(size=1000),rng.normal(loc=5, scale=2, size=1000)))
# plt.hist(a, bins='auto') # plt.hist passes it's arguments to np.histogram
# plt.title("Histogram with 'auto' bins")
#===== //==============
# gaussian_numbers = np.random.randn(1000)
# plt.hist(gaussian_numbers)
# plt.title("Gaussian Histogram")
# plt.xlabel("Value")
# plt.ylabel("Frequency")
# plt.show()
###########3################
gender = ['male','male','female','male','female']
import matplotlib.pyplot as plt
from collections import Counter
c = Counter(pred)
men = c[1]
print "No. of prodictions for Chris", men
women = c[0]
print "No. of prodictions for Sara", women
bar_heights = (men, women)
x = (1, 2)
fig, ax = plt.subplots()
width = 0.4
ax.bar(x, bar_heights, width)
ax.set_xlim((0, 3))
ax.set_ylim((0, max(men, women)*1.1))
ax.set_xticks([i+width/2 for i in x])
ax.set_xticklabels(['Cris', 'Sarah'])
plt.show()
## IT IS PENDING ADDING CODE TO DISPLAY HOW MANY EMAILS WERE PREDICTED TO BE CHRIS AND SARA,
## WWHAT EMAILS WENT TO CHRIS AND SARA
## DISPLAY GRAPHS