summaryrefslogtreecommitdiff
path: root/boston.py
blob: 5996d59031ac3824366bf158f4c862c608b5f8b7 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
#!/usr/bin/python3
from sklearn import datasets
from sklearn import preprocessing
from sklearn.cross_validation import train_test_split
import tensorflow as tf

dataset = datasets.load_boston()
data, target = dataset.data, dataset.target

data_scaler = preprocessing.MinMaxScaler()
data = data_scaler.fit_transform(data)

# use sklearn to fit the data values between 0 and 1
x_train, x_test, y_train, y_test = train_test_split(
    data, target, train_size=0.9
)


with tf.device('/cpu:0'):
    sess = tf.InteractiveSession()

    vecUnfittedInput = tf.placeholder(tf.float32, [13], name="input")
    numTarget = tf.placeholder(tf.float32, [1], name="target")

    # numbers to scale the output between 0-1 and back to prices
    numFactor = tf.placeholder(tf.float32, [1], name="factor")
    # ^ max(input_matrix) - min(input_matrix)
    numOffset = tf.placeholder(tf.float32, [1], name="offset")
    # ^ min(input_matrix)

    # scale the unscaled input
    vecInput = tf.div(tf.sub(vecUnfittedInput, numOffset), numFactor)

    vecBias = tf.Variable(tf.truncated_normal([13], stddev=0.1), name="bias")
    vecWeights = tf.Variable(tf.truncated_normal([1, 13], stddev=0.1), name="weight")

    matWeighted = tf.mul(tf.add(vecInput, vecBias), vecWeights)
    vecNetinput = tf.reduce_sum(matWeighted)
    vecNetinput = tf.Print(vecNetinput, [vecNetinput], "netinput: ")  # DEBUG

    vecUnfittedOutput = tf.sigmoid(vecNetinput)

    # unscale the output
    vecOutput = tf.add(tf.mul(vecUnfittedOutput, numFactor), numOffset)
    # since this is the last layer, vecOutput is a numOutput
    numOutput = vecOutput

    numDifference = tf.sub(numOutput, numTarget)
    numMSE = tf.reduce_mean(tf.square(numDifference))

    summaryMSE = tf.scalar_summary("MSE", numMSE)
    summaryDifference = tf.scalar_summary("Mean difference", numDifference)
    summaries = tf.merge_all_summaries()
    writer = tf.train.SummaryWriter("/tmp/boston_logs", sess.graph_def)

    # all variables have to be specified here
    sess.run(tf.initialize_all_variables())

    # mse = tf.Print(mse, [mse], "mse: ")
    train_step = tf.train.GradientDescentOptimizer(0.01).minimize(numMSE)


factor = 0  #max(target) - min(target)
offset = 0  #min(target)

for count in range(0, len(x_train)):
    trainsteps = 100
    for i in range(0, trainsteps):  # 100 epochs
        #if i % 10 == 9:
            #feed = {
                #"input": x_test[testcount],  # x_test,
                #"target": y_test[testcount],  # [y_test],
                #"factor": factor,
                #"offset": offset
            #}
            #result = sess.run([summaries, numMSE], feed_dict=feed)
            #writer.add_summary(result[0], count * trainsteps + i)

        y_t = 1 #float(y_train[count].tolist())
        print(y_t)
        sess.run(train_step,
                 feed_dict={
                     "input": 1,  #x_train[count].tolist(),
                     "target": 1,  #y_t,
                     "factor": 1,  #[[factor]],
                     "offset": 1,  #[[offset]]
                 })

print("finished training")

#yt = [y_test]
## debug
#print(" --------- ")
#print("mean difference to test data: ")
#print(sess.run(mse, feed_dict={input_matrix: x_test, real: yt, fact: factor, offs: offset}))
#print(sess.run(diff, feed_dict={input_matrix: x_test, real: yt, fact: factor, offs: offset}))