commit 4996187a8c377823286306f9737691ebbdde3836
parent e07e9ce182c11a597a41a5393be87748fbc25b77
Author: David Freifeld <freifeld.david@gmail.com>
Date: Thu, 30 Jul 2020 23:34:55 -0700
Benchmark update + better figures
Diffstat:
10 files changed, 29 insertions(+), 12 deletions(-)
diff --git a/bench/benchmark.py b/bench/benchmark.py
@@ -43,8 +43,18 @@ for i in range(1):
# plt.show()
# print(y)
+plt.plot([0.5025448077098257, 0.06346596189658764, 0.047917547554598754, 0.04105092616154024, 0.03523913263298199, 0.031739621835796136, 0.028315013611574093, 0.026019472083075448, 0.024406521083899426, 0.022933075354279908, 0.021600019709607246, 0.020935506796379722, 0.020341872339561994, 0.019887265494774282, 0.019298826768017932, 0.019021222173712722, 0.018694215646250782, 0.01869210122639975, 0.01835323961580634, 0.01805996506773795, 0.017826852610588563, 0.017619089276841214, 0.0173522467286016, 0.017464391304677935, 0.01741090491941103, 0.016986178006763254, 0.016966580251704732, 0.016847677484476468, 0.01688444716072269, 0.016816666155598632, 0.016728257525321137, 0.01650929572067639, 0.016523535019764574, 0.016360246686629634, 0.016583477585886453, 0.016515186448995516, 0.016167543447082024, 0.016142403505456843, 0.016143194300608676, 0.01618973151673178, 0.01584432253715314, 0.01614457198639892, 0.015996006825746038, 0.015867076906405406, 0.015746370663870562, 0.015635421002396035, 0.015604364669764566, 0.01580412467568558, 0.015422517508656697, 0.01570749596837556], label="Keras")
+plt.plot([0.22765946708964985, 0.05371553719734514, 0.03755848292286033, 0.031034886701316232, 0.029416448738894714, 0.028644813125990085, 0.027122890983788384, 0.025797686040006727, 0.025556786300380104, 0.025362928792686257, 0.025110247483267293, 0.024511196780086352, 0.025183815891478237, 0.024482555564073537, 0.02378877399131557, 0.024125701865020967, 0.02412329411861912, 0.022786135199350237, 0.022480409765085637, 0.021458697166071822, 0.019833367598545273, 0.02021988033199843, 0.01960400892236572, 0.019237394278116573, 0.019515914248845335, 0.018973920885966906, 0.019074104276072507, 0.019292046152524008, 0.01877620817772699, 0.018388098750312793, 0.019136795989971683, 0.01865537591056516, 0.018765945022999747, 0.018489238997267572, 0.018801562214387422, 0.018801799720918007, 0.018268155148376198, 0.019060883800566985, 0.018474303461944327, 0.018674987837030792, 0.01909543504778123, 0.018523710002715618, 0.018588384030547165, 0.017887256971702275, 0.01847935954667011, 0.018385809867809348, 0.018235951924190814, 0.018927972964227, 0.018075138759114685, 0.018163896239800564], label="Keras (set val split)")
+plt.plot([0.6356955872469517, 0.4474506605850878, 0.3890421118235231, 0.3383762977820027, 0.2570263108725408, 0.17083878727236804, 0.1206560030408328, 0.09028454783518607, 0.06834034661147965, 0.05482004367813022, 0.046493575456135125, 0.04067401763115656, 0.03674237276445186, 0.03355837383721336, 0.03091509716139769, 0.028653709084133477, 0.026661936578240484, 0.024985219962964603, 0.023432621666806992, 0.022137642019742405, 0.020931669067147784, 0.019861774528818207, 0.018922685342936688, 0.018049194997636995, 0.017238716312126513, 0.016458408508859233, 0.0158025484137779, 0.015156084850422274, 0.014549496406066243, 0.014012374438633365, 0.013499269858957372, 0.013027227670698194, 0.01258420495404383, 0.012184903993599385, 0.011772133709941639, 0.01141798619621345, 0.011055273510342334, 0.01075009333315004, 0.010450198537204518, 0.010153955363072054, 0.009865946080363197, 0.009610634188130004, 0.009366072731556546, 0.009125463513132877, 0.008900918802492984, 0.008680248000794285, 0.008473303210025152, 0.008258737640210677, 0.008091226337141558, 0.007901268149935282], label = "SciKit Learn")
+plt.plot([0.147248, 0.0459116, 0.0426555, 0.0399301, 0.038506, 0.0373933, 0.0364922, 0.0356698, 0.0350383, 0.0344831, 0.0340619, 0.0336462, 0.0332532, 0.0329713, 0.032643, 0.0323429, 0.0320908, 0.0318416, 0.0316404, 0.0315699, 0.0313803, 0.0311661, 0.0310349, 0.030904, 0.030829, 0.0307032, 0.0306185, 0.0305414, 0.0305478, 0.0304674, 0.0304535, 0.0303114, 0.0303842, 0.0302684, 0.0302227, 0.0301911, 0.0301535, 0.0301598, 0.0300992, 0.0301056, 0.0300929, 0.0301216, 0.0299648, 0.0300934, 0.0299637, 0.030046, 0.0299663, 0.0299561, 0.0299137, 0.0298976], label = "Jacobian")
+plt.legend()
+plt.title("Loss over epochs")
+plt.xlabel("Epoch")
+plt.ylabel("Loss")
+plt.show()
+
x = ['Keras', 'Scikit-Learn', 'Jacobian (Python)', 'Jacobian (C++)']
-speed = [4.71297559738, 2.074756145477295,0.05295228958129883, 0.0521543]
+speed = [2.87004685402, 2.074756145477295,0.05295228958129883, 0.0521543]
x_pos = [i for i, _ in enumerate(x)]
diff --git a/bench/kerasdemo.py b/bench/kerasdemo.py
@@ -17,6 +17,7 @@
import time
import numpy
+import matplotlib.pyplot as plt
# import tensorflow
from numpy import loadtxt
import keras
@@ -37,8 +38,9 @@ def kerasbench(batch_sz, layers):
model.add(Dense(1, activation='linear'))
opt = keras.optimizers.SGD(lr=0.0155)
model.compile(loss='mse', optimizer=opt, metrics=['accuracy'])
- model.fit(X, y, epochs=50, batch_size=batch_sz)
+ history = model.fit(X, y, epochs=50, batch_size=batch_sz)
end = time.time()
+ print(history.history['loss'])
return (end-init)
# sum = 0
@@ -46,9 +48,4 @@ def kerasbench(batch_sz, layers):
# sum += kerasbench(10, 1)
# print(sum/10)
-y2 = []
-y2.append(kerasbench(1, 1))
-y2.append(kerasbench(5, 1))
-y2.append(kerasbench(10, 1))
-y2.append(kerasbench(15, 1))
print(kerasbench(16,1))
diff --git a/bench/scikit.py b/bench/scikit.py
@@ -18,9 +18,11 @@ def bench(batch_sz):
X_train.append(i[:-1])
y_train.append(i[-1])
- clf = MLPClassifier(solver="sgd", batch_size=batch_sz, hidden_layer_sizes=(5))
+ clf = MLPClassifier(solver="sgd", momentum=0, learning_rate_init=0.0155, batch_size=batch_sz, max_iter=50, hidden_layer_sizes=(5))
clf.fit(X_train, y_train)
end = time.time()
+ print(clf.loss_curve_)
+ print(len(clf.loss_curve_))
return end-start
# i = 1
diff --git a/example.cpp b/example.cpp
@@ -19,9 +19,17 @@ double bench(int batch_sz)
net.add_layer(5, "relu");
net.add_layer(2, "linear");
net.initialize();
+ std::vector<float> vals;
for (int i = 0; i < 50; i++) {
net.train();
+ vals.push_back(net.get_cost());
}
+ std::cout << "[";
+ for (int i = 0; i < vals.size(); i++) {
+ if (i == 49) std::cout << vals[i];
+ else std::cout << vals[i] << ", ";
+ }
+ std::cout << "]";
auto end = std::chrono::high_resolution_clock::now();
return std::chrono::duration_cast<std::chrono::nanoseconds>(end - start).count() / pow(10,9);
}
diff --git a/pictures/batch_size.png b/pictures/batch_size.png
Binary files differ.
diff --git a/pictures/full_batch_size.png b/pictures/full_batch_size.png
Binary files differ.
diff --git a/pictures/loss.png b/pictures/loss.png
Binary files differ.
diff --git a/pictures/runtime2.png b/pictures/runtime2.png
Binary files differ.
diff --git a/pictures/updated_runtime.png b/pictures/updated_runtime.png
Binary files differ.
diff --git a/readme.md b/readme.md
@@ -12,13 +12,13 @@ Jacobian is a work-in-progress machine learning library written in C++ designed
## Benchmark Info
-Batch size of model vs. total runtime for this project and Keras (Keras was slow enough that it moves in steps of 20 and starts at a batch size of 20. In reality, the spike at the beginning is much larger for lower batch sizes but it throws the graph off so much you can't see any detail from Jacobian):
+One of the tradeoffs of Jacobian is that as of now it doesn't train nearly as close to perfection as other available libraries (with the benefit being the added speed. Here's a graph of Jacobian's loss over epochs on a simple task (banknote dataset with batch size 16) as compared to other libraries.
-
+
-Here's a runtime comparison for a simple example task (banknote dataset with batch size 16) between Jacobian and some other popular ML libraries:
+Here's a runtime comparison for a simple example task (same as before) between Jacobian and some other popular ML libraries:
-
+
**Coming soon:** A more detailed and current rundown of the speed of Jacobian vs popular machine learning libraries for Python (and eventually comparisons to C++ libraries as well) as well as a handy and flexible Python script for creating benchmark graphs on the fly.