chore: 添加Stock-Prediction-Models项目文件

添加了Stock-Prediction-Models项目的多个文件,包括数据集、模型代码、README文档和CSS样式文件。这些文件用于股票预测模型的训练和展示,涵盖了LSTM、GRU等深度学习模型的应用。
This commit is contained in:
zhoujie committed 2025-04-27 16:28:06 +08:00
1 parent f57150dae8
commit 2757a4d0d2
200 files changed
+79402

No files matched your search

@@ -0,0 +1,682 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.LSTMCell(size_layer, state_is_tuple = False)\n",
"\n",
" rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer = tf.placeholder(\n",
" tf.float32, (None, num_layers * 2 * size_layer)\n",
" )\n",
" self.outputs, self.last_state = tf.nn.dynamic_rnn(\n",
" drop, self.X, initial_state = self.hidden_layer, dtype = tf.float32\n",
" )\n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value = np.zeros((1, num_layers * 2 * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" ) \n",
" init_value = last_state\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value = np.zeros((1, num_layers * 2 * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value = last_state\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0812 10:02:17.549519 140290267916096 deprecation.py:323] From <ipython-input-6-d01d21f09afe>:12: LSTMCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.LSTMCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0812 10:02:17.551540 140290267916096 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f975091ada0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 10:02:17.552432 140290267916096 deprecation.py:323] From <ipython-input-6-d01d21f09afe>:16: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0812 10:02:19.808033 140290267916096 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0812 10:02:19.816455 140290267916096 deprecation.py:323] From <ipython-input-6-d01d21f09afe>:27: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0812 10:02:20.147778 140290267916096 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 10:02:20.154457 140290267916096 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:961: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 10:02:20.564182 140290267916096 deprecation.py:323] From <ipython-input-6-d01d21f09afe>:29: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [01:10<00:00, 4.33it/s, acc=97.2, cost=0.00221]\n",
"W0812 10:03:39.929984 140290267916096 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f975091add8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:09<00:00, 4.33it/s, acc=97.4, cost=0.00193]\n",
"W0812 10:04:50.024182 140290267916096 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f974694f240>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:09<00:00, 4.34it/s, acc=97.2, cost=0.00212]\n",
"W0812 10:05:59.904235 140290267916096 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f9746a5af28>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:09<00:00, 4.30it/s, acc=97.3, cost=0.00195]\n",
"W0812 10:07:10.197728 140290267916096 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f9704151390>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:09<00:00, 4.31it/s, acc=97.2, cost=0.00208]\n",
"W0812 10:08:20.024446 140290267916096 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f96b8051f98>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:09<00:00, 4.31it/s, acc=97.1, cost=0.00224]\n",
"W0812 10:09:30.567560 140290267916096 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f96a40a6fd0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:09<00:00, 4.30it/s, acc=97, cost=0.00229] \n",
"W0812 10:10:40.653531 140290267916096 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f968d66ac88>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:09<00:00, 4.23it/s, acc=97.5, cost=0.00168]\n",
"W0812 10:11:50.874499 140290267916096 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f96941b8438>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:10<00:00, 4.32it/s, acc=97.3, cost=0.00193]\n",
"W0812 10:13:01.677561 140290267916096 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f968b7442e8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:09<00:00, 4.28it/s, acc=97.8, cost=0.00115]\n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdeXxNZ/7A8c9dElm5aUpJYontIGFQFYTQjqXVGV3H1JKKH6ZqidrKoBlB/RJLLKGoGa1S7TBTZSytXzviWqot0hL0pFKExJJGUomI696b3x/3JoIECcnN8n2/Xnm59zzPOef7nDySfO/znOdo8vLyEEIIIYQQQghR+WkdHYAQQgghhBBCiEdDEjwhhBBCCCGEqCIkwRNCCCGEEEKIKkISPCGEEEIIIYSoIiTBE0IIIYQQQogqQhI8IYQQQgghhKgiJMETQgghhBBCiCpC7+gAhBBCCFE6iqJogGnAG4AB2AH8RVXVq3fUewxQAVVV1a7FHOs1IBKoC9wAdgJj84+lKEr2Hbu4Au+pqjpWUZRBwKpCZVp7eQdVVQ8rirIT6Fao3NkeS+tSNFsIIcQ9yAieEEKIR05RlGr1AaID2/s6EAoEAz7YkqrYIupFAyfvc6z9QLCqqrWAxtg+BJ6TX6iqqkf+F7Yk8DqwyV728R3lo4BfgCP28ufuKD+Qv68QQohHq1r9AhZCCAGKokwFRgB1gHPAdFVVNyuKUgO4BHRVVTXBXrc2kAw0VFX1sqIof8D2R38j4AQwUlXVo/a6Z4AVwCDbW8UdmFTUuez1dcA8YAiQBSzElpw4qapqVhSlFhAD9AWswAfA31RVtRTRpo7AEqAltsTj38AEVVVN9vIAYDHwJHATWKKq6lx7DFOAYfYYE4EXAR1wOj8W+zHigPWqqv5dUZQwe7u+w5ZkrVAU5QNgNfA7IA/4Ehitqmqmff/69hi7YfuA9RNgAnAR6K6q6jF7vTrAGfs1T7vnNxP+CPxDVdVz9n2jgf8qivKmqqo59m1dgEDgfXs7i5R/jEIsQNNiqr8CXAb2FlM+BPhIVdW8OwsURWmE7RqEFReLEEKI0pMRPCGEqH6SsP2BXQvblLz1iqLUU1X1BvAZMKBQ3f7AHnty1w5Yg206oDe2KXlb7YlhvgHA84DBnhgVeS573RHAc0BboD22xKqwDwEztiSjHdAbGF5MmyzAeOBxoDPwe2yjSCiK4gl8BXyBbZSrKfC1fb8J9pj7AjWB/wFyijnHnYKwjVI9AbwLaID/tZ+jJVAfmGmPQQdsA85iS459gU/tCeinwOBCxx0AfJ2f3CmKkqkoSpHTKu00d7yuATQrdN5lwBhsSec9KYrSVVGU37Al3K9gS4qLcq8EriEQAnxUzL6vA3tVVT1zv3iEEEKUnIzgCSFENaOqauGpcf9UFOWvQEdgC7ABW+I23V4+kFv3Vv0FWKWq6rf292sVRZkGdAL22LctLTwSdJ9z9cc2knYeQFGUKGyJGYqiPIEt6TKoqnoduKYoyqL8GIpo0+FCb88oirIK6I4tQfkDcFFV1YX28lwgvw3DgbdVVVXt73+0n9/zrgt3t1RVVfOnQ5qBU/YvgDRFUWKAv9nfd8SW+E3OHxEE9tn/XQtsUhRlqj1hCsU2spnfNsM9YvgCeFtRlI1ABrbRSAA3+7/hwLf2++Due7+bqqr7gFqKovhiS8DP3FnHnsB1p/jRwPwE7vQ9yucUUyaEEOIhSYInhBDVjKIor2MbuWpk3+SBbeQLYDfgpihKELbpmm2BzfayhsAQRVHGFjqcM7bEJd9t0/zucy6fO+oXft0QcAIuKIqSv0175/ELnac5tumcHbAlN3ogP+mrj20ksSj3KrufO9v6BLemYHra480odJ6zhZK7AqqqfqsoSg7QQ1GUC9hGGLc+YAxr7MeOw9bmhdimbZ5XFMUHW4L3ZMmaBaqqpiiK8gW20cX2dxSHAvvuk8DNLarAPhJZF/hXSWMSQgjxYCTBE0KIasQ++rIa20jZN6qqWhRF+QH7ND/7+43YpgleArapqppl3/0c8K6qqu/e4xQFU/budy7gAuBXaN/6hV6fw7aS4+NFJUVFWAHEAwNUVc1SFOUt4NVCx3qtmP3OAU2AhDu2X7P/6wbkr0hZ9446d05PnGvf1lpV1SuKoryIbXpk/nkaKIqiL6Y9a7FN07wI/EtV1dxi4r2NqqpWbKOEfwNQFKU3kGL/6gfUA07Yk2RXwFVRlIuAb1H3Mt5Bj+3a3Ol1IKqoHRRFyV/spbgEbgjwmaqqd67IKYQQ4hGRBE8IIaoXd2xJSP79XUOxLcBR2AbgcyCdW1M1wZasbVYU5Stsi4u4AT0AY6EksCTn2giMUxRlO7aEKn96IaqqXlAUZRewUFGUd4BswB/wU1V1D3fzxJaIZSuK0gJ4M/+82O59i7EnfSuwjTq2sk81/TswW1GUE9imV7YGUlRVTVMUJQUYbJ/uOYSik507Y/gN+M0+xXFyobLvsCW0UYqi/A3bPYNPqqq6316+Htv00CxsI2QPxP74Ay9s9wK2xDaKOUtVVav90QSNClX/M7Ypty8Us1DNIGxTK5Ptyfm73LpXMb9OF2z3Dxa3AuYQ4N9F9QdFUVyxTct96UHbJ4QQouRkkRUhhKhGVFU9gW0a3zfYRuhaY1sev3Cdb7ElXD7YnoWWv/0QtvuylmGbeniKe6yE+ADnWg3sAo5iG33bge1etvzk43VsydgJ+/n+hW1EqiiTsCUvWfbj/rNQHFlAL2xTFy8CPwNP24tjsCWau7AliP/ANtKFva2TsSW6AdiW9r+XSGzTGX8DtmNbsCY/Bov9/E2xrUp6HlvClV9+DtsjBfK4Y2VKRVGyFUUp/Ay5wh7Hdt2uYfterVFV9X37MW+oqnox/8se1037axRFaWA/dgP7sVoBBxRFuYbt+6Tar0Fh+SNwRSVwLtgSuLXFxPoikIltGrAQQogyosnLu++iWkIIIUSZUxTlOWClqqoNHR2LIyiKsgbbwi0zHB2LEEKIykumaAohhHAI+5S9p7GNnj2B7T6yzffcqYqyPxvuZWyPgxBCCCFKTaZoCiGEcBQNtmmNGdimaJ4EIhwakQMoijIb2yIv8++xMqUQQgjxQGSKphBCCCGEEEJUETKCJ4QQQgghhBBVRGW8B68G8BS25abv9wwfIYQQQgghhKhqdNhWlv4e23NjC1TGBO8p7lhCWgghhBBCCCGqoW7AvsIbKmOCdwEgI+MaVmvFun/Q29uD9PRsR4chKgDpCyKf9AWRT/qCKEz6g8gnfUHkK0lf0Go1eHm5gz03KqwyJngWAKs1r8IleECFjEk4hvQFkU/6gsgnfUEUJv1B5JO+IPKVoi/cdcuaLLIihBBCCCGEEFWEJHhCCCGEEEIIUUVUximaRbJYzGRkpGE2mxwWw+XLWqxWq8POX9FotTpcXT3w8KiFRqNxdDhCCCGEEEJUeVUmwcvISMPFxQ1397oOSyb0ei1msyR4AHl5eVgsZrKyMsnISOOxx+o4OiQhhBBCCCGqvCozRdNsNuHuXlNGiioIjUaDXu+EweCNyZTr6HCEEEIIIYSoFqpMggdIclcBaTRaQFaGEkIIIYQQojxUqQRPCCGEEEIIIaozSfDKiNEYx6BBrzJ06ECSk884Opy7ZGVl8fHHa4stN5lMTJgwluef/z3PP//7coxMCCGEEEIIUVqS4JWRLVs+Y9iwkXzwwQYaNGj0wPtZLHc9q7BMZGdnsWHDR8WWa7VaBgwYzOLF75VLPEIIIYQQQlQk+/YZef75Xhw79qOjQymRKrOKZkWydOlCjh6NJzn5LJs3byI2dhUHDx5g1aplWK1WDAYvJk+ehp9ffY4cOcSSJQtQlJYkJqqMGPEmbdu2IzZ2EUlJP2MymWjXrgNjx45Hp9ORlnaZxYvnc/78OQB69uxDaOhQdu36gk2bPsFsvgnA6NFv0aFDR6xWKzEx8zhy5HucnJxxc3NlxYo1xMREk52dTVjYQFxcXFi5cs1tbdDr9Tz1VBAXLqSW+/UTQgghhBDCUUwmE9HR77Js2WIaN25CnTp1HR1SiVTZBG//sQvsO3qhTI7dtU09glvXK7Y8PHwiiYkqAwaEEhzcjYyMK8yZE0Fs7Pv4+zdm27bPiYycwerVtimSp0//wuTJ0wgMbANAVNRs2rZtz9Sp72C1WomMnMH27Vvp1+8lZs16h86dg3n33fkAZGZmAhAU1Ilevfqg0WhITj7DuHGj2Lx5B6dOJRIff4j16zeh1Wq5evUqABMmTGH48FA+/HBDmVwjIYQQQgghKpukpJ8ZOXI4P/4YT2joUGbNmou7u7ujwyqRKpvgVSTHjyfQpElz/P0bA9C3bz8WLowmJ+caAH5+9QuSO7ANB588eZxPP/0YgNzcXOrUeYKcnBwSEo6yaNHygroGgwGAlJTzzJw5nbS0NPR6PVeupJOe/is+Pn6YzWaiombTvn0HunTpVl7NFkIIIYQQolLIy8tjw4Z1TJ/+NjVq1OCDDz7m+ef/6OiwSqXKJnjBre89ylaRuLq63bElj7lzF+Dr63fb1pycnGKPMXPmdMaMGU9ISA+sVis9e3bFZDLh7f0469ZtJD7+MIcOfceKFbGsWbO+DFohhBBCCCFE5ZORcYWJE8exbdsWunXrzrJlq6hXz8fRYZWaLLJSDgICWpOUlMjZs2cA2LlzG82aKbi5FT3cGxwcwvr1awsWXMnMzCQ1NQU3NzcCA9uwceOtaZX5UzSzs7MLOuL27VsxmUwAZGRkkJubS1BQZ0aOHIOHhwepqSm4u7uTm5uL2Wwuq2YLIYQQQghRoe3bZ6RHjy58+eUOIiJms2nTlkqd3EEVHsGrSLy8vJgxYxaRkdOxWCwYDF5ERMwutv64cRN5772lhIUNQKPR4OTkTHj4RHx8fImImE1MTDShof3RanX06tWHwYPDCA+fwLRpk/D09CQoqAuLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,680 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.LSTMCell(size_layer, state_is_tuple = False)\n",
"\n",
" rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer = tf.placeholder(\n",
" tf.float32, (None, num_layers * 2 * size_layer)\n",
" )\n",
" _, last_state = tf.nn.dynamic_rnn(\n",
" drop, self.X, initial_state = self.hidden_layer, dtype = tf.float32\n",
" )\n",
" \n",
" with tf.variable_scope('decoder', reuse = False):\n",
" rnn_cells_dec = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)], state_is_tuple = False\n",
" )\n",
" drop_dec = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells_dec, output_keep_prob = forget_bias\n",
" )\n",
" self.outputs, self.last_state = tf.nn.dynamic_rnn(\n",
" drop_dec, self.X, initial_state = last_state, dtype = tf.float32\n",
" )\n",
" \n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 11,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value = np.zeros((1, num_layers * 2 * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" ) \n",
" init_value = last_state\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value = np.zeros((1, num_layers * 2 * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value = last_state\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 12,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0813 21:47:16.666563 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f69f3ff7908>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 21:47:16.753933 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f69fd7aa860>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 21:47:16.834197 140095600830272 deprecation.py:323] From <ipython-input-10-f89ab136c7c7>:41: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:36<00:00, 3.11it/s, acc=97.9, cost=0.00101] \n",
"W0813 21:48:54.353741 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f69fd79e9e8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 21:48:54.437589 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f69f2dedeb8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.05it/s, acc=98.3, cost=0.00069] \n",
"W0813 21:50:34.225154 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f69f35367f0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 21:50:34.305581 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f696417da20>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.06it/s, acc=97.7, cost=0.00117] \n",
"W0813 21:52:13.825603 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f69e80d60f0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 21:52:13.908980 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f695cdf5518>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:37<00:00, 3.08it/s, acc=98.4, cost=0.000614]\n",
"W0813 21:53:52.767824 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f695d0ed1d0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 21:53:52.849310 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f693eab10f0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.03it/s, acc=98.2, cost=0.000755]\n",
"W0813 21:55:32.572073 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f693ed38cf8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 21:55:32.654169 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f693c7376a0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.07it/s, acc=98.3, cost=0.000681]\n",
"W0813 21:57:12.073868 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f693ce4e080>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 21:57:12.156364 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f693a339e80>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.01it/s, acc=97.7, cost=0.00126] \n",
"W0813 21:58:51.933507 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f693ab0ffd0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0813 21:58:52.153095 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f693801aba8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.04it/s, acc=98.5, cost=0.000589]\n",
"W0813 22:00:31.650501 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f69380c7f98>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:00:31.732362 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6935bf8ef0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.01it/s, acc=98.4, cost=0.000625]\n",
"W0813 22:02:11.445839 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6936470550>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:02:11.528598 140095600830272 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f693387feb8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:39<00:00, 3.08it/s, acc=96.8, cost=0.0027] \n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 13,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdeVxVZf7A8c+97JtAiiKruD05aj81E5VynFJbXNpmKkvT0horRbPMRs1EzdFyx1yysSyz0sYV02wZx9wql8m0OriDC4IIAsLleu+5vz/OhUDBFbyI3/frdV/c+5znnPN9Do/gl+c5zzE5HA6EEEIIIYQQQtz4zK4OQAghhBBCCCFExZAETwghhBBCCCGqCUnwhBBCCCGEEKKakARPCCGEEEIIIaoJSfCEEEIIIYQQopqQBE8IIYQQQgghqglJ8IQQQgghhBCimnB3dQBCCCGEuDpKKRMwAvg7EAR8CTyvaVrOefVuATRA0zTtznKO5QVMBB4HfIBPgcGapp1zbq8HzAbaAYXAF8AQTdNszu3vAX8GGgHPapr2YYljzwV6lTidB2DVNC3gGpovhBCiDDKCJ4QQosIppW6qPyC6sL1PA72BOCAMIzFLLKPeJOC3SxzrdaA10AxoDLQCRpXYPhtIB+oCLTCSuRdLbP/Z+Xnn+QfWNG2Apmn+RS+M5HHppRonhBDiyt1Uv4CFEEKAUup14DmgNpAKjNQ0bblzBOckcKemaXucdUOAFCBa07R0pVQ3YDxQD/gVGKBp2m5n3cPAHOAp46PyA14t61zO+m7A20AfIBeYgpGceGiaZlNKBQJTgQcAHfgAeFPTNHsZbWoDzACaAAXAv4GhmqZZndubAtOB24FzwAxN0yY4YxgO9HPGmAw8BLgBh4picR5jA7BI07T3lVJ9ne36ESPJmqOU+gCYD/wf4AC+Al7SNC3buX+kM8a7MP7A+ikwFEgD/qxp2i/OerWBw85rnnHRbyZ0B/6laVqqc99JwHdKqRc0Tct3lrXHSNrec7bzYseapGnaaed+MzESwzed22OAWZqmWYA0pdQ6oGnRzpqmvevcz3KxgJ394lGg2yXaJoQQ4irICJ4QQtx8DmAkGYFAArBIKVVX07RCYBnQs0Tdx4D/OpO7lsACjOmANYF5wCpnYlikJ9AVCHImRmWey1n3OeB+jNGgVhiJVUkfAjagIdAS6AL0L6dNduBloBbGFMJ7cI4uKaUCgG+AdRijXA2Bb537DXXG/ABQA3gWyC/nHOeLBQ4CdYC3ABPwT+c5mgCRwBhnDG5AEnAEIzkOBz5zJqCfUXr6Yk/g26LkTimVrZQqc1qlk+m8914Y0ySLzjsLGIiRdF7K+ceKcCbaYCTITyilfJVS4Rjfu3WXcczzPQpkABuvYl8hhBCXICN4Qghxk9E0reTUuM+VUv8A2gArgcUYidtI5/YnnZ8BngfmaZr2g/PzQqXUCKAt8F9n2cyi0aTLONdjGCNpRwGUUhMxEjOUUnUwkq4gTdMKgLNKqWlFMZTRph0lPh5WSs3DmEI4HWOkKE3TtCnO7RagqA39gdc0TdOcn392nv9y7g07rmla0XRIG7Df+QLIUEpN5Y/RrzYYid+wohFBYJPz60JgqVLqdU3THBhTLt8u0bagi8SwDnhNKbUEyMIYjQTwdX6NB37QNG2HUqr5JdqzDhislPoPxghmfIljncFIyJ4HcpzbFwIrLnHMsvQBPnK2VQghRAWTBE8IIW4ySqmnMUau6jmL/DFGvgD+A/gqpWIxpmu2AJY7t0UDfZRSg0oczhMjcSmSWuL9pc4Vdl79ku+jMRbiOKGUKiozn3/8EudpjDGdszVGQuIOFCV9kRgjiWW52LZLOb+tdfhjCmaAM96sEuc5UiK5K6Zp2g9KqXygo1LqBMYI46rLjGGB89gbMNo8BWOq5VGlVBhGknb7ZR7rLYyFWv6HsYjKfIyR05NKKTNGAvge0B7j+7gAYwrna5d5fJRSUUBHjNFbIYQQlUASPCGEuIkopaIx/uN+D7BV0zS7Uup/OKfmOT8vwZgmeBJI0jQt17l7KvCWpmlvXeQUxaMylzoXcAKIKLFvZIn3qRhJRq2ykqIyzAF2AT01TctVSg0B/lriWE+Us18q0ADYc175WedXX4wRK4DQ8+qcPwI1wVnWXNO000qphzCmRxadJ0op5V5OexZiTNNMA75w3ud2SZqm6RijhG8CKKW6AMecrx4YC6L86kySfQAfpVQaEH7+vYzOkdKBzhdKqeeBHZqm6UqpWkAUxj14hUCh857D8VxBgocxOrlZ07SDV7CPEEKIKyAJnhBC3Fz8MJKQovu7nsFYgKOkxRhT7zL5Y6omGMnacqXUNxiLi/hijMZsLJEEXsm5lmBMCVyDkVAVTS9E07QTSqn1wBSl1BtAHsYiHxGapv2XCwVgJGJ5SqlbgReKzotx79tUZ9I3B2PU8U/OqabvA+OUUr9iTK9sDhzTNC1DKXUM6OWc7tkHIxG8mACMqYxnnPeoDSux7UeMhHaiUupNjHsGb9c0bbNz+yKM6aG5GEnQZXE+/iAY417AJhijmGOdSdla/hg5BePxB08CD5azUE04xvfrBMb9hW/gXJRF07RTSqlDwAtKqckYI3h9gN0l9vfEGLU0AR5KKW+MRyHoJU7zNMaonxBCiEoii6wIIcRNRNO0XzGm8W3FGKFrDmw+r84PGAlXGLC2RPl2jKl1szCmHu4H+l7DueYD6zGShF0Yz3CzYSQ/YCQDnhirdWZhPHetLmV7FSN5yXUe9/MSceQCnTGmLqYB+4C/ODdPxUg012MkiP/CGOnC2dZhGIluU2BLeW11SsBYLOYMsAZjwZqiGOzO8zfEWJX0KEbCVbQ9FePxAg7g+5IHVUrlKaXuKuectTCu21mM79UCTdPecx6zUNO0tKKXM65zzvcopaKcx45yHquBs41nMUYUX9c0bX2Jcz0C3IeROO/HWI305RLb12OsYNoeYypnAdChRDvaYYzYyuMRhBCiEpkcDrnHWQghhOsppe4H5mqaFu3qWFxBKbUAY+GWUZesLIQQQpRDpmgKIYRwCaWUD8ZI2nqMRw28yR8LutxUlFL1MEbIWro4FCGEEDc4maIphBDCVUwY0xqzMKZo/gaMdmlELqCUGoexyMs7mqYdcnU8QgghbmwyRVMIIYQQQgghqokbcQTPHWNVMJleKoQQQgghhLgZlZsT3YhJUjTG6l13YaxCJoQQQgghhBA3kwiMVZcbAgdKbrgRE7yiJbK/v2gtIYQQQgghhKje6lINErwTAFlZZ9H1qnX/YM2a/mRm5rk6DFEFSF8QRaQviCLSF0RJ0h9EEekLosiV9AWz2URwsB84c6OSbsQEzw6g644ql+ABVTIm4RrSF0QR6QuiiPQFUZL0B1FE+oIochV9wX5+wY24yIoQQgghhBBCiDJIgieEEEIIIYQQ1cSNOEVTCCGEEEIIUcHsdhtZWRnYbFZXh3JTSk83o+t6qTJ3d0+Cg0Nwc7v8tE0SPCGEEEIIIQRZWRl4e/vi5xeKyWRydTg3HXd3MzbbHwmew+Hg7NkcsrIyqFWr7kX2LE2maAohhBBCCCGw2az4+dWQ5K6KMJlM+PnVuOIRVUnwhBBCCCGEEACS3FUxV/P9kARPCCGEEEIIIaoJSfCEEEIIIYQQVc7GjRt46qm/8swzT5KSctjV4VwgNzeXTz5ZWO52q9XK0KGD6Nr1Hrp2vee6xSUJnhBCCCGEEKLKWblyGf36DeCDDxYTFVXvsvez2y949vcVczgcZGWdZt8+jfz8/DLr5OXlsnjxR+Uew2w207NnL6ZPn33N8VwJWUVTCCGEEEIIUaXMnDmF3bt3kZJyhOXLl5KYOI9t27Ywb94sdF0nKCiYYcNGEBERyc6d25kxYzJKNSE5WeO5516gRYuWJCZO48CBfVitVlq2bM2gQS/j5uZGRkY606e/w9GjqQB06nQvvXs/w/r161i69FOsVivnzll58MFHaNHidtzc3Jg8eSI7d/6Eh4cnvr4+zJmzgKlTJ5GXl0ffvk/i7e3N3LkLSrXB3d2dO+6I5cSJ49f12kmCJ4QQQgghhChl8y8n2LT7RKUc+87b6hLX/OLL/sfHv0JyskbPnr2Ji7uLrKzTjB8/msTE94iJqU9S0goSEkYxf74xRfLQoYMMGzaCZs1uA2DixHG0aNGK119/A13XSUgYxZo1q+jR42HGjn2Ddu3ieOutdwDIzs4GoFWr1tx6axNOnz7NqVPpJCZO47HHnmTfPo1du7azaNFSzGYzOTk5AAwdOpz+/Xvz4YeLK+U6XS1J8IQQQgghhBBV2t69e2jQoDExMfUBeOCBHkyZMon8/LMAREREFid3AJs2beS33/by2WefAGCxWKhduw75+fns2bObadPeLa5bo0YNMjLS2bZtCytXLiMvLxcvL2+ys7M4fTqTsLAIbDYbEyeOo1Wr1rRvf9d1bPmVkwRPCCGEEEIIUUpc80uPslUlPj6+55U4mDBhMuHhEaVKz7+fLicnh+PHj2KxWFiw4D0GDhzCPfd0Qdd1OnW6E6vVSs2atfj44yXs2rWD7dt/ZM6cRBYsWFTJLbp6ssiKEEIIIYQQokpr2rQ5Bw4kc+TIYQDWrk2iUSOFr69fmfXj4jqwaNHC4gVXsrOzOX78GL6+vjRrdhuLF3/EwYMHOHhwPzk5OcTENKCwsLB4MZc1a1ZhtRoPGM/KysJisRAb244BAwbi7+/P8ePH8PPzw2KxYLPZKr39V0JG8IQQQgghhBBVWnBwMKNGjSUhYSR2u52goGBGjx5Xbv3Bg19h9uyZ9O3bE5PJhIeHJ/Hxr1CnTih///tLzJ49k9WrV+Dp6cV99z1AmzZtiY8fyogRrxIQEEBsbHsCAwMBSE8/yaRJ47Hb7djtdtq2bU/Tps0xm8106XI/ffo8QUBAjQsWWQHo3/9pMjJOkpuLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,736 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.LSTMCell(size_layer, state_is_tuple = False)\n",
"\n",
" backward_rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" forward_rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop_backward = tf.contrib.rnn.DropoutWrapper(\n",
" backward_rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" forward_backward = tf.contrib.rnn.DropoutWrapper(\n",
" forward_rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.backward_hidden_layer = tf.placeholder(\n",
" tf.float32, shape = (None, num_layers * 2 * size_layer)\n",
" )\n",
" self.forward_hidden_layer = tf.placeholder(\n",
" tf.float32, shape = (None, num_layers * 2 * size_layer)\n",
" )\n",
" _, last_state = tf.nn.bidirectional_dynamic_rnn(\n",
" forward_backward,\n",
" drop_backward,\n",
" self.X,\n",
" initial_state_fw = self.forward_hidden_layer,\n",
" initial_state_bw = self.backward_hidden_layer,\n",
" dtype = tf.float32,\n",
" )\n",
" \n",
" with tf.variable_scope('decoder', reuse = False):\n",
" backward_rnn_cells_decoder = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" forward_rnn_cells_decoder = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" drop_backward_decoder = tf.contrib.rnn.DropoutWrapper(\n",
" backward_rnn_cells_decoder, output_keep_prob = forget_bias\n",
" )\n",
" forward_backward_decoder = tf.contrib.rnn.DropoutWrapper(\n",
" forward_rnn_cells_decoder, output_keep_prob = forget_bias\n",
" )\n",
" self.outputs, self.last_state = tf.nn.bidirectional_dynamic_rnn(\n",
" forward_backward_decoder, drop_backward_decoder, self.X, \n",
" initial_state_fw = last_state[0],\n",
" initial_state_bw = last_state[1],\n",
" dtype = tf.float32\n",
" )\n",
" self.outputs = tf.concat(self.outputs, 2)\n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value_forward = np.zeros((1, num_layers * 2 * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * 2 * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" ) \n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value_forward = np.zeros((1, num_layers * 2 * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * 2 * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 11,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0813 22:30:03.664880 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6c6a29f9e8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:30:03.666436 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6c6a0dba58>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:30:03.827417 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6c6a0db5f8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:30:03.828239 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6c6a0dbe48>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0813 22:30:03.988492 140106178451264 deprecation.py:323] From <ipython-input-10-79385dfa86b9>:67: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [02:29<00:00, 2.03it/s, acc=96.4, cost=0.00318] \n",
"W0813 22:32:35.430384 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6c68a2de10>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:32:35.431268 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6c689d96a0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:32:35.592414 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6c689d91d0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:32:35.593283 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6c680bc630>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:30<00:00, 2.00it/s, acc=98.1, cost=0.000912]\n",
"W0813 22:35:08.073616 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6c68a6df28>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:35:08.074523 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6c40475208>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:35:08.237059 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bbbf9afd0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:35:08.237945 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bbbf56fd0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:31<00:00, 1.99it/s, acc=98.2, cost=0.000814]\n",
"W0813 22:37:40.822348 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6c40403080>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:37:40.823194 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bba08cac8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:37:40.984025 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bb9234278>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:37:40.984853 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bb90d59e8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:31<00:00, 1.98it/s, acc=98.2, cost=0.000732]\n",
"W0813 22:40:14.461846 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bb230a7f0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:40:14.462596 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bb31e4588>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:40:14.624744 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bb2361198>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:40:14.625597 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bb223d5c0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:28<00:00, 2.04it/s, acc=98.6, cost=0.000437]\n",
"W0813 22:42:44.319911 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6baf458dd8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:42:44.320685 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bb0355400>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:42:44.481708 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6baf534d30>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:42:44.482524 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6baf3f2c88>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:30<00:00, 2.00it/s, acc=98.8, cost=0.000304]\n",
"W0813 22:45:16.281273 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bb04d4240>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:45:16.282183 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bad522c18>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:45:16.443265 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bad522940>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:45:16.444107 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6bac5314e0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:31<00:00, 1.99it/s, acc=98.5, cost=0.000517]\n",
"W0813 22:47:49.574930 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6ba974ef28>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:47:49.575763 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6baa69bc18>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:47:49.735839 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6ba9794a20>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:47:49.736666 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6ba7e914a8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:31<00:00, 1.99it/s, acc=96.9, cost=0.00272] \n",
"W0813 22:50:22.981087 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6ba69149b0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:50:22.982089 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6ba7828080>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:50:23.141866 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6ba69434e0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:50:23.142687 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6ba68067b8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:31<00:00, 1.99it/s, acc=98.2, cost=0.000705]\n",
"W0813 22:52:55.940449 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6ba3aacbe0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:52:55.941309 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6ba49a4080>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:52:56.101360 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6ba21a14e0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0813 22:52:56.102182 140106178451264 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f6ba39be898>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:27<00:00, 2.03it/s, acc=98.6, cost=0.000453]\n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 12,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdeXgUVdbA4V9vWTo7kADZIGylsgiIBIjgBqgIuM04IiI44IjKJoo6gAwBVFBWoyzioAiiwoxsAdQPR2VYRJYogk6FfUkCCZCQjU6nu/r7ozohgQQCJHQI532efrq76lbVqcql6dP31r0Gl8uFEEIIIYQQQojrn9HTAQghhBBCCCGEqByS4AkhhBBCCCFEDSEJnhBCCCGEEELUEJLgCSGEEEIIIUQNIQmeEEIIIYQQQtQQkuAJIYQQQgghRA0hCZ4QQgghhBBC1BBmTwcghBBCiCujKIoBGA08BwQDa4G/qaqafV65WoAKqKqq3lHOvgYA/wTOlljcU1XVH9zrJwIPAzcDk1RVHV9i29HuOIqYAG8gTFXVk+7jzwG6Ai7gG+D58+MUQghx9aQFTwghRKVTFOWG+gHRg+f7NNAPiAPCAV8goYxyU4A/KrC/Laqq+pd4/FBi3T7gVWDN+RupqvpWye3cx/tBVdWT7iKTgBAgBmgM1AXGVyAeIYQQl+mG+g9YCCEEKIryOvAsEAYcBcaoqrpcURRv4ARwh6qqu91lQ4EjQANVVdMVRemJ/mW9IfA7MFhV1V3usofQW2n66m8VP+CVso7lLm8C3gH6AznANPTkxKKqqkNRlCBgOtAD0ICPgX+oquos45zaA7PQW5fOAv8GRqqqanevbw7MBG4DCoFZqqq+5Y7hNWCgO8Zk9FYqE3CwKBb3Pn4AFquq+pG7tetZ4Gf0JGuOoigfA/OBWznXSvWiqqpZ7u2j3DF2Rv+B9XNgJHAcuFNV1d/c5cKAQ+5rnnHRPyb0Av6pqupR97ZTgP8oivK8qqr57mWdgBbAh+7zvCKqqi5076/vxcq5WxWfBuJLLI4BVhS12CmKshzofaWxCCGEKJ+04AkhxI1nP3qSEYT+JXyxoij1VVUtAL4C+pQo+zjwozu5awMsQO8OWBuYB6xyJ4ZF+gAPAsHuxKjMY7nLPgs8ALQG2qInViV9AjiAJkAboDswqJxzcgIvAXWAjsC9wAsAiqIEAOuBr9FbuZoA37m3G+mOuQcQCPwVyC/nGOeLBQ6gt0a9CRiAt93HuBmIwt1K5U4kE4HD6MlxBPCFOwH9AniqxH77AN8VJXeKomQpilJmt0o3w3mvvYGmJY77PjAEPem8lDaKopxUFCVZUZQ3rrBlsjN6svzvEss+AHoqihKiKEoI8Biw7gr2LYQQ4hKkBU8IIW4wqqouK/H2S0VR/g60B1YCS9ATtzHu9U+63wP8DZinqupW9/uF7nuvOgA/upe9V9SaVIFjPY7eknYMQFGUyeiJGYqi1EVPuoJVVT0L5CmKMqMohjLOaUeJt4cURZkH3IneatcTOK6q6jT3ehtQdA6DgFdVVVXd7391Hz/gggt3oVRVVYu6QzrQuzDuc7/PUBRlOvAP9/v26InfqKIWQWCj+3khsExRlNdVVXWhd7l8p8S5BV8khq+BVxVFWQpkordGAljdz8OAraqq7lAUpeUlzmcDekvfYaA58KX7vN6+xHbn6w/8S1XV3BLLdgJewCn3+++A2Ze5XyGEEBUgCZ4QQtxgFEV5Gr3lqqF7kT96yxfA94BVUZRY9O6arYHl7nUNgP6KogwtsTsv9MSlyNESry91rPDzypd83QCwAGmKohQtM56//xLHaYbenbMdenJjBoqSvij0lsSyXGzdpZx/rnU51wUzwB1vZonjHC6R3BVTVXWroij5wF2KoqShtzCuqmAMC9z7/gH9nKehd9s8pihKOHqCd1tFdqSq6oESb39TFGUCMIrLSPAURbECfwYeOm/VUmCXe7kBmAosRk/yhRBCVCJJ8IQQ4gaiKEoD9PvE7kUfUMOpKMovuLv5ud8vRe8meAJIVFU1x735UeBNVVXfvMghirsBXupYQBoQWWLbqBKvjwIFQJ2ykqIyzAGSgD6qquYoijIC+FOJfT1RznZH0Qf92H3e8jz3sxUoGumx3nllzu/y+JZ7WUtVVU8rivIwevfIouNEK4piLud8FqJ30zyO3vplKyfeUlRV1dBbCf8BoChKdyDF/egN1Ad+dyfJvoCvoijHgYiy7mUs4/wMlyhzvkeA0+gJZ0mt0e9HzHPHOZdzLZhCCCEqkSR4QghxY/FD/+JedH/XM+jd8kpaAqxA7043psTy+cByRVHWow8uYgXuAjaUSAIv51hLgeGKoqxBT6iKuheiqmqaoijfAtMURXkDyEUfqCNSVdUfuVAAeiKWqyjKTcDzRcdFv/dtujvpm4Pe6niLu6vpR8BERVF+R+9e2RJIUVU1Q1GUFOApd3fP/uiJ4MUEAGeAM4qiRKC3fhX5GT2hnawoyj/Q7xm8TVXVTe71i9G7h+agd9GsEPf0AyHo9wLejN6KOUFVVU1RlHWcazkF+At6l9uHyhmo5gFgp6qqJ9zX8A1gWYn1FvTBZ4yAWVEUH6DwvH31Bz51dzUtaRswSFGUV93v/4beoieEEKKSySArQghxA1FV9Xf0bnxb0FvoWgKbziuzFT3hCqfEQBiqqm5HHxjlffSuh/uAAVdxrPnAt+hf9JPQ53BzoCc/oI/E6IU+Wmcm8C/0FqmyvIKevOS49/tliThygG7oXRePA3uBu92rp6Mnmt+iJ4j/RG/pwn2uo9AT3ebA5vLO1S0efbCYM+hTCXxVIgan+/hN0EclPYaecBWtP4p+n5oL+G/JnSqKkqsoSudyjlkH/brlof+tFqiq+qF7nwWqqh4verjjKnS/RlGUaPe+o937uhfYpShKnnufX6G3ShaZjz5CaR/0xP8sJZJRd1J7D/BpGXH+FT3ZPIbeutgIPRkUQghRyQwuV0UG1RJCCCGqlrsFaa6qqg08HYsnKIqyAH3glrGejkUIIcT1S7poCiGE8AhFUXzRW9K+RZ9q4B+cG9DlhqIoSkPgUfTpIIQQQogrJl00hRBCeIoBvVtjJnoXzT+AcR6NyAMURZmIPsjLu6qqHvR0PEIIIa5v0kVTCCGEEEIIIWoIacETQgghhBBCiBrierwHzxu4HX246UvN4SOEEEIIIYQQNY0JfWTpbejzxha7HhO82zlvCGkhhBBCCCGEuAF1BjaWXHA9JnhpAJmZeWha9bp/sHZtf06dyvV0GKIakLogikhdEEWkLoiSpD6IIlIXRJHLqQtGo4GQED9w50YlXY8JnhNA01zVLsEDqmVMwjOkLogiUhdEEakLoiSpD6KI1AVR5ArqwgW3rMkgK0IIIYQQQghRQ0iCJ4QQQgghhBA1xPXYRVMIIYQQQghRyZxOB5mZGTgcdk+HckNKTzeiaVqpZWazFyEhoZhMFU/bJMETQgghhBBCkJmZgY+PFT+/ehgMBk+Hc8Mxm404HOcSPJfLRV5eNpmZGdSpU7/C+5EumkIIIYQQQggcDjt+foGS3FUTBoMBP7/Ay25RlQRPCCGEEEIIASDJXTVzJX8PSfCEEEIIIYQQooaQBE8IIYQQQghR7WzY8AN9+/6JZ555kiNHDnk6nAvk5OTw2WcLy11vt9sZOXIoDz54Lw8+eO81i0sSPCGEEEIIIUS1s3LlVwwcOJiPP15CdHTDCm/ndF4w9/dlc7lcZGaeZu9elfz8/DLL5ObmsGTJp+Xuw2g00qfPU8ycOfuq47kcMoqmEEIIIYQQolp5771p7NqVxJEjh1m+fBkJCfP46afNzJv3PpqmERwcwqhRo4mMjGLnzu3MmjUVRbmZ5GSVZ599ntat25CQMIP9+/dit9tp06YdQ4e+hMlkIiMjnZkz3+XYsaMAdO16H/36PcO3337NsmWfY7fbKSy089BDj9K69W2YTCamTp3Mzp3bsFi8sFp9mTNnAdOnTyE3N5cBA57Ex8eHuXMXlDoHs9nM7bfHkpaWek2vnSR4QgghhBBCiFI2/ZbGxl1pVbLvO1rVJ67lxYf9HzbsZZKTVfr06UdcXGcyM08zadI4EhI+JCamEYmJK4iPH8v8+XoXyYMHDzBq1GhatGgFwOTJE2ndui2vv/4GmqYRHz+WNWtW0bv3I0yY8AYdO8bx5pvvApCVlQVA27btuOmmmzl9+jQnT2aQkDCdxx9/kr17VZKStrN48TKMRiPZ2dkAjBz5GoMG9eOTT5ZUyXW6UpLgCSGEEEIIIaq1PXt207hxM2JiGgHQo0dvpk2bQn5+HgCRkVHFyR3Axo0b+OOPPXzxxWcA2Gw2wsLqkp+fz+7du5gx44PisoGBgaSnp7N162ZWrvyK3NwcvL19yMrK5PTpU4SHR+JwOJg8eSJt27ajU6fO1/DML58keEIIIYQQQohS4lpeupWtOvH1tZ63xMVbb00lIiKy1NLz76fLzs4mNfUYNpuNBQs+ZMiQEdx7b3c0TaNr1zuw2+3Url2HRYuWkpS0g+3bf2bOnAQWLFhcxWd05WSQFSGEEEIIIUS11rx5S/bvT+bw4UMArFuXSNOmClarX5nl4+K6sHjxwuIBV7KyskhNTcFqtdKiRSuWLPmUAwf2c+DAPrKzs4mJaUxBQUHxYC5r1qzCbtcnGM/MzMRmsxEb25HBg4fg7+9PamoKfn5+2Gw2HA5HlZ//5ZAWPCGEEEIIIUS1FhISwtixE4iPH4PT6SQ4OIRx4yaWW3748JeZPfs9Bgzog8FgwGLxYtiwl6lbtx7PPfcis2e/x+rVK/Dy8ub++3vQvn0Hhg0byejRrxAQEEBsbCeCgoIASE8/wZQpk3A6nTidTjp06ETz5i0xGo107/4A/fs/QUBA4AWDrAAMGvQ0GRknyMnJ4ZFHehAb25HXX3+jyq4TgMHlcl2ykKIoU4HLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,721 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" lambda_coeff = 0.5\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.LSTMCell(size_layer, state_is_tuple = False)\n",
"\n",
" rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer = tf.placeholder(\n",
" tf.float32, (None, num_layers * 2 * size_layer)\n",
" )\n",
" _, last_state = tf.nn.dynamic_rnn(\n",
" drop, self.X, initial_state = self.hidden_layer, dtype = tf.float32\n",
" )\n",
" \n",
" self.z_mean = tf.layers.dense(last_state, size)\n",
" self.z_log_sigma = tf.layers.dense(last_state, size)\n",
" \n",
" epsilon = tf.random_normal(tf.shape(self.z_log_sigma))\n",
" self.z_vector = self.z_mean + tf.exp(self.z_log_sigma)\n",
" \n",
" with tf.variable_scope('decoder', reuse = False):\n",
" rnn_cells_dec = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)], state_is_tuple = False\n",
" )\n",
" drop_dec = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells_dec, output_keep_prob = forget_bias\n",
" )\n",
" x = tf.concat([tf.expand_dims(self.z_vector, axis=0), self.X], axis = 1)\n",
" self.outputs, self.last_state = tf.nn.dynamic_rnn(\n",
" drop_dec, self.X, initial_state = last_state, dtype = tf.float32\n",
" )\n",
" \n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.lambda_coeff = lambda_coeff\n",
" \n",
" self.kl_loss = -0.5 * tf.reduce_sum(1.0 + 2 * self.z_log_sigma - self.z_mean ** 2 - \n",
" tf.exp(2 * self.z_log_sigma), 1)\n",
" self.kl_loss = tf.scalar_mul(self.lambda_coeff, self.kl_loss)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits) + self.kl_loss)\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value = np.zeros((1, num_layers * 2 * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_x = np.random.binomial(1, 0.5, batch_x.shape) * batch_x\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" ) \n",
" init_value = last_state\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value = np.zeros((1, num_layers * 2 * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value = last_state\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0816 15:26:45.502804 139658996016960 deprecation.py:323] From <ipython-input-6-d907d7a4dee6>:13: LSTMCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.LSTMCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0816 15:26:45.505823 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f04dbc873c8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0816 15:26:45.507445 139658996016960 deprecation.py:323] From <ipython-input-6-d907d7a4dee6>:17: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0816 15:26:45.829126 139658996016960 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0816 15:26:45.832581 139658996016960 deprecation.py:323] From <ipython-input-6-d907d7a4dee6>:28: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0816 15:26:46.024316 139658996016960 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0816 15:26:46.031064 139658996016960 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:961: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0816 15:26:46.507349 139658996016960 deprecation.py:323] From <ipython-input-6-d907d7a4dee6>:31: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"W0816 15:26:46.696353 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f04778236d8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0816 15:26:46.879564 139658996016960 deprecation.py:323] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/math_grad.py:1205: add_dispatch_support.<locals>.wrapper (from tensorflow.python.ops.array_ops) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use tf.where in 2.0, which has the same broadcast rule as np.where\n",
"train loop: 100%|██████████| 300/300 [01:47<00:00, 2.80it/s, acc=97, cost=0.00235] \n",
"W0816 15:28:35.363878 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f0455d7deb8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0816 15:28:35.471002 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f044b3bf198>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:46<00:00, 2.82it/s, acc=96.9, cost=0.00305]\n",
"W0816 15:30:22.970038 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f044b349f60>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0816 15:30:23.075726 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f03f03e0e10>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:47<00:00, 2.77it/s, acc=95.1, cost=0.00633]\n",
"W0816 15:32:11.926008 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f03f043bfd0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0816 15:32:12.031505 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f03a4bb2a58>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:47<00:00, 2.86it/s, acc=95.9, cost=0.00422]\n",
"W0816 15:34:00.252120 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f03f0194f60>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0816 15:34:00.478516 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f044c079320>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"train loop: 100%|██████████| 300/300 [01:48<00:00, 2.75it/s, acc=96.3, cost=0.00351]\n",
"W0816 15:35:49.588577 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f03965d9940>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0816 15:35:49.693055 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f0393f722e8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:47<00:00, 2.79it/s, acc=96.2, cost=0.00384]\n",
"W0816 15:37:38.517486 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f0393f489e8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0816 15:37:38.625684 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f039189c940>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:47<00:00, 2.79it/s, acc=95.6, cost=0.00472]\n",
"W0816 15:39:27.256033 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f0391929fd0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0816 15:39:27.363451 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f038f1f4d68>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:47<00:00, 2.78it/s, acc=96.1, cost=0.00394]\n",
"W0816 15:41:15.619689 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f038f286940>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0816 15:41:15.724680 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f038cb1fda0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:47<00:00, 2.82it/s, acc=97.3, cost=0.00223]\n",
"W0816 15:43:04.145420 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f038d74f630>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0816 15:43:04.251741 139658996016960 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f0388c03d68>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:45<00:00, 2.82it/s, acc=96.6, cost=0.00292]\n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdeXxM9/7H8dfMJJF9EWskiO2QhKJaWykatNqry7231xZLS6/aaZUfqoL2osQSimrVVlp6aV26aDW2qrZKq7YTeyKJSiORTMaYzPL7YyYRBAmSSeLzfDw8kjnnzDmfc/JF3vP9nu/R2Gw2hBBCCCGEEEKUfVpnFyCEEEIIIYQQ4v6QgCeEEEIIIYQQ5YQEPCGEEEIIIYQoJyTgCSGEEEIIIUQ5IQFPCCGEEEIIIcoJCXhCCCGEEEIIUU5IwBNCCCGEEEKIcsLF2QUIIYQQ4u4oiqIBJgD/BvyBL4FXVFXNdKxfAfQCTPne5qeqquUO+90OdAJcVVU1O5ZNA54DGgHTVVWdUoQ6KgCLgX8ABmCWqqox93LuQgghCiY9eEIIIe47RVEeqA8QnXi+fYEooC0QBHgAsTdsM0tVVe98f+4U7noDrgWsOgm8AWy9izqmAPWBWkBH4A1FUZ68/akJIYS4Gw/Uf8BCCCFAUZTxwCCgCpAITFRVdZOjl+VP4DFVVQ87tq0MJAC1VFW9qCjKM8B0oDZwFBisquohx7ZnsffS9La/VLyA1ws6lmN7HTAL6AdkAXOwhwJXVVXNiqL4ATFAN8AKfAS8VVBAURTlUWA+9t6lK8B/gTGqqpoc68OBecDDQA4wX1XVdxw1jANedtQYj72XSgec4foerB3AGlVVP1AUpb/jvH7GHm4WK4ryEbAMeAiwAd8AQ1VVzXC8P8RRYzvsH7CuA8YAF4DHVVX9w7FdFeCs45qn3vaHCX8DPlRVNdHx3pnA94qivKqqquEO772J45q/5TinH/OvU1V1pWOb3ndRRz+gv6qq6UC6oijLgP7A10WtUQghxO1JD54QQjx4TmEPGX5ANLBGUZTqqqpeBTYCPfNt+yKw0xHumgHLsQ/DCwSWApsdwTBXT+BpwN8RjAo8lmPbQcBTQFOgOfZgld8KwAzUA5oBXYCBtzgnCzAaqAS0Bp4AhgAoiuIDfIc9TAQ59rfd8b4xjpq7Ab7AS9iHEBZGS+A0UBV4G9AA/3EcoxEQgr3nKjfMbgHOYQ/HNYBPHAH0E6BPvv32BLbnhjtFUTIURXnsNnVobvi+AvbeslxDFEW5pCjKr4qi/P0O5/QO9pB+4Q7bFboORVECgOrA7/nW/w6E38UxhBBC3IH04AkhxANGVdUN+V5+qijK/wGPAl8Aa7EHt4mO9b0crwFeAZaqqvqT4/VKRVEmAK2AnY5lC3J7cQpxrBex96SdB1AUZQb2YIaiKFWxhy5/VVWvANmKoszNraGAc/o138uziqIsBR7H3mv3DHBBVdU5jvVGIPccBgJvqKqqOl7/7ji+z00X7mbJqqrmDkM0Yx/CeNLxOlVRlBjsvWE4zjkIGJvbIwjscXxdCWxQFGW8qqo27EMdZ+U7N//b1PA19uGO64F07L2RAJ6OrwuA14DL2APyp4qiXFBV9Ycbd6QoSgvsQyxHAsF3Ovki1OHt+P5yvu0vA4W5xkIIIYpIAp4QQjxgFEXpi73nqrZjkTf2ni+AOMBTUZSW2IdrNgU2OdbVAvopijI83+7csAeXXIn5vr/TsYJu2D7/97Ww3weWoihK7jLtjfvPd5wG2IdztsAeKlyA3NAXgr0nsSC3W3cnN55rVa4NwfRx1Jue7zjn8oW7PKqq/qQoigHooChKCvYexs2FrGG5Y987sJ/zHOzDJc879n0g37ZfKoryMfACcF3AUxRFC7wHjHQMjy3k4QtVh96xjS/2cJ37fVZRDyKEEOLOJOAJIcQDRFGUWtjvE3sC+FFVVYuiKL/hGF7neL0e+zDBP4Etqqrm/iKeCLytqurbtzmErbDHAlK4vqcoJN/3icBVoFJBoagAi4GDQE9VVbMURRmFfcbG3H31uMX7EoG6wOEblmc7vnoCmY7vq92wje2G1+84ljVWVfWSoijPAQvzHaemoigutzifldiHaV4APlNV1VjANjdRVdWKvZfwLQBFUboASY4/BbFx/VDKXL7Yw/GnjnCncyw/ryjKP1VV3X23daiqanUE14eAbx1veQg4UphzFEIIUTQS8IQQ4sHihf2X/Nz7uwYAETdssxb4HEjj2lBNsIe1TYqifId9chFPoAOwK18ILMqx1gMjFUXZij1Q5Q7rQ1XVFEVRtgFzFEV5E3svUCgQrKrqTm7mgz2I6RVFaQi8mntc7Pe+xThC32LsvY5hjqGmHwDTFEU5in14ZWPsoSRVUZQkoI9juGc/7EHwdnywDz28rChKDWBsvnU/Yw+0MxRFeQv7PYMP5xsquQb78NAs7EM0C0VRlIpAAPZ7ARth78Wc6ghcKIryD+zDJw1AJPYQ+bcCdnWZ63tiQxw1P8y1n58r9uCnBVwURXEHchzB/bZ1AKuASYqi7Md+z+IgYEBhz1MIIUThySQrQgjxAFFV9Sj24XM/Yu+ha8wNw/UcwScb+y/8X+Vbvh/7L+YLsQ89PIl9JsS7PdYyYBtwCHvv25fY72XLnSWzL/YwdtRxvM+wT9ZRkNex3y+Y5djvp/nqyAI6Yw82F4AT2KfqB3sQWe+oIxP4EPsU/zjOdSz2oBsO7L3VuTpEY58s5jL2RwlszFeDxXH8ethnJT0P/Cvf+kTgAPZAfF1vmaIoekVR2t3imJWwX7ds7D+r5aqqvp9v/UjsPWkZwLvAIFVVdzj2W9Ox75qqqtpUVb2Q+4dr4fjP3JlIsV/XK9h7dyc6vs8No3eq4y3sQ2HPYb9f811VVWUGTSGEKAYam+3GESZCCCFEyVMU5SlgiaqqtZxdizMoirIc+8Qtk5xdixBCiLJLhmgKIYRwCkVRPLD3pG3DPmzvLa5N6PJAURSlNvbJT5o5uRQhhBBlnAzRFEII4Swa7MMa07EP0TwGTHZqRU6gKMo07JO8vKuq6hln1yOEEKJskyGaQgghhBBCCFFOSA+eEEIIIYQQQpQTZfEevArAI9inm7bcYVshhBBCCCGEKG902GeW/gX7c2PzlMWA9wg3TCEthBBCCCGEEA+gdsCe/AvKYsBLAUhPz8ZqLV33DwYGepOWpnd2GaIUkLYgcklbELmkLYj8pD2IXNIWRK6itAWtVkNAgBc4slF+ZTHgWQCsVlupC3hAqaxJOIe0BZFL2oLIJW1B5CftQeSStiBy3UVbuOmWNZlkRQghhBBCCCHKCQl4QgghhBBCCFFOlMUhmgWyWMykp6diNpucVsPFi1qsVqvTjl/aaLU6PDy88fb2Q6PROLscIYQQQgghyr1yE/DS01Nxd/fEy6ua08KEi4sWs1kCHoDNZsNiMZOVlUF6eioVK1ZxdklCCCGEEEKUe+VmiKbZbMLLy1d6ikoJjUaDi4sr/v6BmExGZ5cjhBBCCCHEA6HcBDxAwl0ppNFoAZkZSgghhBBCiJJQrgKeEEIIIYQQQjzIJOAVk127dtC79z8YMKAXCQlnnV3OTbKysvj445W3XG8ymRgzZjhPP/0ETz/9RAlWJoQQQgghhLhbEvCKyRdfbOTllwfz0UdrqVmzdqHfZ7Hc9KzCYqHXZ7F27apbrtdqtfTs2Yd5894rkXqEEEIIIYQoTeLitvP0053544/fnV1KkZSbWTRLkwUL5nDo0EESEs6xadMGYmOXsm/fXpYuXYjVasXfP4CxYycQHBzCgQP7mT9/NorSiPh4lUGDXqVp02bExs7l1KkTmEwmmjVrwfDho9HpdKSmXmTevHc5fz4RgMjIrkRFDWDbtq/ZsGEdZnMOAEOHjqJFi0exWq3ExMziwIFfcHV1w9PTg8WLlxMTMxO9Xk///r1wd3dnyZLl152Di4sLjzzSkpSU5BK/fkIIIYQQQjhLdnY20dGTWLHiQxo0UKhSpZqzSyqSchvwfvgjhT2HUopl3481qU7bxtVvuX7EiNeIj1fp2TOKtm3bkZ5+ienTJxMb+z6hoXXYsuVzoqMnsWyZfYjkmTOnGTt2AhERTQCYMWMaTZs2Z/z4N7FarURHT2Lr1s107/48U6e+SevWbXn77XcByMjIAKBly1Z07twVjUZDQsJZRo4cwqZNX3LyZDwHD+5nzZoNaLVaMjMzARgzZhwDB0axYsXaYrlGQgghhBBClDU///wTw4a9wrlzZxk8eBj/939v4uHh4eyyiqTcBrzS5MiRw9St24DQ0DoAdOvWnTlzZmIwZAMQHBySF+4A9uzZxbFjR/jkk48BMBqNVKlSFYPBwOHDh5g7d1Hetv7+/gAkJZ1nypSJpKam4uLiwqVLaaSl/UVQUDBms5kZM6bRvHkL2rRpV1KnLYQQQgghRJlw9epVZs16h0WL5hMcHMKmTVtp0+YxZ5d1V8ptwGvb+Pa9bKWJh4fnDUtsvPPObGrUCL5uqcFguOU+pkyZyLBho2nfvgNWq5XIyMcwmUwEBlZi9er1HDz4K/v3/8zixbEsX76mGM5CCCGEEEKIsufw4T8YOvQVjh07Qp8+/Zg69R28vX2cXdZdk0lWSkB4eGNOnYrn3LmzAHz11Rbq11fw9PQqcPu2bduzZs3KvAlXMjIySE5OwtPTk4iIJqxff21YZe4QTb1eT/XqQQBs3boZk8kEQHp6OkajkZYtWzN48DC8vb1JTk7Cy8sLo9GI2WwurtMWQgghhBCi1DKbzcybN5uuXTuQlvYXH3+8npiY2DId7qAc9+CVJgEBAUyaNJXo6IlYLBb8/QOYPHnaLbcfOfI13ntvAf3790Sj0eDq6saIEa8RFFSDyZOnERMzk6ioF9FqdXTu3JU+ffozYsQYJkx4HR8fH1q2bIOfnx8AFy/+ycyZ07FYLFgsFlqLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,687 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.GRUCell(size_layer)\n",
"\n",
" rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer = tf.placeholder(\n",
" tf.float32, (None, num_layers * size_layer)\n",
" )\n",
" _, last_state = tf.nn.dynamic_rnn(\n",
" drop, self.X, initial_state = self.hidden_layer, dtype = tf.float32\n",
" )\n",
" \n",
" with tf.variable_scope('decoder', reuse = False):\n",
" rnn_cells_dec = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)], state_is_tuple = False\n",
" )\n",
" drop_dec = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells_dec, output_keep_prob = forget_bias\n",
" )\n",
" self.outputs, self.last_state = tf.nn.dynamic_rnn(\n",
" drop_dec, self.X, initial_state = last_state, dtype = tf.float32\n",
" )\n",
" \n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value = np.zeros((1, num_layers * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" ) \n",
" init_value = last_state\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value = np.zeros((1, num_layers * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value = last_state\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0816 17:29:50.622041 140611410274112 deprecation.py:323] From <ipython-input-6-7a2cc302036d>:12: GRUCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.GRUCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0816 17:29:50.624857 140611410274112 deprecation.py:323] From <ipython-input-6-7a2cc302036d>:16: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0816 17:29:50.941864 140611410274112 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0816 17:29:50.945153 140611410274112 deprecation.py:323] From <ipython-input-6-7a2cc302036d>:27: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0816 17:29:51.137009 140611410274112 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0816 17:29:51.144164 140611410274112 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:564: calling Constant.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0816 17:29:51.155159 140611410274112 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:574: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0816 17:29:51.387444 140611410274112 deprecation.py:323] From <ipython-input-6-7a2cc302036d>:41: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [01:36<00:00, 3.15it/s, acc=97.7, cost=0.00125] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:37<00:00, 3.07it/s, acc=98.1, cost=0.000968]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:34<00:00, 3.14it/s, acc=97.3, cost=0.00201] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:36<00:00, 3.12it/s, acc=97.8, cost=0.00113] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:36<00:00, 3.15it/s, acc=97.6, cost=0.00147] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:36<00:00, 3.11it/s, acc=98.2, cost=0.000856]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:35<00:00, 3.14it/s, acc=97.7, cost=0.00139] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:36<00:00, 3.11it/s, acc=97.1, cost=0.002] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:36<00:00, 3.13it/s, acc=97.8, cost=0.00111] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:37<00:00, 3.06it/s, acc=97.4, cost=0.00152] \n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 12,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOy9ebQlyX3X+YklM++9b6t61XurV3V3dksjjDcxHKPBYNnH2AbPgQFjS5YwA3gY4HiGAwyLGcEAZrUZtgGNQbbkxrYEMxbG9hgfbOPBxkeYGZYxUqe61d1SS+ruqnr16lW9d+/NJSLmj4hc7n33Vb2qrq41vnWiIuIXkXlze5nxjd8SwjlHRERERERERERERERExK0PeaMPICIiIiIiIiIiIiIiIuLaIBK8iIiIiIiIiIiIiIiI2wSR4EVERERERERERERERNwmiAQvIiIiIiIiIiIiIiLiNkEkeBEREREREREREREREbcJIsGLiIiIiIiIiIiIiIi4TRAJXkRERERERERERERExG0CfaMPICIiIiIiIuLqkOe5AP408J3ACeCngD9QFMWF0J4Bfx/4b4Ap8NeKovi+S+zrLwDfAawD/x74Q0VR/OfQvh329W7AAf8C+IOD33oZuBcwYZf/piiKr1vxOz8L/GYgKYqieWNXICIiIiJiGVGDFxERERFxzZHn+R01gXgDz/d9wLcDXwU8AIyBvzNo/3PAk8AjwG8C/kSe519/xL5+J/B7gXcB28AvAz80aP+LwEngMeCteDL355b28VuLolgPaRW5ew+QHP/0IiIiIiKuFHfUBzgiIiIiAvI8/5PA7wfuAV4B/kxRFD8WtD2vA7+hKIpfDX3vBj4HPFIUxek8z78JP9B/FPgk8N8VRfGfQt+X8Rqe9/hqvgb8sVW/Ffor4K8B7wcuAt+LJydJURRNnudbwPcB3wBY4AeADxRF0WqIhuf0TuBvAc8AM+D/AP5oURRVaH878L8CXw7UwN8qiuJ7wjH8T8B/G47x08B/DSjgJQZapjzP/xXwbFEU/zDP898Tzuvf4knW38/z/AeA7we+hF7D9YeKojgftn8oHOO78BOsPwL8UeA14DcWRfH/hX73AC+Ha37mkjcTfivwj4qieCVs+1eBn8vz/A8WRTEN1/b3FEWxC+zmef79wO8BfnrFvh4DfrEoihfDvp4F/sel9o8PNHY/Bvy2yxxfh3A/P4C/Xr983O0iIiIiIq4MUYMXERERcefhM3iSsQX8eeDZPM/vL4qiBP5P4FsHfX8X8AuB3H0p8CG8OeAp4IPAjwdi2OJbgW8ETgRitPK3Qt/fD/wW4NcCX4YnVkP8INAATwBfCnwd8PuOOCeDJyN3Ab8e+BrgvwfI83wD+Jd4UvNA2N/Phu3+aDjmbwA28Rqs6RG/sYxfB7yI12T9JUAAfzn8xjPAQwQNVyCSPwF8Fk+OHwR+NBDQHwXeO9jvtwI/25K7PM/P53n+Gy5xHGKpnAFP5nl+Ergf+I+D9v8IvP2I/fwo8NY8z5/K8zzBk8MhEfx7wDfleX4y7Pt3AP/X0j7+cZ7nZ/I8/5k8z79kqe178BMAr13iXCIiIiIi3iCiBi8iIiLiDkNRFP9kUP1onud/Cngn8M+AH8YTtz8T2r8t1AH+APDBoig+EeofzvP8TwP/JfALQfa3W23SMX7rd+E1aZ8HyPP8r+CJGXme34snXSeKopgBB3me/832GFac0/8zqL6c5/kHgd+I19p9E/BaURTfG9rnQHsOvw/4E0VRFKH+H8Pvbxy6cIfxxaIoWnPIBnghJIAzeZ5/H15jRTjnB4A/PvA7+8WQfxj4J3me/8miKBze5PKvDc7txCWO4afxZpcfA3bx2kiACd6PDmBv0H8POOrcXg3HVOAJ8yt4X7kW/y+QAjuh/rPA/zZof0/oI4DvAv5FnudPF0VxPs/zr8CbkX4X8JZLnE9ERERExBtEJHgRERERdxjyPH8fXnP1aBCt4zVfAD8PTPI8/3V4c81fC/xYaHsEeH+e539ksLsUT1xavDIoX+63HljqPyw/gvfVejXP81Yml/c/+J2n8OacX4EnNxpoSd9DeE3iKlyq7XJYPtd76U0wN8Lx7g5+57OrgooURfGJPM+nwFfnef4qXsP448c8hg+Fff8r/Dl/L95s8/PAfuiziSe1bfniEfv6n4GvDPt7Da9V/Lk8z98ezD0/Bvwn4JvxJO5vAM/iiTpFUfzSYF9/Oc/z9wPvyvP8J/FE8LuC6e0xTy0iIiIi4moQTTQjIiIi7iDkef4I3k/sDwOngnboVwlmfsG/7WN4M8FvBX6iKIqWELwC/KWiKE4M0qQoih8Z/IQ77m/hNUZDbc5Dg/IrQAncNfitzaIojjIv/PvAc8CTRVFs4iNLisG+Hj9iu1fwAUOWcRDyyUB231Ift1T/niB7RziG9y4dw8OXCMby4dD/24F/WhTF/Ih+CyiKwhZF8YGiKB4tiuItwH8GvgB8IfjdvYr3CWzxJaHPKvxa4KNFUXy+KIqmKIofxAdVedug/YNFURwURbEP/AO8lvUoOPz5b+KJ90fzPH8N+JXQ/vk8z991nPOMiIiIiDg+ogYvIiIi4s7CGn7g3fp3fQfwXyz1+WHg43hTvD8zkH8/8GN5nv9LfHCRCfDVwP89IIFX8lsfA74raHgO6M0LKYri1TzPfwb43jzP/yxeG/UY8JaiKH6Bw9gALgD7eZ4/DfzB9nfxvm/fl+f5/4AnginwtmBq+g+Bv5Dn+Sfx5pXvwJOjM3mefwF4bzD3fD+rieDyMewBe3mePwj88UHbv8WTrb+S5/kH8CaQXz7Qej2LNw+9iCd5x0JYuuAk3hfwGbwW838pisKGLh8BvjvP83+H9xX8/fhlEFbhV4Dfmef5j+KvXRvx8oVB++/L8/xPhPofwGv0yPP8YTxB/xX85PEfwWtqfylck6GW96FwPb6c/h5FRERERFwjRA1eRERExB2Eoig+iTfj+2W8CeY78IPwYZ9P4AnXAwyCaBRF8e/wBOHv4k0PX8BHZLza3/p+4GfwJOHf49dwa+jXUXsfnox9MvzeP8UHDVmFP4b3F7wY9vvRwXFcBL4Wb7r4GvA8fskA8IToY+E4LgD/CL/UAOFc/zie6L4d+DdHnWvAn8cHi9kDfhIfsKY9BhN+/wl8VNLPA98yaH8F77/mgH893Gme5/uX0HTdhb9uB/h79aGiKP73QfsH8Caon8X7Sf71oih+Ouz34bDvh0Pfv4onmf8BOI8PWvM72iig+AA0j4Zj/wJeK/r+0LaBJ8+7oe3rgd9SFMVOURSuKIrX2kRP6l5vo5xGRERERFw7COeWLUwiIiIiIiKuP/I8/y3APyiK4pEbfSw3AnmefwgfuOW7b/SxRERERETcuogmmhERERERNwR5no/xmrSfwZsPfoA+oMsdhTzPHwV+O345iIiIiIiIiKtGNNGMiIiIiLhREHizxl28iean8JEc7yjkef4X8MFn/npRFC/d6OOJiIiIiLi1EU00IyIiIiIiIiIiIiIibhNEDV5ERERERERERERERMRtglvRBy/DL8T6Kn2ktYiIiIiIiIiIiIiIiDsFCh9Z+lfw68Z2uBUJ3leyFEI6IiIiIiIiIiIiIiLiDsS7gF8cCm5FgvcqwO7uAdbeXP6Dp06ts7Ozf6MPI+ImQHwWIlrEZyGiRXwWIoaIz0NEi/gsRLS4kmdBSsHJk2sQuNEQtyLBMwDWupuO4AE35TFF3BjEZyGiRXwWIlrEZyFiiPg8RLSIz0JEi6t4Fg65rMUgKxEREREREREREREREbcJIsGLiIiIiIiIiIiIiIi4TRAJXkRERERERERERERExG2CSPAiIiIiIiIiIiIiIiJuE0SCFxERERERERERERERcZsgEryIiIiIiIiIiIiIiIjbBJHgRURERERERERERERE3CaIBC8iIiIiIiIiIiIiIuI2wa240HlERERERERERERERMQVwznHwcE+586d49y5nZDOLeS7u7tdvSznfPCDP8A73vFrbvShHxuR4EVERERERERERETcpphOp/zkT/44n//8KyilSZKEJNFonZAkCUqpIEvQOkFr1bUN21vZsN3nGq0X9yeEuC7n5pzj4sUL7OzssLu7TNLOsbNzbqW8qqqV+xNCcPLkSU6e3GZ7+xQPPfQQ99xzH/fcc891OZ9rhUjwIiIiIiIiIiIiIm4zFMVzfOQjH+JjH/tR9vbO35BjEEJ0SUq5UG8THJb5xCX7Wms5f36XpmlW/raUku3t7Y6sPfLIo3zZl315Vz916lRX3t7eZnt7m62tEyilrus1ejNwWYKX5/nfAH4H8CjwjqIofjXInwI+DJwCdoD3FUXx/Btpi4iIiIiIiIiIiIi4OsxmM/75P/84P/RDP8gnPvHLpGnKN33Tb+N97/u9fPmXfyVN02BMQ103NE1NXdc0TUPTNKFch3Lf3vYftrf9h+19f4NzLiSLcwzqRydYrFtrB3UO9RNCcvLkyQWCdvLkdkfctrZOIOWdGW7kOBq8jwN/C/jXS/J/APy9oiiezfP8vcAHgd/8BtsiIiIiIiIiIiIiIq4Azz//aT7ykQ/x0Y/+MOfPn+fxx9/KBz7wF/mWb/k27rrrrq5flmU38CgjrhcuS/CKovhFgDzPO1me5/cAXwZ8bRD9CPB38zy/GxBX01YUxZk3fDYRERERERERERERdwDKsuQnfuKf8ZGP/AC//Mu/RJIkfMM3/Fbe977v4Ku+6l13rPYq4up98B4CvlAUhQEoisLkef7FIBdX2RYJXkRERERERERERMQl8OKLL/CRj/wgH/3oP2ZnZ4dHHnmU7/7uP8fv/t3vveWCgUS8Obhlg6ycOrV+ow9hJe6+e+NGH0LETYL4LES0iM9CRIub6Vlwzi340QxTVVUr5cvJWtvta1V+XNlx+rf+OEO/nFXly7Vfqq/WmpMLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,726 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.GRUCell(size_layer)\n",
"\n",
" backward_rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" forward_rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop_backward = tf.contrib.rnn.DropoutWrapper(\n",
" backward_rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" forward_backward = tf.contrib.rnn.DropoutWrapper(\n",
" forward_rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.backward_hidden_layer = tf.placeholder(\n",
" tf.float32, shape = (None, num_layers * size_layer)\n",
" )\n",
" self.forward_hidden_layer = tf.placeholder(\n",
" tf.float32, shape = (None, num_layers * size_layer)\n",
" )\n",
" _, last_state = tf.nn.bidirectional_dynamic_rnn(\n",
" forward_backward,\n",
" drop_backward,\n",
" self.X,\n",
" initial_state_fw = self.forward_hidden_layer,\n",
" initial_state_bw = self.backward_hidden_layer,\n",
" dtype = tf.float32,\n",
" )\n",
" \n",
" with tf.variable_scope('decoder', reuse = False):\n",
" backward_rnn_cells_decoder = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" forward_rnn_cells_decoder = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" drop_backward_decoder = tf.contrib.rnn.DropoutWrapper(\n",
" backward_rnn_cells_decoder, output_keep_prob = forget_bias\n",
" )\n",
" forward_backward_decoder = tf.contrib.rnn.DropoutWrapper(\n",
" forward_rnn_cells_decoder, output_keep_prob = forget_bias\n",
" )\n",
" self.outputs, self.last_state = tf.nn.bidirectional_dynamic_rnn(\n",
" forward_backward_decoder, drop_backward_decoder, self.X, \n",
" initial_state_fw = last_state[0],\n",
" initial_state_bw = last_state[1],\n",
" dtype = tf.float32\n",
" )\n",
" self.outputs = tf.concat(self.outputs, 2)\n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value_forward = np.zeros((1, num_layers * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" ) \n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value_forward = np.zeros((1, num_layers * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0816 18:33:46.362064 140384958228288 deprecation.py:323] From <ipython-input-6-2500790da2db>:12: GRUCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.GRUCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0816 18:33:46.364130 140384958228288 deprecation.py:323] From <ipython-input-6-2500790da2db>:16: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0816 18:33:46.687459 140384958228288 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0816 18:33:46.692470 140384958228288 deprecation.py:323] From <ipython-input-6-2500790da2db>:42: bidirectional_dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.Bidirectional(keras.layers.RNN(cell))`, which is equivalent to this API\n",
"W0816 18:33:46.693083 140384958228288 deprecation.py:323] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn.py:464: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0816 18:33:46.884588 140384958228288 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0816 18:33:46.891244 140384958228288 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:564: calling Constant.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0816 18:33:46.900250 140384958228288 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:574: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0816 18:33:47.374557 140384958228288 deprecation.py:323] From <ipython-input-6-2500790da2db>:67: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [02:28<00:00, 2.02it/s, acc=97.7, cost=0.00125] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:26<00:00, 2.05it/s, acc=98.3, cost=0.000708]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:29<00:00, 2.01it/s, acc=98.1, cost=0.000848]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:27<00:00, 2.03it/s, acc=98.5, cost=0.000662]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:30<00:00, 2.01it/s, acc=97.4, cost=0.0017] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:29<00:00, 2.01it/s, acc=97.7, cost=0.00127] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:30<00:00, 1.99it/s, acc=98.3, cost=0.000625]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:29<00:00, 2.01it/s, acc=98.2, cost=0.000883]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:29<00:00, 2.01it/s, acc=98.5, cost=0.000547]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:29<00:00, 2.00it/s, acc=96.9, cost=0.00229] \n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA4IAAAFBCAYAAAAi6hFSAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdeZwU1b3//1dVd88+MDiyjwgqFIomiEQ2JYtboonZjDeoRLzqvSZxiSZe/akx4pIrLrgQt3gvakQ0aNyCS7xJvkZxR4jErRCUfRuBgRnGmZ6uqt8fVdVdPdPDOjM9zLyfj0dPV51zqvrT1ad76lOnutrwPA8RERERERHpPsx8ByAiIiIiIiIdS4mgiIiIiIhIN6NEUEREREREpJtRIigiIiIiItLNKBEUERERERHpZpQIioiIiIiIdDNKBEVERERERLqZeL4DEBERkfZlWVZv4A7gJMAFnrdt+/Sg7gNg/0jzIuAF27a/k2M9BnAF8J9ABfA88B+2bW8N6gcCdwNHA/XA9bZt3xtZ/vfAV4GhwL/btv1gpG4K8L/AF5GH/LZt2y/vwVMXEZFWaERQRETywrKsbnUwMs/P90lgHTAI6APcElbYtj3Ctu0y27bLgHJgJfB4K+v5CTAZmAAMAIqBGZH6WcBnQF/8pPO3lmV9PVL/HvAzYEEr638jjCW4vbxLz1JERHZat/onLCIiO8eyrMuBc/GThpXAlbZtP2VZViGwHjjKtu33g7a9gRXA/rZtb7As69vA9cBg4EPgPNu2FwVtlwH3AKf7s1Yp8KtcjxW0jwE3AWcCtcCt+IlHwrbtlGVZPYHpwIn4I10PAL+xbdvJ8ZyOxB8VOxh/1OlPwCW2bSeD+hHA7cARQBNwh23bvw1iuAw4O4hxMfA9IIaf9CRs204F63gZmGXb9v8EI1znAm/jJ1D3WJb1AHA/8GXAA/4C/Ny27Zpg+f2CGI/GP1j7KHAJfhL3Vdu2/xW06wMsC7Z59Q5ey+OB/YCvRbbLwlaaTwT2DbZNLt8B/te27ZXBuqcBf7cs66dBvF8DTrVtuwl4z7KsJ4B/B/4fgG3bdwXLNWwvZhERaX8aERQRkVyW4icjPYGpwCzLsvrbtt2IP7o0KdL2VOAfQRJ4ODAT/9TBSuA+4NkggQxNwh8tqggSqJyPFbQ9F/gWMBIYhZ+ART0IpICDgMOB44FzWnlODnAxfqIzDjgGf3QKy7LKgb8CL+KPdB0E/C1Y7pIg5hOBHviJTX0rj9HcGOBT/BGyGwAD+O/gMQ7GT9CuCWKIAXOB5fhJ9EDgsSBRfQw4I7LeScDfwiTQsqway7KOaiWGsYANPGRZ1kbLst6xLOurrbQ9E/iTbdvbtvOcjGbThfinehqt1B+6nXU1d7hlWZ9blrXYsqxfd7dRYxGRjqQPWBERacG27eipgX+0LOv/A44EngFm4yd4Vwb1pwXzAP8B3Gfb9lvB/EOWZV2Bn4z8Iyi7MxxR2onHOhV/ZG4VgGVZN+IncFiW1Rc/OauwbfsLYJtlWbeFMeR4Tu9GZpdZlnUf/vfVbge+DayzbfvWoL4BCJ/DOcB/2bZtB/PvBY9f3mLDtbTGtu3w1MkUsCS4AVRbljUd+E0wfyR+gnhpOMIIzAvuHwIetyzrctu2PfzTM2+KPLeK7cRQRSZBPgv4IfCMZVkH2bb9edjIsqwS4BTg5O2s60XgvyzLmgNsxh8pBSixbbvWsqzXgF9blnUpcEjwWNsdsYx4BT9pXA6MAP6Iv83+eyeXFxGRXaBEUEREWrAs6yf4I2GDg6Iy/JE08E/zK7Esawz+aaIjgaeCuv2BMy3LuiCyugL8BCe0MjK9o8ca0Kx9dHp/IAGstSwrLDObrz/yOMPwTyMdDZTg/w8Mk8P98Ecmc9le3Y40f659yZz6WR7EuznyOMsjSWCabdtvWZZVD3zNsqy1+COWz+5kDF8Ay2zb/t9g/jHLsq7E/57fM5F2PwA2kUnYc5kZxPky/va7Ff900VVB/enAXfjP+1P87wyO2Jkgbdv+NDL7L8uyrgUuRYmgiEi7UCIoIiJZLMvaH/97bMfgX7zDsSzrnwSn/AXzc/BPT1wPzLVtuzZYfCVwg23bN2znIbydfSxgLf6IVmi/yPRKoBHYN1fylMM9+N+NmxSMXv0CfwQsXNePW1luJXAg8H6z8vD0yRJgazDdr1kbr9n8b4Oyw2zb3mRZ1veA30UeZ5BlWfFWns9D+KeHrgOesG17Z79ntwg/WdteXOCfFvqHYMQxJ9u2XfwRzN9A+vuHq4Mbtm0vxx9dJaifjf8dyd3hkX2aqYiItCElgiIi0lwp/k54+P2zs2j5Pa/ZwNPARjKniIKf1D1lWdZf8ROAEvwLiLwSSRZ35bHmABdZlvUcfuIVnoqIbdtrLct6CbjVsqxfA3XAEKDKtu1co1rl+AlbnWVZw4GfkjltcS4wPUgO78EfxTwkOMX1f4DrLMv6EP+0zsOA1bZtV1uWtRo4IzjN9Ez8hHF7yoEtwJbgpxYujdS9jZ/43mhZ1m/wv9N4hG3brwX1s/BPS63FPzV0Zz0F3GJZ1pnBOr6Pn1yH68WyrCrg68B521uRZVn7AL3wR/sOxh9hvTZIELEs62D80cFG/NN6jw/ahcsX4I+CGkDCsqwiIGnbtmtZ1reABbZtrw9en1/T+tVLRURkD+liMSIiksW27Q/xT/l7A3/E7zAiSUPQ5i38xGwA8EKkfD7+BV5+h3/K4xJgyh481v3AS/ijWgvxf7cuhZ8kgX81zgL8q5NuBp4A+pPbr/C/z1gbrPePkThqgePwR87WAZ/gJ0bgJztzgji24v/WXXFQdy5+MrcR/xTI11t7roGp+Be92QI8h3/hnTAGJ3j8g/CvwroK+LdI/Ur8n13wgFejK7Usq86yrKNzPaBt25vwv/f3q+BxLwe+G/1+IH5i+YZt2y1OgW227n3xX4Nt+K/7TNu2fx9pfgJ+krgZP6n8ZrOrmr6Ef6rqeOD3wfTEoO4YYJFlWduCx3gSfwRVRETageF5rZ4BIiIi0qkEo0b32ra9/w4bd0GWZc3EvwDNVfmORURE9m46NVRERDoty7KK8UfmXsL/CYbfkLkwTbdiWdZg/Au6HJ7nUEREpAvQqaEiItKZGfinU27GPzX0I+DqvEaUB5ZlXYd/sZqbbdv+LN/xiIjI3k+nhoqIiIiIiHQzXXVEMI7/e1Q69VVERERERLqj7eZEXTVR2h//SnVHk/mRWxERERERke6iCv8q0wcBLa4K3VUTwfDS4a9ut5WIiIiIiEjX1p9ulAiuBdi8eRuu27m+A1lZWcbGjXX5DkM6AfUFCakvSJT6g4TUFySkviChXekLpmnQq1cpBLlRc101EXQAXNfrdIkg0CljkvxQX5CQ+oJEqT9ISH1BQuoLEtqNvuDkKuyqF4sRERERERGRVigRFBERERER6WaUCIqIiIiIiHQzSgRFRERERES6GSWCIiIiIiIi3YwSQRERERERkW5GiaCIiIiIiEg302a/I2hZ1i3AD4HBwGG2bb8flA8DHgIqgY3AT2zb/mRP6kRERERERGT3teUPyj8N3AG82qz8XuAu27ZnWZZ1BnAf8I09rBMRERERkZ3geR51dbVs3ryZLVtq2Lx5MzU1m1vM19dvA8AwDMDAMIxgmvR09JarHJqX02r7XFor31Fd+LjNua6D4zg4jovjpIJph1QqFdS56elUKpWzrT/tRqbDdTi4roPrulx22ZVMmnTGduLrfNosEbRtex6AZVnpMsuy+gCjgOOCokeB31mW1Rv/1drlOtu2q9sqZhGRrsjzPFzXjfxDy/xjAwj/j27vH3Puf+bbn29et/1/2O3L8zxSqRTJZJKmpiSNjf59MplMl4XTmfkmksnGYL6JxsbGdHkq1ZRed67nHd1WO3ffel1paSF1dQ14nofnecHzodm8lzUPrde1tlzzZaOPsb1l/PLt14frC+t35/XbddnL5FpH87IdzbdWZpomsVgs62aa/n08HicWM9Pz/i0e3Jvptn67cFkzvWx0uZ49i6mpqcd13fT7Grz0fFgWvW9enrtN9nwsFqOgoJDCwvBWRFFRYYuygoICioqKssrj8Xhe3+vdTSqVoqamJkjiNm03qQvvwzLHcVpdb2FhIb167UNJSQmQ63MEMp8VO/p8iH7utNY+dxzbe+/vTl3Yv/33XuY957/XWr7v4vHs92wikaCoqCjHe9lvG33fDx06rNX4Oqu2HBHMZT9gtW3bDoBt245lWWuCcmM363Y6EaysLGvTJ9NWevcuz3cIrXI9DzfYGXAi067n4abr/XI38kY2w6M9ZO5zlRkGGBh+Hdn10bId8TwPr1m8rgcuzeYj7cIyj0xdum2LdXk4zbaDG0x7QV3zbeE2a+e6wXpdDxcPx/Uf23GDx1m3CYPMTmGwF4bjpPAcF9dN4aYcPDcz7boOruPgOf696zq4qRSu5+IFO/2Zcv/ec11ipkksZhIPP7xisUhZnHjMJGaaxOMx4jF/JyZu+u0TwU5NPBYjHrRPBB+qiZhJPB7Ox/zkw3FJOQ6u59+nHBfX9Ug5Do7r0uQ4WfPhMk5wVM5xPRzX8dfhejhucATOzdSBgWGafv8xYximCYaBaZpgmBimiRkzMTDBDMpNEyOoM0wDAxMjFvPLATMWwzOCNjEDwzD9naVUE25TE56Twks24aaSOKkm3KYkblMTTlMTTlPSvyWTpJqSpIKyVLKRpmSSVJBoNDX6803JZDrhSDb695mkLXNEcnfL/J3FziF8ncxw+xsGRvCamOFraGSmo+Vm0Eej5WZ6OoZpGniel96GyWSSxsj07iUT3cv2jvTvaCRgV+t3N7Y9WSbXOnbUZkfLhAlU8/dh9JZKpbpN/zNNk8LCQoqKinZ4n0gkdphItHXZ7t6aJ9atJTzNba8/7Wrfi87X19ezadMmamtrt/t69OzZk33Line truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,704 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" lambda_coeff = 0.5\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.GRUCell(size_layer)\n",
"\n",
" rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer = tf.placeholder(\n",
" tf.float32, (None, num_layers * size_layer)\n",
" )\n",
" _, last_state = tf.nn.dynamic_rnn(\n",
" drop, self.X, initial_state = self.hidden_layer, dtype = tf.float32\n",
" )\n",
" \n",
" self.z_mean = tf.layers.dense(last_state, size)\n",
" self.z_log_sigma = tf.layers.dense(last_state, size)\n",
" \n",
" epsilon = tf.random_normal(tf.shape(self.z_log_sigma))\n",
" self.z_vector = self.z_mean + tf.exp(self.z_log_sigma)\n",
" \n",
" with tf.variable_scope('decoder', reuse = False):\n",
" rnn_cells_dec = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)], state_is_tuple = False\n",
" )\n",
" drop_dec = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells_dec, output_keep_prob = forget_bias\n",
" )\n",
" x = tf.concat([tf.expand_dims(self.z_vector, axis=0), self.X], axis = 1)\n",
" self.outputs, self.last_state = tf.nn.dynamic_rnn(\n",
" drop_dec, self.X, initial_state = last_state, dtype = tf.float32\n",
" )\n",
" \n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.lambda_coeff = lambda_coeff\n",
" \n",
" self.kl_loss = -0.5 * tf.reduce_sum(1.0 + 2 * self.z_log_sigma - self.z_mean ** 2 - \n",
" tf.exp(2 * self.z_log_sigma), 1)\n",
" self.kl_loss = tf.scalar_mul(self.lambda_coeff, self.kl_loss)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits) + self.kl_loss)\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value = np.zeros((1, num_layers * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_x = np.random.binomial(1, 0.5, batch_x.shape) * batch_x\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" ) \n",
" init_value = last_state\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value = np.zeros((1, num_layers * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value = last_state\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0816 23:54:04.861056 140552998012736 deprecation.py:323] From <ipython-input-6-f18f06dc1a5f>:13: GRUCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.GRUCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0816 23:54:04.862557 140552998012736 deprecation.py:323] From <ipython-input-6-f18f06dc1a5f>:17: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0816 23:54:05.179484 140552998012736 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0816 23:54:05.182720 140552998012736 deprecation.py:323] From <ipython-input-6-f18f06dc1a5f>:28: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0816 23:54:05.374030 140552998012736 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0816 23:54:05.380675 140552998012736 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:564: calling Constant.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0816 23:54:05.389776 140552998012736 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:574: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0816 23:54:05.536239 140552998012736 deprecation.py:323] From <ipython-input-6-f18f06dc1a5f>:31: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"W0816 23:54:05.986564 140552998012736 deprecation.py:323] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/math_grad.py:1205: add_dispatch_support.<locals>.wrapper (from tensorflow.python.ops.array_ops) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use tf.where in 2.0, which has the same broadcast rule as np.where\n",
"train loop: 100%|██████████| 300/300 [01:48<00:00, 2.73it/s, acc=96, cost=0.00448] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:49<00:00, 2.74it/s, acc=95.6, cost=0.00512]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:48<00:00, 2.76it/s, acc=96.2, cost=0.0037] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:48<00:00, 2.75it/s, acc=95.5, cost=0.00715]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:48<00:00, 2.78it/s, acc=96.6, cost=0.0041] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:48<00:00, 2.75it/s, acc=97.3, cost=0.00204]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:47<00:00, 2.81it/s, acc=62, cost=7.74] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:48<00:00, 2.80it/s, acc=95, cost=0.00699] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:48<00:00, 2.76it/s, acc=96.8, cost=0.00279]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:48<00:00, 2.75it/s, acc=97.1, cost=0.00215]\n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdd3gVZfr/8fecknLSE0oSAkkIMCAd6dgFrOt+XSsqgqK/dS2oqKtrWxF1xYIorui6i6AICiqIoIi6a0FUQAIIwoQEkkACpJeT5OSUmd8fcwIBEgiQCvfrurxMZubMPDMZkvM5zz3PoxiGgRBCCCGEEEKIts/S0g0QQgghhBBCCNE4JOAJIYQQQgghxClCAp4QQgghhBBCnCIk4AkhhBBCCCHEKUICnhBCCCGEEEKcIiTgCSGEEEIIIcQpQgKeEEIIIYQQQpwibC3dACGEEEKcOFVV7wGmADFAGnCfpmmr/esigVeBS/ybv6Fp2lP17CcAWAAMBhKB8zVN+7bW+vuBe4B2gBP4EHhI0zSvqqod/Mc5FwgBtgBTNE37xf/aOOAt/77jgGRN0zIb5woIIYSoTXrwhBBCNDpVVU+rDxBb6nxVVR0GPA9cDUQA/wGWqKpq9W/yCuAAkoChwHhVVW85yi5XAzcB++pYtwwYpGlaONAH6A9M9q8LBdYBZwLRwDxghaqqof71OrASuOr4z1IIIcTxOK3+AAshhABVVR8Bbgc6ALuBxzRNW6KqaiCwHzhL07Qt/m3bA9lAoqZpeaqqXg48gxkYfgfu0DRts3/bTGA2cKP5rRoCPFjXsfzbW4EXgAlAOfAyMAuw+3uFIoAZwKWYAeEd4O+apvnqOKehmD1IvYAq4GPMHiS3f31vYCZmAPEAr2qa9py/DQ8Dk/xtTAP+D7ACu2ra4t/Ht8B8TdP+rarqRP95rQVuBmarqvoO8DZm8DGAL4G7NE0r8b++s7+NZ2N+wLoQs+dtH3Cupmm/+bfrAGT6r3n+UX+Y5s9hq6Zpv/pf+y7whv9c9gJ/AC7RNK0SyFRV9T/Arf5reQj/tZrp388R11jTtIxa3yqYP5Nu/nU7MX9WNf6lqupLgAr8qmnafuCN0y34CyFES5AePCGEOP1kYIaMCGAqMF9V1ThN06qBT4Bxtba9FvjOH+4GAnOAP2OWA74FLPMHwxrjgMuASH8wqvNY/m1vxywdHAAMwgxWtc0FvJghYiAwFritnnPyAfdjlg+OAC4E7gRQVTUM+BqzBynev79v/K+b4m/zpUA4ZviprOcYhxsG7AQ6As9ihp5/+I/RC+gMPOVvgxVYDmRhhrJOwAf+UPUBZq9ZjXHANzXhTlXVElVVz6qnDV8AVlVVh/mPcSuwkUN74JTDvu7TwPM7gqqqN6iqWgYUYAbZt+rZbgAQAKSf6LGEEEKcGPkkTQghTjOapi2u9e2Hqqr+DbN871PMZ7DeAh7zr7+Bg2/i/x/wVs1zVcA8VVUfBYYD3/mXvaZp2u4GHutazJ60PQCqqj6PGcxQVbUjZuiK1DStCqhQVfWVmjbUcU6/1vo2U1XVtzCfB5sJXA7s0zTtZf96F1BzDrcBf9U0TfN/v8l//LAjLtyRcjVNm+X/2osZZmoCTb6qqjOAv/u/H4oZ/B6q6RHELIcEs5xxsaqqj2iaZgDjMXs2a84t8ihtKMfsrVyNGd5KMHvsDP/6lcAjqqpOwAyit2KWbJ4QTdMWAAtUVe2O2XO5//BtVFUNB94DpmqaVnqixxJCCHFiJOAJIcRpRlXVmzF7rpL8i0Ixe74A/gc4/M927cfsXVviX5cITPAP6lEjADO41Nhd6+tjHSv+sO1rf50I2IG9qqrWLLMcvv9ax+mBWSI4GDPA2ICa0NcZsyexLkdbdyyHn2tHDpZghvnbW1zrOFm1wt0Bmqb9oqpqJXCeqqp7MXsYlzWwDZOAW4DemOFyLLBcVdWBmqblYj4jNwvYARRiloWOq2dfDaZp2g5VVbdiloP+qWa5qqrBwGfAz5qm/eNkjyOEEOL4ScATQojTiKqqiZjPiV0I/KRpmk9V1Y34y/j83y/CDAH7geWappX7X74beFbTtGePcoianqNjHgvzGbGEWq/tXOvr3UA10K6uUFSH2UAqME7TtHJVVe/DHHikZl/X1/O63UAK5qiPtVX4/+8Ayvxfxx62jXHY98/5l/XVNK1IVdX/A16vdZwuqqra6jmfeRwc3OQjTdNc9bT3cAMwf0Zp/u9X+kPiSP9+ijCfiQRAVdXnMJ8bbAw2zGtXs+9AYCmwB7OMVwghRAuQgCeEEKeXEMwQUvN81y0c+UzWAsw36oUcLNUEM6wtUVX1a8yQ4ADOA76vFQKP51iLgHtVVV2BGagerlmhadpeVVVXAS+rqvoE5rD8yUCCpmnfcaQwzCDmVFW1J/CXmuNiPvs2wx/6ZmP2Op7hLzX9NzBNVdXfMXvA+gI5mqblq6qaA9zkL/ecQK0wU48woBQoVVW1E/BQrXVrMQPt86qq/h3zmcEzNU370b9+PmZ5aDlmiWZDrQMeU1V1FuagMKOBHvgDq6qqKZhlmyWYvXv/D7N0tU7+kFYTwANUVQ0CqjVNM1RVvQ1Y5n8e8wzgb5gDyaCqqh34CHOAmwmapul17DsIc/AagEBVVYOOI8gKIYRoIBlkRQghTiOapv2OOVrlT5g9dH2BHw/b5hfMwBWPOYhHzfL1mAOjvI5ZepgOTDyJY70NrAI2Y/a+fY75LFvNCI43Y4ax3/3H+whzDrW6PIj5vGC5f78f1mpHOTAGc0TJfZjliuf7V8/ADJqrMAPif4Bg/7rbMUNaIWYJ5Jr6ztVvKuZgMaXACswBa2ra4PMfvxvmqKR7gOtqrd8NbMAMxD/U3qmqqk5VVc+u55jvYg7S8q2//a8Bf9Y0bbt//ZnAb5jX5R/AjZqmba21762qqt5Ya38aZkjrhBneqjDLZQFGAb+pqlqB+bP6HHjUv24k5rOOY4ESf5sPb3cVZlAH2O7/XgghRCNTDOPwChMhhBCi+amqegnwpqZpicfc+BSkquoczIFbHm/ptgghhGi7pERTCCFEi/APyHE+Zu9ZR8wRJ5cc9UWnKFVVkzAHKxnYwk0RQgjRxkmJphBCiJaiYJY1FmOWaG4DnmzRFrUAVVWnYT4z96Kmabtauj1CCCHaNinRFEIIIYQQQohThPTgCSGEEEIIIcQpoi0+gxcIDMEcbtp3jG2FEEIIIYQQ4lRjxRxZeh3mvLEHtMWAN4TDhpAWQgghhBBCiNPQ2cDq2gvaYsDbC1BcXIGut67nB2NiQiksdB57Q3HKk3tB1JB7QdSQe0HUJveDqCH3gqhxPPeCxaIQFRUC/mxUW1sMeD4AXTdaXcADWmWbRMuQe0HUkHtB1JB7QdQm94OoIfeCqHEC98IRj6zJICtCCCGEEEIIcYqQgCeEEEIIIYQQp4i2WKIphBBCCCGEaGQ+n5fi4ny8XndLN+W0lJdnQdf1Q5bZbAFERbXHam14bJOAJ4QQQgghhKC4OJ+gIAchIbEoitLSzTnt2GwWvN6DAc8wDCoqyiguzqddu7gG70dKNIUQQgghhBB4vW5CQsIl3LUSiqIQEhJ+3D2qEvCEEEIIIYQQABLuWpkT+XlIwBNCCCGEEEKIU4QEPCGEEEIIIUSr8/3333LjjVdzyy03kJ2d2dLNOUJ5eTnvvz+v3vVut5spU+7hsssu5LLLLmy2dskgK0IIIYQQQohW59NPP2HSpDu44ILRhyyvdHkoLq/GYlGwWS3YrDX/N/9T0LHZmj7mOJ3lLFjwLjfeOKHO9RaLhXHjbiIyMpL77ruzydtTQwKeEEIIIYQQolV57bWX2bw5lezsLJYsWcysWW/x889rmD17Fm6Pj4iISP581wO069CJ3zZv4L05r5PUtTtZu9K5ZtytnNFnAO/PfYPsrJ14PW76DxjEnXfeR2CAnaKifF599SX27NkNwOjRFzF+/C2sWrWSxYsX4vV6ALjrrvsYPHgouq4zY8YLbNiwDrs9AIcjmNmz5zBjxnScTicTJ95AUFAQb74555BzsNlsDBkyjL17c5v12knAE0IIIYQQQhzix9/2snrz3ibZ91n94hjV9+jD/k+e/ABpaRrjxo1n1KizKS4uYtq0J3n0qRkkJ6ewfs1XvPHqc7z99jzyox3k7MnkwQf/Ro+evfH6DF595TnO6NOf2+98CI/HyxuvPcuijz/m/NGX8+xTjzHozOHcM2UqNptChbMMZ5WH/gOHcP4FY7BZLezencW9997JkiWfk56eRmrqeubPX4zFYqGsrAyAKVMe5rbbxjN37oImuU4nSgKeEEIIIYQQolVL3biJhC5dSU5OoWN0MJdffgWvvDKdysoKABISOnPmoIEHtl+/dg27MjRWLv8IMHC5XHTuFI8jQGeHtpWnn3sFwzCodPkwLMEUlFSRsSOdxR/MobioAJvNRmFhIem7cujYMR6v18vzz09j0KDBjBx5dgtdhYaRgCeEEEIIIYQ4xKi+x+5lay5V1V5KK9woikLH6GCsFgu6rh+yTXCw47BXGTz33Et06pRwyNLKykoUBdpHBB94Tk/XDbw+nYcmP8ek/3c3Q4edjdvjZdzVYymrqKRzQizvvbeI1NRfWb9+LbNnz2LOnPlNeconRUbRFEIIIYQQQrRKHo+PvOIqevbsze6sDPbszgbgiy+W0727isMRUufrRo06h/nz5+Hz+QAoKSkhNzcHh8NBnz79WLToYFllWVkpAXYrFRVOUpISiQ4PYt2ar/F43MTFhFDpLMPlcjFs2AjuuONuQkNDyc3NISQkBJfLhdfrbfoLcRykB08IIYQQQgjR6uiGQbHTTXerQo/keJ544mmmTn0Mn89HZGQUTz45rd7X3nvvA7zxxmtMnDgORVGw2wOYPPkB4uM78eST05gxYzrjx1+LxWJlzJiLuOmmiUyePIVHH32QsLALine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,717 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"def layer_norm(inputs, epsilon=1e-8):\n",
" mean, variance = tf.nn.moments(inputs, [-1], keep_dims=True)\n",
" normalized = (inputs - mean) / (tf.sqrt(variance + epsilon))\n",
"\n",
" params_shape = inputs.get_shape()[-1:]\n",
" gamma = tf.get_variable('gamma', params_shape, tf.float32, tf.ones_initializer())\n",
" beta = tf.get_variable('beta', params_shape, tf.float32, tf.zeros_initializer())\n",
" \n",
" outputs = gamma * normalized + beta\n",
" return outputs\n",
"\n",
"def multihead_attn(queries, keys, q_masks, k_masks, future_binding, num_units, num_heads):\n",
" \n",
" T_q = tf.shape(queries)[1] \n",
" T_k = tf.shape(keys)[1] \n",
"\n",
" Q = tf.layers.dense(queries, num_units, name='Q') \n",
" K_V = tf.layers.dense(keys, 2*num_units, name='K_V') \n",
" K, V = tf.split(K_V, 2, -1) \n",
"\n",
" Q_ = tf.concat(tf.split(Q, num_heads, axis=2), axis=0) \n",
" K_ = tf.concat(tf.split(K, num_heads, axis=2), axis=0) \n",
" V_ = tf.concat(tf.split(V, num_heads, axis=2), axis=0) \n",
"\n",
" align = tf.matmul(Q_, tf.transpose(K_, [0,2,1])) \n",
" align = align / np.sqrt(K_.get_shape().as_list()[-1]) \n",
"\n",
" paddings = tf.fill(tf.shape(align), float('-inf')) \n",
"\n",
" key_masks = k_masks \n",
" key_masks = tf.tile(key_masks, [num_heads, 1]) \n",
" key_masks = tf.tile(tf.expand_dims(key_masks, 1), [1, T_q, 1]) \n",
" align = tf.where(tf.equal(key_masks, 0), paddings, align) \n",
"\n",
" if future_binding:\n",
" lower_tri = tf.ones([T_q, T_k]) \n",
" lower_tri = tf.linalg.LinearOperatorLowerTriangular(lower_tri).to_dense() \n",
" masks = tf.tile(tf.expand_dims(lower_tri,0), [tf.shape(align)[0], 1, 1]) \n",
" align = tf.where(tf.equal(masks, 0), paddings, align) \n",
" \n",
" align = tf.nn.softmax(align) \n",
" query_masks = tf.to_float(q_masks) \n",
" query_masks = tf.tile(query_masks, [num_heads, 1]) \n",
" query_masks = tf.tile(tf.expand_dims(query_masks, -1), [1, 1, T_k]) \n",
" align *= query_masks\n",
" \n",
" outputs = tf.matmul(align, V_) \n",
" outputs = tf.concat(tf.split(outputs, num_heads, axis=0), axis=2) \n",
" outputs += queries \n",
" outputs = layer_norm(outputs) \n",
" return outputs\n",
"\n",
"\n",
"def pointwise_feedforward(inputs, hidden_units, activation=None):\n",
" outputs = tf.layers.dense(inputs, 4*hidden_units, activation=activation)\n",
" outputs = tf.layers.dense(outputs, hidden_units, activation=None)\n",
" outputs += inputs\n",
" outputs = layer_norm(outputs)\n",
" return outputs\n",
"\n",
"\n",
"def learned_position_encoding(inputs, mask, embed_dim):\n",
" T = tf.shape(inputs)[1]\n",
" outputs = tf.range(tf.shape(inputs)[1]) # (T_q)\n",
" outputs = tf.expand_dims(outputs, 0) # (1, T_q)\n",
" outputs = tf.tile(outputs, [tf.shape(inputs)[0], 1]) # (N, T_q)\n",
" outputs = embed_seq(outputs, T, embed_dim, zero_pad=False, scale=False)\n",
" return tf.expand_dims(tf.to_float(mask), -1) * outputs\n",
"\n",
"\n",
"def sinusoidal_position_encoding(inputs, mask, repr_dim):\n",
" T = tf.shape(inputs)[1]\n",
" pos = tf.reshape(tf.range(0.0, tf.to_float(T), dtype=tf.float32), [-1, 1])\n",
" i = np.arange(0, repr_dim, 2, np.float32)\n",
" denom = np.reshape(np.power(10000.0, i / repr_dim), [1, -1])\n",
" enc = tf.expand_dims(tf.concat([tf.sin(pos / denom), tf.cos(pos / denom)], 1), 0)\n",
" return tf.tile(enc, [tf.shape(inputs)[0], 1, 1]) * tf.expand_dims(tf.to_float(mask), -1)\n",
"\n",
"def label_smoothing(inputs, epsilon=0.1):\n",
" C = inputs.get_shape().as_list()[-1]\n",
" return ((1 - epsilon) * inputs) + (epsilon / C)\n",
"\n",
"class Attention:\n",
" def __init__(self, size_layer, embedded_size, learning_rate, size, output_size,\n",
" num_blocks = 2,\n",
" num_heads = 8,\n",
" min_freq = 50):\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" \n",
" encoder_embedded = tf.layers.dense(self.X, embedded_size)\n",
" encoder_embedded = tf.nn.dropout(encoder_embedded, keep_prob = 0.8)\n",
" x_mean = tf.reduce_mean(self.X, axis = 2)\n",
" en_masks = tf.sign(x_mean)\n",
" encoder_embedded += sinusoidal_position_encoding(self.X, en_masks, embedded_size)\n",
" \n",
" for i in range(num_blocks):\n",
" with tf.variable_scope('encoder_self_attn_%d'%i,reuse=tf.AUTO_REUSE):\n",
" encoder_embedded = multihead_attn(queries = encoder_embedded,\n",
" keys = encoder_embedded,\n",
" q_masks = en_masks,\n",
" k_masks = en_masks,\n",
" future_binding = False,\n",
" num_units = size_layer,\n",
" num_heads = num_heads)\n",
"\n",
" with tf.variable_scope('encoder_feedforward_%d'%i,reuse=tf.AUTO_REUSE):\n",
" encoder_embedded = pointwise_feedforward(encoder_embedded,\n",
" embedded_size,\n",
" activation = tf.nn.relu)\n",
" \n",
" self.logits = tf.layers.dense(encoder_embedded[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.001"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Attention(size_layer, size_layer, learning_rate, df_log.shape[1], df_log.shape[1])\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y\n",
" },\n",
" ) \n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" )\n",
" },\n",
" )\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0)\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0)\n",
" },\n",
" )\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0817 12:08:12.096583 140064997701440 deprecation.py:323] From <ipython-input-6-24d2a24c36ef>:91: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"W0817 12:08:12.104836 140064997701440 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0817 12:08:12.294501 140064997701440 deprecation.py:506] From <ipython-input-6-24d2a24c36ef>:92: calling dropout (from tensorflow.python.ops.nn_ops) with keep_prob is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `rate` instead of `keep_prob`. Rate should be set to `rate = 1 - keep_prob`.\n",
"W0817 12:08:12.305350 140064997701440 deprecation.py:323] From <ipython-input-6-24d2a24c36ef>:73: to_float (from tensorflow.python.ops.math_ops) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use `tf.cast` instead.\n",
"W0817 12:08:12.446460 140064997701440 deprecation.py:323] From <ipython-input-6-24d2a24c36ef>:33: add_dispatch_support.<locals>.wrapper (from tensorflow.python.ops.array_ops) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use tf.where in 2.0, which has the same broadcast rule as np.where\n",
"train loop: 100%|██████████| 300/300 [01:41<00:00, 2.97it/s, acc=96.7, cost=0.00409] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.99it/s, acc=97.3, cost=0.00184] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.98it/s, acc=96.7, cost=0.00351] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.98it/s, acc=97.9, cost=0.00112] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:41<00:00, 2.97it/s, acc=98, cost=0.00113] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.98it/s, acc=97.5, cost=0.00165] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:41<00:00, 2.96it/s, acc=95.8, cost=0.00513]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:41<00:00, 2.98it/s, acc=98, cost=0.000974] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.99it/s, acc=96.8, cost=0.00322] \n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdeVyU1f7A8c/MACqLgisiqGh6NLXcEpVETdS2a8u9v26kJJbeNBVX0lxI3IJUXDCXvFmWmmllerW6hFlmZmnSNcseXEDcRcUEccRZfn/MgKDgCgzL9/16zWtmnnPmeb7n4Yh855znPDqr1YoQQgghhBBCiLJP7+gAhBBCCCGEEEIUDUnwhBBCCCGEEKKckARPCCGEEEIIIcoJSfCEEEIIIYQQopyQBE8IIYQQQgghyglJ8IQQQgghhBCinJAETwghhBBCCCHKCSdHByCEEEKIu6OU0gETgFcAT+AL4F+apl28rl51QAM0TdMeLmRf/YFwoAlwEVgNTNA0zXRdvSbAb8Anmqb1y7N9ODAaqAEkASM1TdtuL4sA+gMNgLPAIk3TZt1b64UQQhRERvCEEEIUOaVUhfoC0YHtfREIBQIBH6AKEFdAvRhg/y325QqMBGoCAUAPYGwB9d4GduXdoJQKAKKBfwDVgHeB9Uopg72Kzh6rF/AoMEwp9fwt4hFCCHEXKtR/wEIIIUApNR4YBNQGjgITNU1br5SqBJwGHtY0bZ+9bi0gFWigadoZpdSTwHSgIfAHMFjTtL32uinAYqCv7a1yw5Yg3HAse30D8Ba2kZ0MYA625MRZ0zSTUqoaEAs8DliA94A3NE0zF9CmDsB8oDlwGfgUGK1pWra9vAUwD2gHXAXma5o20x7DOOBle4xJwNOAAUjOicW+j2+BlZqm/VspFWZv18/YEpfFSqn3gGXAg4AV+C8wVNO0C/bP+9lj7ILtC9aPsI14nQK6apr2m71ebSDFfs7TbvrDhL8B72qadtT+2RjgG6XUEE3TsuzbOgMtgXfs7SyQpmmL87w9rpRaBXS/7jw/D1wAdgD35SlqCPyuadov9nofAIuwndOTmqa9lfdQSqkN2JLSNbdonxBCiDskI3hCCFHxHMKWZFQDooCVSqm6mqZdAT4DQvLUfQ74zp7ctQGWY5sOWANYCmy0J4Y5QoAnAE97YlTgsex1BwGPAa2BttgSq7zeB0zYEok2QC9gYCFtMgOjsI0+dcI2+vQqgFLKA0gAvsI2ynUfsMX+udH2mB8HqgIvAVmFHON6AcBhoA4wA9so1Zv2YzQH/IAp9hgMwCbgCLZkqB6wxp6ArgH65dlvCLAlJ7lTSl1QShU4rdJOd93rStimWeYcdyEwDFvSeSeCgN9z3iilqgJTsZ2z630JGJRSAfZjvgT8ii15zcc+rbRL3n0LIYQoOjKCJ4QQFYymaevyvP1YKfU60AHYgO26q6XARHv5C/b3AP8Clmqa9pP9/Qql1ASgI/CdfduCnNGk2zjWc9hG0o4BKKWisSVmKKXqYEu6PDVNuwxcUkrNzYmhgDb9kudtilJqKdAV26jdk8ApTdPm2MuNQE4bBgKvaZqm2d//z358jxtO3I1OaJqWMx3SBBy0PwDSlFKxwBv29x2wJX4Rea5p225/XgGsU0qN1zTNim3KZe6Il6ZpnjeJ4SvgNaXUWiAd22gk2KZbgu2aup80TftFKdXqNtoEgFLqJaA9+RPqadhGC48ppa7/SAa2UdPt2JLMC8Bj9vZcbwq2L5jfu914hBBC3D5J8IQQooJRSr2IbRSmoX2TO7aRL4CtgKv9mqrT2EbX1tvLGgD97Ytp5HDBlrjkOJrn9a2O5XNd/byvGwDOwMk8yYT++v3nOU5TbNM522NLbpyAnKTPD9tIYkFuVnYr17e1DtemYHrY403Pc5wj1y9YAqBp2k9KqSygm1LqJLYRxo23GcNy+76/xdbmOdimbR5TSvlgS/Da3UmjlFJPYxuJDNY07ax9W2sgGNtIakFeBgYALbAlub2ATUqpNpqmnciz72HYprR2sY8YCyGEKGKS4AkhRAWilGqA7TqxHsCPmqaZlVK/Yp/mZ3+/Fts0wdPAJk3TMuwfPwrM0DRtxk0OkTtic6tjAScB3zyf9cvz+ihwBahZUFJUgMVAIhCiaVqGUmoktgU/cvZV2IIeR4HGwL7rtl+yP7tiW1ESwPu6OtePTs20b2uladp5e6K0MM9x6iulnAppzwps0zRPYVud0lhIvPlommbBNkr4BoBSqhdw3P7oA9QF/rAnyVWAKkqpU0C9Qq5lfBTbz+yJnGsC7bphS9JT7ftyxzYl835N09pi+yJgk6ZpSfb6X9mT1c7AJ/Z9vwSMB4JyRm2FEEIUPUnwhBCiYnHDloTkXN81ANsCHHmtBj4HznFtqibY/vBfr5RKwLa4iCu2P/y35UkC7+RYa4ERSqnN2BKqnOmFaJp2UikVD8xRSk0GMgF/wFfTtO+4kQe2RCxTKdUMGJJzXGzXvsXak77F2EYd77dPNf03ME0p9Qe2kadWwHFN09KUUseBfvbpnv2xJYI34wH8BfyllKoHROQp+xlbQhutlHoD2zWD7TRN+8FevhLb9NAMbFM0b4v99gde2K4FbI5tFHOqpmkWpdSXXBs5Bfgntim3TxWS3D0CrAKe0TTt5+uK3yH/gihj7fseYn+/C5iolIrDtjhNMNAUe+KslOqLLQHurmna4dttnxBCiDsni6wIIUQFomnaH9im8f2IbYSuFfDDdXV+wpZw+WBbPCNn+25sC6MsxDb18CAQdg/HWgbEA3uxjb59ge1atpzk40Vsydgf9uN9gm1EqiBjsSUvGfb9fpwnjgygJ7api6eAA1xbHTIWW6IZjy1BfBfbSBf2tkZgS3RbYFs58maisC0W8xewGduCNTkxmO3Hvw/bqqTHsCVcOeVHgT3YEuLv8+5UKZWplOpSyDFrYjtvl7D9rJZrmvaOfZ9XNE07lfOwx3XV/hqlVH37vuvb9zUZ22I4X9i3Z9qTRDRNy7puX5mAMc8qnx9gSwC/tZ/HBcArmqb9aS+fjm1hnl159r3kFudTCCHEXdBZrXe6qJYQQghR9JRSjwFLNE1r4OhYHEEptRzbwi2THB2LEEKIskumaAohhHAIpVQVbCNp8dhuNfAG1xZ0qVCUUg2BZyl8ERMhhBDitsgUTSGEEI6iwzatMR3bFM39QKRDI3IApdQ0bNeqzdI0LdnR8QghhCjbZIqmEEIIIYQQQpQTMoInhBBCCCGEEOVEWbwGrxLwELblpm9Y5lkIIYQQQgghyjkDtpWld2G7b2yuspjgPcR1S0gLIYQQQgghRAXUBdied0NZTPBOAqSnX8JiKV3XD9ao4c65c5mODkOUAtIXRA7pCyKH9AWRl/QHkUP6gshxJ31Br9fh5eUG9twor7KY4JkBLBZrqUvwgFIZk3AM6Qsih/QFkUP6gshL+oPIIX1B5LiLvnDDJWuyyIoQQgghhBBClBOS4AkhhBBCCCFEOVEWp2gWyGw2kZ6ehsmU7bAYzpzRY7FYHHb80kavN1Clijvu7tXQ6XSODkcIIYQQQohyr9wkeOnpaVSu7Iqbm7fDkgknJz0mkyR4AFarFbPZREbGBdLT06hevbajQxJCCCGEEKLcKzdTNE2mbNzcqspIUSmh0+lwcnLG07MG2dlGR4cjhBBCCCFEhVBuEjxAkrtSSKfTA7IylBBCCCGEECWhXCV4QgghhBBCCFGRSYJXTLZt+5a+ff/BgAEvkJqa4uhwbpCRkcGqVSsKLc/Ozmb06OE88UQPnniiRwlGJoQQQgghhLhbkuAVkw0bPuPllwfz3nurqV+/4W1/zmy+4V6FxSIzM4PVqz8otFyv1xMS0o958xaVSDxCCCGEEEKUJl9//RWPPx7Mb7/9z9Gh3JFys4pmabJgwRz27k0kNfUI69evIy5uKTt37mDp0oVYLBY8Pb2IiJiAr68fe/bsZv782SjVnKQkjUGDhtC6dRvi4uZy6NABsrOzadOmPcOHj8JgMJCWdoZ582Zx7NhRAIKDexMaOoD4+K9Yt+4jTKarAAwdOpL27TtgsViIjX2LPXt24ezsgqtrFRYvXk5sbAyZmZmEhb1A5cqVWbJkeb42ODk58dBDAZw8eaLEz58QQgghhBCOkp5+nkmTxrNu3RqaNWtO7drejg7pjpTbBO+H306yfe/JYtn3ww/UJbBV3ULLw8PHkJSkERISSmBgF9LTzzN9eiRxce/g79+ITZs+JypqEsuW2aZIJicfJiJiAi1bPgBAdPQ0Wrduy/jxk7FYLERFTWLz5o306fMMU6dOplOnQGbMmAXAhQsXAAgI6EjPnr3R6XSkpqYwYsSrrF//BQcPJpGYuJuVK9eh1+u5ePEiAKNHj2PgwFDef391sZwjIYQQQgghyppNmzYybtxo0tPPM2bMOEaOHEulSpUcHdYdKbcJXmny++/7aNy4Kf7+jQB4/PE+zJkTQ1bWJQB8ff1ykzuA7du3sX//76xZswoAo9FI7dp1yMrKYt++vcyd+3ZuXU9PTwCOHz/GlCkTSUtLw8nJifPnz3Hu3Fl8fHwxmUxER0+jbdv2dO7cpaSaLYQQQgghRJmQlpbG66+PZePG9bRq9SBr1nxGq1YP3PqDpVC5TfACW918lK00qVLF9botVmbOnE29er75tmZlZRW6jylTJjJs2CiCgrphsVgIDn6Y7OxsatSoyYcfriUx8Rd27/6ZxYvjWL58ZTG0QgghhBBCiLLFarXy2WfrmDjxNTIzM5kwIZKhQ0fg7Ozs6NDumiyyUgJatGjFoUNJHDmSAsCXX26iSROFq6tbgfUDA4NYuXJF7oIrFy5c4MSJ47i6utKy5QOsXXttWmXOFM3MzEzq1vUBYPPmjWRnZwOQnp6O0WgkIKATgwcPw93dnRMnjuPm5obRaMRkMhVXs4UQQgghhCi1Tp48wYsLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,718 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"def encoder_block(inp, n_hidden, filter_size):\n",
" inp = tf.expand_dims(inp, 2)\n",
" inp = tf.pad(\n",
" inp,\n",
" [\n",
" [0, 0],\n",
" [(filter_size[0] - 1) // 2, (filter_size[0] - 1) // 2],\n",
" [0, 0],\n",
" [0, 0],\n",
" ],\n",
" )\n",
" conv = tf.layers.conv2d(\n",
" inp, n_hidden, filter_size, padding = 'VALID', activation = None\n",
" )\n",
" conv = tf.squeeze(conv, 2)\n",
" return conv\n",
"\n",
"\n",
"def decoder_block(inp, n_hidden, filter_size):\n",
" inp = tf.expand_dims(inp, 2)\n",
" inp = tf.pad(inp, [[0, 0], [filter_size[0] - 1, 0], [0, 0], [0, 0]])\n",
" conv = tf.layers.conv2d(\n",
" inp, n_hidden, filter_size, padding = 'VALID', activation = None\n",
" )\n",
" conv = tf.squeeze(conv, 2)\n",
" return conv\n",
"\n",
"\n",
"def glu(x):\n",
" return tf.multiply(\n",
" x[:, :, : tf.shape(x)[2] // 2],\n",
" tf.sigmoid(x[:, :, tf.shape(x)[2] // 2 :]),\n",
" )\n",
"\n",
"\n",
"def layer(inp, conv_block, kernel_width, n_hidden, residual = None):\n",
" z = conv_block(inp, n_hidden, (kernel_width, 1))\n",
" return glu(z) + (residual if residual is not None else 0)\n",
"\n",
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" kernel_size = 3,\n",
" n_attn_heads = 16,\n",
" dropout = 0.9,\n",
" ):\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
"\n",
" encoder_embedded = tf.layers.dense(self.X, size_layer)\n",
"\n",
" e = tf.identity(encoder_embedded)\n",
" for i in range(num_layers):\n",
" z = layer(\n",
" encoder_embedded,\n",
" encoder_block,\n",
" kernel_size,\n",
" size_layer * 2,\n",
" encoder_embedded,\n",
" )\n",
" z = tf.nn.dropout(z, keep_prob = dropout)\n",
" encoder_embedded = z\n",
"\n",
" encoder_output, output_memory = z, z + e\n",
" g = tf.identity(encoder_embedded)\n",
"\n",
" for i in range(num_layers):\n",
" attn_res = h = layer(\n",
" encoder_embedded,\n",
" decoder_block,\n",
" kernel_size,\n",
" size_layer * 2,\n",
" residual = tf.zeros_like(encoder_embedded),\n",
" )\n",
" C = []\n",
" for j in range(n_attn_heads):\n",
" h_ = tf.layers.dense(h, size_layer // n_attn_heads)\n",
" g_ = tf.layers.dense(g, size_layer // n_attn_heads)\n",
" zu_ = tf.layers.dense(\n",
" encoder_output, size_layer // n_attn_heads\n",
" )\n",
" ze_ = tf.layers.dense(output_memory, size_layer // n_attn_heads)\n",
"\n",
" d = tf.layers.dense(h_, size_layer // n_attn_heads) + g_\n",
" dz = tf.matmul(d, tf.transpose(zu_, [0, 2, 1]))\n",
" a = tf.nn.softmax(dz)\n",
" c_ = tf.matmul(a, ze_)\n",
" C.append(c_)\n",
"\n",
" c = tf.concat(C, 2)\n",
" h = tf.layers.dense(attn_res + c, size_layer)\n",
" h = tf.nn.dropout(h, keep_prob = dropout)\n",
" encoder_embedded = h\n",
"\n",
" encoder_embedded = tf.sigmoid(encoder_embedded[-1])\n",
" self.logits = tf.layers.dense(encoder_embedded, output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = test_size\n",
"epoch = 300\n",
"dropout_rate = 0.7\n",
"future_day = test_size\n",
"learning_rate = 1e-3"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], \n",
" dropout = dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value = np.zeros((1, num_layers * 2 * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {modelnn.X: batch_x, modelnn.Y: batch_y},\n",
" ) \n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" )\n",
" },\n",
" )\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0)\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0)\n",
" },\n",
" )\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0818 16:16:28.504163 139649888855872 deprecation.py:323] From <ipython-input-6-6c0655f4345e>:55: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"W0818 16:16:28.507718 139649888855872 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0818 16:16:28.696973 139649888855872 deprecation.py:323] From <ipython-input-6-6c0655f4345e>:13: conv2d (from tensorflow.python.layers.convolutional) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use `tf.keras.layers.Conv2D` instead.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0818 16:16:28.910956 139649888855872 deprecation.py:506] From <ipython-input-6-6c0655f4345e>:66: calling dropout (from tensorflow.python.ops.nn_ops) with keep_prob is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `rate` instead of `keep_prob`. Rate should be set to `rate = 1 - keep_prob`.\n",
"train loop: 100%|██████████| 300/300 [00:43<00:00, 7.09it/s, acc=96.6, cost=0.00251]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:43<00:00, 7.08it/s, acc=96.9, cost=0.00232] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:43<00:00, 6.99it/s, acc=94.1, cost=0.00764] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:43<00:00, 6.98it/s, acc=96.6, cost=0.00273]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:43<00:00, 7.02it/s, acc=97.7, cost=0.00113] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:43<00:00, 7.06it/s, acc=97.7, cost=0.00117]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:43<00:00, 6.98it/s, acc=96.4, cost=0.00286]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:43<00:00, 6.97it/s, acc=94.7, cost=0.00573] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:43<00:00, 6.94it/s, acc=93.9, cost=0.00807] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:43<00:00, 7.05it/s, acc=94.6, cost=0.006] \n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdd3gU1frA8e+WbHojvVGFASkCF0VAsYEotmunRUDBi0ovVy4igiA/EKmRZkGRJqCgXFCu14qoXEWQAOKg9JC26Qkpmy2/P2Z3SUjohA3wfp5nnp2dOTP7zuSw7DvnzBmdw+FACCGEEEIIIcSVT+/pAIQQQgghhBBCXBqS4AkhhBBCCCHEVUISPCGEEEIIIYS4SkiCJ4QQQgghhBBXCUnwhBBCCCGEEOIqIQmeEEIIIYQQQlwlJMETQgghhBBCiKuE0dMBCCGEEOLCKIqiA8YB/wBCgM+AZ1VVLXCu9wYWAo8BxcDrqqrOOs2+FgF9KizyAiyqqgY61y8H7gL8gXTnvt6psL0f8AbwhHPbXaqqdnaumwi8BJRV2H8rVVUPXszxCyGEqEpa8IQQQlxyiqJcUxcQPXi8TwGJQCcgFvAFkiqsnwg0BuoBdwD/VBTlnup2pKrqIFVVA1wTsApYW6HI/wH1VVUNAh4EpiiK8rcK698C6gDNnK8jTvmI1RX3L8mdEELUjGvqP2AhhBCgKMpYYCAQCRwDXlJVdb2ztScDuEVV1T3OshHAUaCeqqqZiqLcD0wB6gO/A4NUVU12lj2M1lrUW3ur+AOjq/ssZ3kD8DrQFygEZqIlJ16qqloVRQkGZgHdATvwHvCKqqq2ao7pJmAuWnJRAnwMjFRV1eJc3xyYA/wNKAfmqqo61RnDi8Azzhj3A38HDMAhVyzOfXwLLFdV9R1FUfo5j+tntCRroaIo7wFvAzcADuA/wAuqquY5t09wxngr2gXWVcBItNaw21RV3e0sFwkcdp5z8xn/mPAA8K6qqsec204HvlYU5TlVVYud57afqqq5QK6iKG8D/YDNZ9qp82/3KHC/a5mqqnsrFHE4p0bAr4qiNEVL+uJdrYfAr2eJXQghRA2QFjwhhLj2HEBLMoKBScByRVFiVFUtA9YBPSuUfQL4zpnctQGWoHUHDAMWAxuciaFLT+A+IMSZGFX7Wc6yA4F7gdZAW7TEqqL3AStwHdAGuBsYcJpjsqG1GIUDHdC6Ej4PoChKIPAlWlIT69zfV87tRjpj7g4EAU+jdWU8F+2Bg0AU8BqgQ2vlikVLNBPQWtBcyexG4AhachwHfOhMQD+kctfInsBXruROUZQ8RVFuOUMculPmvYHGiqKEAjHArgrrdwHNz+HYHgXMwJaKCxVFWaAoSjHwB5CG1iUU4CbnsU1SFCVLUZTdiqI8eso+H1AUJUdRlL2Kojx3DjEIIYS4ANKCJ4QQ1xhVVSt2u1utKMq/0H6gfwqsREvcXnKu7+V8D/AssFhV1f853y9VFGUccDPwnXPZPFdr0jl81hNoLWkpAIqiTENLzFAUJQot6QpRVbUEOKEoymxXDNUcU8XWosOKoiwGbkNrtbsfSFdVdaZzfSngOoYBwD9VVVWd73c5Pz+wyomrKlVVVVd3SCvwl3MCMCuKMgt4xfn+JrTEb4yrRRDY6nxdCqxVFGWsqqoOtC6Xr1c4tpAzxLAZrdvlGiAXrTUSwA8IcM7nVyifD5zLsfUFPnDG46aq6vOKogxBS6Jv5+Q9dfFAC7SW01jn+k2Kovyuquo+YA1aF84MtMT4Y0VR8lRVXXUOsQghhDgPkuAJIcQ1RlGUp9Baruo7FwWgtXwBfAP4KYrSHu3HeGtgvXNdPaCv8we+iwntB73LsQrzZ/us2FPKV5yvhzZQR5qiKK5l+lP3X+FzmqB152yHltwYOdlFMAGtJbE6Z1p3NqceaxQnu2AGOuPNrfA5Ryokd26qqv7P2Sp2u6IoaWgtjBvOMYYlzn1/i3bMM9G6baYARc4yQWhJrWu+8Ew7VBSlLlryNrC69c4uslsVRekDPAfMQ+sWWw5McR7jd4qifIPW6rpPVdXfK+ziR0VR5qIN/CIJnhBCXGKS4AkhxDVEUZR6aPeJ3QX8pKqqTVGU33B283O+X4PWTTAD2KiqqishOAa8pqrqa2f4CHeLz9k+C62LX3yFbRMqzB9Dax0Kry4pqsZCYCfQU1XVQkVRhqMlEK599TjNdsfQ7iPbc8ryE85XP8B1T1n0KWUcp7yf6lzWUlXVHEVR/g68WeFz6iqKYjzN8SxF66aZDnykqmppNWWqUFXVjtZK+AqAoih3A8eB46qq2p0J4w3Af52b3ADsrW5fFSQCP5zDIChGtHMHkFzN+lPPz6nrdGdYL4QQ4gJJgieEENcWf7Qf1677u/qjda2raCXwCZDNya6aoCVr6xVF+RJtcBE/tJaeLRWSwPP5rDXAMEVRNqElVK7uhaiqmqYoyhfATEVRXkZrjWqANojHd1QViJaIFTkH/HjO9blo977NciZ9C9FaHa93djV9B5isKMrvaN0rW6IlR2ZFUY4DfZzdPftyMpk5nUC0LpD5iqLEAWMqrPsZLaGdpijKK2j3DP5NVdUfnOuXo3UPLURLsM6Joih1gFC0ewGbobVivupM/AA+AMYrirId7V7BgUD/s+z2KWD6KZ8TCdyJdi5LgC5oFwFc92tuQRuM51+KovwfWjfMO4B/Ord/yFkmD7gRGIr2eAchhBCXmAyyIoQQ1xBnV7mZwE9oLXQtgR9OKfM/tIQrFvi8wvLtaAnCm2hdD/9CG5HxQj/rbeALtNafnWgDdljRkh/QEg0T2miducBHaIOGVGc02v2Chc79rq4QRyHQFa3rYjrwJ1ryAVpCtMYZRwHwLtqjBnAe6xi0RLc58OPpjtVpEtpgMfnAJrQBa1wx2Jyffx1aIpQCPFlh/TFgB1pC/H3FnSqKUqQoyq2n+cxwtPN2Au1vtURV1bcqrH8FrQvqEbT7JGeoqrrZud+6zn3XrfBZHdBaVSveO4kzruecceeiPe9uuKqqG5zxlwMPod03mY/2N3hKVdU/nNv3QKsvhWhJ53RVVZee5piEEEJcBJ3DcaYeFEIIIcTloSjKvcAiVVXreToWT1AUZQnawC3jPR2LEEKIK5d00RRCCOERiqL4orWkfYHWffAVTg7ock1RFKU+8Aja4yCEEEKICyZdNIUQQniKDq1bYy5aF819wASPRuQBiqJMRhvkZYaqqoc8HY8QQogrm3TRFEIIIYQQQoirhLTgCSGEEEIIIcRV4kq8B88bbYjlNE6OtCaEEEIIIYQQ1woD2sjSv6A9N9btSkzwbuSUIaSFEEIIIYQQ4hp0K7C14oIrMcFLA8jNPYHdXrvuHwwLCyA7u8jTYYhaQOqCcJG6IFykLoiKpD4IF6kLwuV86oJeryM01B+cuVFFV2KCZwOw2x21LsEDamVMwjOkLggXqQvCReqCqEjqg3CRuiBcLqAuVLllTQZZEUIIIYQQQoirhCR4QgghhBBCCHGVuBK7aAohhBBCCCEuMZvNSm6uGavV4ulQrkmZmXrsdnulZUajidDQCAyGc0/bJMETQgghhBBCkJtrxsfHD3//aHQ6nafDueYYjXqs1pMJnsPh4MSJAnJzzYSHx5zzfqSLphBCCCGEEAKr1YK/f5Akd7WETqfD3z/ovFtUJcETQgghhBBCAEhyV8tcyN9DEjwhhBBCCCGEuEpIgieEEEIIIYSodbZs+ZbevR+jf/9eHD162NPhVFFYWMiKFUtPu95isTBy5BDuu+8u7rvvrssWlyR4QgghhBBCiFrn00/X8cwzg3jvvZXUrVv/nLez2ao8+/u8WSwW0tPT2L9fpbi4uNoyRUWFrFz5wWn3odfr6dmzD3PmLLjoeM6HjKIphBBCCCGEqFXmzZtJcvJOjh49wvr1a0lKWsy2bT+yePGb2O12QkJCGTNmHPHxCezYsZ25c99AUZqxf7/KwIHP0bp1G5KSZnPgwJ9YLBbatGnHkCEjMBgMmM2ZzJkzg5SUYwB06dKNxMT+/Oc/n7F69UpKS0ux2208/PDjtGt3EwaDgTfemMaOHb/g5WXCz8+XhQuXMGvWdIqKiujXrxc+Pj4sWrSk0jEYjUZuvLE9aWmpl/XcSYInhBBCCCGEqOSH3WlsTU6rkX3f0iqGTi3PPOz/0KGj2L9fpWfPRDp1upXc3BymTJlAUtJbNGjQkI0bP2HSpPG8/bbWRfLQoYOMGTOOFi1aATBt2mRat27L2LEvY7fbmTRpPJs2beDBBx/m1VdfpkOHTrz22gwAMjMzSE1NJSQklCFDRmAyeVNSUsyUKa/w5JO92L//D3bu3M7y5WvR6/UUFBQAMHLkiwwYkMj776+skfN0oSTBE0IIIYQQQtRqe/fuoVGjJjRo0BCA7t0fZObM6RQXnwAgPj7BndwBbN26hX379vLhhysAKC0tJTIyiuLiYvbsSWbmzCRyc3PJzs6iqKgQnU5HQUEBGzasIzc3B6PRi9zcHLKzs4iNjcdqtTJt2mTatm1Hx463Xv4TcB4kwRNCCCGEEEJU0qnl2VvZahNfX79TljiYOvUN4uLiKy3Nzc3B4XCwb99eHA4HJpM30dGx1KlTh9dem8jgwSPo3Pl27HY7XbrcgsViISwsnGXL1rBz569s3/4zCxcmsWTJ8st3cOdJBlkRQgghhBBC1GrNm7fkwIH9HDlyGIDPP99I48YKfn7+1Zbv1Kkzy5cvxWazYbfbOXz4ID/+uJUjRw7ToEEjvv/+Oxo2vI5mza7Hx8cHk8lEUVERMTGxAGzatAGLRXvAeG5uLqWlpbRv34FBgwYTEBBAaupx/P39KS0txWq1XpZzcK6kBU8IIYQQQghRq4WGhjJ+/KtMmvQSNpuNkJBQJkyYfNryw4aNYt68WfTu/Rh2ux2DwUjPnn1o374jr732OklJs3jhhQHo9Qa6du1Line truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,706 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"def position_encoding(inputs):\n",
" T = tf.shape(inputs)[1]\n",
" repr_dim = inputs.get_shape()[-1].value\n",
" pos = tf.reshape(tf.range(0.0, tf.to_float(T), dtype=tf.float32), [-1, 1])\n",
" i = np.arange(0, repr_dim, 2, np.float32)\n",
" denom = np.reshape(np.power(10000.0, i / repr_dim), [1, -1])\n",
" enc = tf.expand_dims(tf.concat([tf.sin(pos / denom), tf.cos(pos / denom)], 1), 0)\n",
" return tf.tile(enc, [tf.shape(inputs)[0], 1, 1])\n",
"\n",
"def layer_norm(inputs, epsilon=1e-8):\n",
" mean, variance = tf.nn.moments(inputs, [-1], keep_dims=True)\n",
" normalized = (inputs - mean) / (tf.sqrt(variance + epsilon))\n",
" params_shape = inputs.get_shape()[-1:]\n",
" gamma = tf.get_variable('gamma', params_shape, tf.float32, tf.ones_initializer())\n",
" beta = tf.get_variable('beta', params_shape, tf.float32, tf.zeros_initializer())\n",
" return gamma * normalized + beta\n",
"\n",
"def cnn_block(x, dilation_rate, pad_sz, hidden_dim, kernel_size):\n",
" x = layer_norm(x)\n",
" pad = tf.zeros([tf.shape(x)[0], pad_sz, hidden_dim])\n",
" x = tf.layers.conv1d(inputs = tf.concat([pad, x, pad], 1),\n",
" filters = hidden_dim,\n",
" kernel_size = kernel_size,\n",
" dilation_rate = dilation_rate)\n",
" x = x[:, :-pad_sz, :]\n",
" x = tf.nn.relu(x)\n",
" return x\n",
"\n",
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" kernel_size = 3,\n",
" n_attn_heads = 16,\n",
" dropout = 0.9,\n",
" ):\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
"\n",
" encoder_embedded = tf.layers.dense(self.X, size_layer)\n",
" encoder_embedded += position_encoding(encoder_embedded)\n",
" \n",
" e = tf.identity(encoder_embedded)\n",
" for i in range(num_layers): \n",
" dilation_rate = 2 ** i\n",
" pad_sz = (kernel_size - 1) * dilation_rate \n",
" with tf.variable_scope('block_%d'%i):\n",
" encoder_embedded += cnn_block(encoder_embedded, dilation_rate, \n",
" pad_sz, size_layer, kernel_size)\n",
" \n",
" encoder_output, output_memory = encoder_embedded, encoder_embedded + e\n",
" g = tf.identity(encoder_embedded)\n",
"\n",
" for i in range(num_layers):\n",
" dilation_rate = 2 ** i\n",
" pad_sz = (kernel_size - 1) * dilation_rate\n",
" with tf.variable_scope('decode_%d'%i):\n",
" attn_res = h = cnn_block(encoder_embedded, dilation_rate, \n",
" pad_sz, size_layer, kernel_size)\n",
"\n",
" C = []\n",
" for j in range(n_attn_heads):\n",
" h_ = tf.layers.dense(h, size_layer // n_attn_heads)\n",
" g_ = tf.layers.dense(g, size_layer // n_attn_heads)\n",
" zu_ = tf.layers.dense(\n",
" encoder_output, size_layer // n_attn_heads\n",
" )\n",
" ze_ = tf.layers.dense(output_memory, size_layer // n_attn_heads)\n",
"\n",
" d = tf.layers.dense(h_, size_layer // n_attn_heads) + g_\n",
" dz = tf.matmul(d, tf.transpose(zu_, [0, 2, 1]))\n",
" a = tf.nn.softmax(dz)\n",
" c_ = tf.matmul(a, ze_)\n",
" C.append(c_)\n",
"\n",
" c = tf.concat(C, 2)\n",
" h = tf.layers.dense(attn_res + c, size_layer)\n",
" h = tf.nn.dropout(h, keep_prob = dropout)\n",
" encoder_embedded += h\n",
"\n",
" encoder_embedded = tf.sigmoid(encoder_embedded[-1])\n",
" self.logits = tf.layers.dense(encoder_embedded, output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = test_size\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 5e-4"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], \n",
" dropout = dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value = np.zeros((1, num_layers * 2 * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {modelnn.X: batch_x, modelnn.Y: batch_y},\n",
" ) \n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" )\n",
" },\n",
" )\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0)\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0)\n",
" },\n",
" )\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0829 00:04:33.873839 140104212150080 deprecation.py:323] From <ipython-input-6-1aeaade5f897>:44: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"W0829 00:04:33.883059 140104212150080 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0829 00:04:34.265801 140104212150080 deprecation.py:323] From <ipython-input-6-1aeaade5f897>:4: to_float (from tensorflow.python.ops.math_ops) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use `tf.cast` instead.\n",
"W0829 00:04:34.294613 140104212150080 deprecation.py:323] From <ipython-input-6-1aeaade5f897>:24: conv1d (from tensorflow.python.layers.convolutional) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use `tf.keras.layers.Conv1D` instead.\n",
"W0829 00:04:36.600379 140104212150080 deprecation.py:506] From <ipython-input-6-1aeaade5f897>:82: calling dropout (from tensorflow.python.ops.nn_ops) with keep_prob is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `rate` instead of `keep_prob`. Rate should be set to `rate = 1 - keep_prob`.\n",
"train loop: 100%|██████████| 300/300 [00:14<00:00, 20.69it/s, acc=93, cost=0.0106] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:14<00:00, 20.99it/s, acc=97.6, cost=0.00116]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:14<00:00, 20.94it/s, acc=95.2, cost=0.00553]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:14<00:00, 20.97it/s, acc=95.4, cost=0.00442]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:14<00:00, 21.88it/s, acc=95.6, cost=0.00393]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:14<00:00, 21.01it/s, acc=95.3, cost=0.00454]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:14<00:00, 21.05it/s, acc=96.7, cost=0.00229]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:14<00:00, 21.01it/s, acc=97.1, cost=0.00178]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:14<00:00, 20.80it/s, acc=95.3, cost=0.00492]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:14<00:00, 20.94it/s, acc=90.6, cost=0.0192] \n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdeVhV1frA8S+HQUYBEVQExXFlDjklKubtllNO1a28oZJWejNTTNPsOuN01ZySnK5dyzTHbk6o6c9uZWpWzmm6UVJBUEEF5YhwPMPvj3NARFRU4CC+n+fxOeestfbe795nRbystdd2sFgsCCGEEEIIIYR49OnsHYAQQgghhBBCiMIhCZ4QQgghhBBClBKS4AkhhBBCCCFEKSEJnhBCCCGEEEKUEpLgCSGEEEIIIUQpIQmeEEIIIYQQQpQSkuAJIYQQQgghRCnhZO8AhBBCCPFglFIOwAjgHcAH2Az8Q9O0q7b6L4DugCHXZt6appnusK8JwJuAJ3AAeE/TtKO52rQBpgEKSAWGaJq2WilVHlgPPAE4AseAoZqm7bJt1xv4D3A91yE7a5r2w8NdASGEEHlJgieEEKLQKaWcNE0z2juO4mLH830DiADCsCZcXwHRQK9cbaZpmjaqAPt6DXgLaAWcASYCS4HGAEqpJ4Hltn3/H+CNNakE0Nu2PQFYgBeBjUqpgFzX5WdN01o92GkKIYQoKEnwhBDiMaOU+gjoCwQACcBITdPWKqXKABeAVpqmHbG19QfigaqapiUrpTpj/cU/BPgD6Kdp2mFb29PAfKCH9aPyAIbmdyxbe0eso0G9gHRgBtbkxFnTNKNSyhuYCXQEzMDnwNg7jD41Az4B6mAdJfov1tElg62+LjAbaALcAD7RNG2yLYbhwNu2GGOBl7COQp3KjsW2jx+AZZqmfWYbkeoL/Io1yZqvlPocWAQ8hTXJ2Yp1BCzNtn2wLcZnsN4isQIYApwH/qJp2u+2dgHAads1T7nrlwldgP9ompZg23Yq8D+l1LuapmXcY9u8qgE7NU3707avZcDgXPWjgIWapm2xfb5k+4emaZmAZttOB5gAX6AckHyfcQghhHgIcg+eEEI8fuKwJhneQBSwTClVSdO0LOAbIDxX227Aj7bkrhGwGOt0QD9gIbDBlhhmCwc6AT62xCjfY9na9gVeABpiHSV6KU+cXwBGoCbQCGgH9LnDOZmwJiPlgRbA80B/AKWUF7Ad+BYItO3vO9t2Q2wxdwTKYh2FKmhiFAr8CVQAJgEOwL9sx6gDBAPjbDE4AjFYR8ZCgMrASlsCuhLomWu/4cB32cmdUipNKXW3kS+HPO/LALVylfVXSl1WSu1TSr1yl/2sBGoopWorpZyxJt7f5qpvbovnd6XUOaXUMqVUudw7UEodBjKBDcBnmqblTu4aKaUuKqVilVKjlVLyR2YhhCgC8sNVCCEeM5qmrcn1cZVS6p9AM6z3UC3HmriNtNV3t30G+AfWEZxfbJ+XKKVGYP3F/0db2Zzs0aQCHKsb1pG0swBKqSlYEzOUUhWwJl0+mqZdB64ppWZlx5DPOe3L9fG0Umoh8Beso3adgfOaps2w1WcC2efQB/hQ0zTN9vmQ7fhet1242yVpmhZte28ETtr+AaQopWYCY22fm2FN/IblmrK40/a6BFijlPpI0zQL1imX03KdW/Y0yPx8C3yolFqNdYrmcFu5u+11DvABcAVrgrxKKXU++964PM7ZYtKwJswJwHO56oNssbUDkmxxR2Mdsc2OtYFSyhV4GXDJte0OoB7WBLcusArrNfvXXc5NCCHEA5AETwghHjNKqTewjlyF2Io8sY58AXwPuCulQrFO12wIrLXVVQV6KaUG5tqdC9bEJVtCrvf3OlZgnva531cFnIFzSqnsMl3e/ec6Tm2s0zmbYk1unIDspC8Y60hifu5Wdy95z7UCN6dgetniTc11nDP53aenadovSqkM4Fml1DmsI4wbChjDYtu+f8B6zjOwTts8a9v3/lxtNyulvgL+BuSX4I0Bnrbt7zzWUcX/KaXq2qZ7Xgc+1zQt1na+k7GOjOY9n0xghVLqmFLqoKZph7Knfdr8rpQaDwxDEjwhhCh0kuAJIcRjRClVFet9Ys9jXfTCpJQ6iG2an+3zaqzTBC8AMZqmpds2TwAmaZo26S6HsBT0WFhHjIJybRuc630CkAWUL+DiJfOxrvoYrmlaulLqfeDVXPt6/Q7bJQA1gCN5yq/ZXt2Bq7b3FfO0seT5PNlWVl/TtMtKqZeAT3Mdp8pdFmNZgjWhOg98bUuS7knTNDPWUcKxAEqpdkCi7V9+LNw6pTO3hsCq7BFV4Aul1GzgSWAvcJhbzznv+eflDFTHNip6H3EIIYR4CJLgCSHE48UD6y/X2fd3vYl16lxuy4F1WBfQGJmrfBGwVim1HeviIu7As8COXEng/RxrNTBIKbUJa0KVPb0QTdPOKaW2ATOUUqOxrtJYDQjSNO1HbueFNRHTK6WeAN7NPi7We99m2pK++VhHHZ+0TTX9DJiglPoD6/TK+kCipmkpSqlEoKdtumcvrIng3XhhnQp5RSlVGesIVbZfsSa0U5RSY7FOgWySa6rkMqyJUDrWaZAFYrsHzhfrvYB1sI5ijrclfiilXsU6jTMDaIM1iexyh939BrymlFqJ9dr1wJqkZU87/RwYbVt85TzwEdZri1KqOdbfKX7FukBNJNZ7E3+x1b8A7Nc07YLt+xkN5J6+K4QQopDIIitCCPEY0TTtD6zT+H7GOkJXnzzT9WyJzzWsUyi35Crfi3VhlE+xTj08CfR+iGMtArZhHRk6gPUZbkasyQ9YV6d0wbpaZyrwNVCJ/A3Fer9gum2/q3LFkQ60xZrYnMe6lP9fbdUzsSaa27AmiP8B3Gx1fbEmaZew3je2+07nahOFdbGYK8AmrAvWZMdgsh2/JtZVSc8Cf89VnwDsx5oQ/5R7p0opvVLqmTscszzW63YN63e1WNO0f+eqH4R1NC8N+Bjom/3sOaVUFdu+q9jaTsWaZB60tR8MvJK9CqimaYuBL7EmbWewjrBG2rYtA8zFeq0Ssd4/2UnTtCRb/fPAYaXUNVu832Ad8RRCCFHIHCyWe82wEEIIIYqebZRngaZpVe0diz0opRZjXbilIM+sE0IIIfIlUzSFEELYhVLKDetI2jas0/nGcnNBl8eKUioE6+InjewcihBCiEecTNEUQghhLw5YpzWmYp2ieQzrSo6PFaXUBKyLvHysadope8cjhBDi0SZTNIUQQgghhBCilHgUR/CcsD5PSaaXCiGEEEIIIR5Hd8yJHsUkqSrWlduewfYgVyGEEEIIIYR4jARhXXW5JhCXu+JRTPCyl8j+6a6thBBCCCGEEKJ0q0QpSPDOAaSmXsNsLln3D/r5eXLpkt7eYYgSQPqCyCZ9QWSTviByk/4gsklfENnupy/odA74+nqALTfK7VFM8EwAZrOlxCV4QImMSdiH9AWRTfqCyCZ9QeQm/UFkk74gsj1AXzDlLXgUF1kRQgghhBBCCJEPSfCEEEIIIYQQopR4FKdo5stkMpKamoLRaLBbDMnJOsxms92OX9LodI64uXni6emNg4ODvcMRQgghhBCi1Cs1CV5qagquru54eFS0WzLh5KTDaJQED8BisWAyGUlPTyM1NYVy5QLsHZIQQgghhBClXqmZomk0GvDwKCsjRSWEg4MDTk7O+Pj4YTBk2jscIYQQQgghHgulJsEDJLkrgRwcdICsDCWEEEIIIURxKFUJnhBCCCGEEEI8ziTBKyI7dvxAjx6v8uab3YmPP23vcG6Tnp7OV18tuWO9wWBgyJCBdOr0PJ06PV+MkQkhhBBCCCEelCR4RWT9+m94++1+fP75cqpUCSnwdibTbc8qLBJ6fTrLl395x3qdTkd4eE9mz55XLPEIIYQQQghRUlgsFtau/ZqOHdtw+PBBe4dzX0rNKpolyZw5Mzh8+ADx8WdYu3YN0dEL2bNnNwsXforZbMbHx5dhw0YQFBTM/v17+eST6ShVh9hYjb5936Vhw0ZER88iLu4EBoOBRo2aMnDgYBwdHUlJSWb27I85ezYBgDZt2hMR8Sbbtn3LmjUrMBpvAPDee+/TtGkzzGYzM2dOY//+33B2dsHd3Y358xczc+ZU9Ho9vXt3x9XVlQULFt9yDk5OTjz9dCjnziUV+/UTQgghhBDCXn7//RAjRw5nz57d1K//FBUrBto7pPtSahO8Xb+fY+fhc0Wy71YNKhFWv9Id6yMjPyA2ViM8PIKwsGdITb3MxIljiI7+N9WqVScmZh1RUaNYtMg6RfLUqT8ZNmwE9eo1AGDKlAk0bNiYjz4ajdlsJipqFJs2baBr15cZP340LVqEMWnSxwCkpaUBEBranLZt2+Pg4EB8/GkGDerP2rWbOXkylgMH9rJs2Rp0Oh1Xr14FYMiQ4fTpE8EXXywvkmskhBBCCCHEo+TixYv8618TWLbsC8qVK8eMGXPo3j0CR0dHe4d2X0ptgleSHD16hBo1alOtWnUAOnbsyowZU8nIuAZAUFBwTnIHsHPnDo4dO8rKlV8BkJmZSUBABTIyMjhy5DCzZs3Naevj4wNAYuJZxo0bSUpKCk5OTly+fIlLly4SGBiE0WhkypQJNG7clJYtnymu0xZCCCGEEKLEu3HjBosX/5uPP55CRsY1/vGPdxk69CO8vX3sHdoDKbUJXlj9u4+ylSRubu55SixMnjydypWDbinNyMi44z7GjRvJgAGDad36WcxmM23atMJgMODnV56lS1dz4MA+9u79lfnzo1m8eFkRnIUQQgghhBCPlu+//47Roz8iNlbjr399ngkTplC7trJ3WA9FFlkpBnXr1icuLpYzZ04DsGVLDLVqKdzdPfJtHxbWmmXLluQsuJKWlkZSUiLu7u7Uq9eA1atvTqvMnqKp1+upVMk6P3jTpg0YDAYAUlNTyczMJDSLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,705 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.LSTMCell(size_layer, state_is_tuple = False)\n",
"\n",
" backward_rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" forward_rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop_backward = tf.contrib.rnn.DropoutWrapper(\n",
" backward_rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" forward_backward = tf.contrib.rnn.DropoutWrapper(\n",
" forward_rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.backward_hidden_layer = tf.placeholder(\n",
" tf.float32, shape = (None, num_layers * 2 * size_layer)\n",
" )\n",
" self.forward_hidden_layer = tf.placeholder(\n",
" tf.float32, shape = (None, num_layers * 2 * size_layer)\n",
" )\n",
" self.outputs, self.last_state = tf.nn.bidirectional_dynamic_rnn(\n",
" forward_backward,\n",
" drop_backward,\n",
" self.X,\n",
" initial_state_fw = self.forward_hidden_layer,\n",
" initial_state_bw = self.backward_hidden_layer,\n",
" dtype = tf.float32,\n",
" )\n",
" self.outputs = tf.concat(self.outputs, 2)\n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value_forward = np.zeros((1, num_layers * 2 * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * 2 * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" ) \n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value_forward = np.zeros((1, num_layers * 2 * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * 2 * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0812 10:20:04.613218 140016646534976 deprecation.py:323] From <ipython-input-6-32a8ad1d5669>:12: LSTMCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.LSTMCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0812 10:20:04.617547 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f579b74fd68>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 10:20:04.620435 140016646534976 deprecation.py:323] From <ipython-input-6-32a8ad1d5669>:16: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n",
"W0812 10:20:04.623959 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f579b6efa20>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0812 10:20:04.949644 140016646534976 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0812 10:20:04.954938 140016646534976 deprecation.py:323] From <ipython-input-6-32a8ad1d5669>:42: bidirectional_dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.Bidirectional(keras.layers.RNN(cell))`, which is equivalent to this API\n",
"W0812 10:20:04.955546 140016646534976 deprecation.py:323] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn.py:464: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0812 10:20:05.149145 140016646534976 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 10:20:05.156026 140016646534976 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:961: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 10:20:05.712592 140016646534976 deprecation.py:323] From <ipython-input-6-32a8ad1d5669>:45: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [01:39<00:00, 3.04it/s, acc=97.8, cost=0.00113] \n",
"W0812 10:21:46.695034 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f58216a3208>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 10:21:46.695935 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f5790ef9e10>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 3.03it/s, acc=97.1, cost=0.00187] \n",
"W0812 10:23:27.984155 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f57919d69e8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 10:23:27.985092 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f57008a7b70>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.97it/s, acc=97.8, cost=0.00118] \n",
"W0812 10:26:50.307250 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f5791f0d2e8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 10:26:50.308161 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f56dbd71160>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.97it/s, acc=97.1, cost=0.00237]\n",
"W0812 10:28:31.638492 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f56dbe750b8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 10:28:31.639337 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f56d982db38>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:37<00:00, 3.11it/s, acc=97.5, cost=0.00143] \n",
"W0812 10:30:09.934609 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f56d99130b8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 10:30:09.935530 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f56d7b95320>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:39<00:00, 3.01it/s, acc=97.4, cost=0.00163] \n",
"W0812 10:31:50.447502 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f56d7384cf8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 10:31:50.448328 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f56d56a2748>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.05it/s, acc=96.6, cost=0.00322] \n",
"W0812 10:33:30.276075 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f56d4e9bba8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 10:33:30.276944 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f56d2868da0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.07it/s, acc=97.7, cost=0.00133] \n",
"W0812 10:35:09.746517 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f56d290ccf8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 10:35:09.747369 140016646534976 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f56d03129b0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:39<00:00, 3.03it/s, acc=97.5, cost=0.00142] \n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdeVxV1fr48c85DDKIgqggoojTstSuU6GilCaiNvxut+8tJ1LTblaKqZkmauKUQ6JJzl3NNC3tZpoTVGpkNjiVabZRE3AWERRExDP8/tgHRMUZOIDP+/U6Lzhrr733s/dZKA9r2Aar1YoQQgghhBBCiNLPaO8AhBBCCCGEEEIUDknwhBBCCCGEEKKMkARPCCGEEEIIIcoISfCEEEIIIYQQooyQBE8IIYQQQgghyghJ8IQQQgghhBCijJAETwghhBBCCCHKCEd7ByCEEEKIe6OUMgAjgVcBT2AD8B9N0y7Ytk8FugEVgTRgvqZpk25xvIHAEMAbSADe1DRtW75zTQb62ap/BIzQNM2qlKoMrAEaAA7AAeAtTdN+zHfswcBwwA34AnhN07TLhXEfhBBCXCU9eEIIIQqdUuqB+gOiHa/3JSAcCAb8AFcgJt/2/wINNE2rALQGeiil/lXQgZRSQegJ3P+hJ4T/BVYrpRxsVf4D/BP4B/AI8Ax6YgmQCbwMVAG8gCnA17n3RSkVBowAngQCgNpA1H1euxBCiAI8UP8BCyGEAKXUCOAVoCpwFIjUNG21UqoccBpoo2naPlvdKkAyEKBp2hml1NPABKAW8CfQX9O0vba6icBcoIf+VrkDbxV0Llt9B2Aq0AvIAKajJydOmqaZlFIVgWigC2ABFgPvappmLuCaHgM+AB4CLgH/A4ZompZj294QmAk0B64AH2iaNskWw3Cgry3GBPQkxgE4khuL7RhbgWWapn2klOptu65f0ZOsuUqpxcBC9ATICsQCb2ialm7bv4Ytxrbof2Bdgd5bdgp4XNO0P2z1qgKJtnuecssPU0+y/qtp2lHbvlOAzUqp1zRNy9I0TbuuvgWoe5Nj1QL2a5q2y3asT4A5tvtyEv1zmq5p2jHb9um2ezBP07RsQLOVGwEzeqJXCThj2/e/mqbtt9UZD3yKnvQJIYQoRNKDJ4QQD57D6ElGRfRelGVKqWq24XJfog/py/UC8L0tuWsKLELvtfEG5gNrbYlhrm7AU4CnLTEq8Fy2uq8AnYEmQDP0xCq/jwETekLSFOjI1eGB1zMDg4HKQCv0nqLXAZRSHsC3wCb0Xq66wHe2/YbYYu4CVEDvhcq6yTmuFwT8DfgAEwED8J7tHA8BNYCxthgcgHVAEnoiVR34zJaAfgb0zHfcbsB3ucmdUipdKdXmFnEYrvu+HFAvt0ApNUIplQkcA9yB5Tc5zkbAQSkVZIv3ZeA39AQUoCHwe776v9vK8iil9gLZwFrgI03TztxiXx+llPctrksIIcQ9kB48IYR4wGiatirf28+VUu8Aj6HPoVqOnrhF2rZ3t70HfYjefE3TfrG9X6KUGgm0BL63lc3K7U26g3O9gN6TltsjNBk9MUMp5YOedHlqmnYJuKiUmpEbQwHXtCvf20Sl1HzgcfReu6eBU5qmTbdtzwZyr6Ef8Ha+nq7fbef3uOHG3eiEpmm5wyFNwCHbCyBFKRUNvGt7/xh64jcst0cQ2Gb7ugRYpZQaoWmaFX3I5dR81+Z5ixg2AW8rpVaiz7Ebbit3y7f/ZFvPXhP0JPr8TY6Vgd7zuQ09UUwHOttiAih/3b7ngfJKKUNuHU3THlFKuQDPAc756ha0L4AHkHqL6xNCCHGXJMETQogHjFLqJfSeq1q2ovLoPV8AWwA323ys0+hJwWrbtgCgl20hjlzO6IlLrqP5vr/dufyuq5//+wDACTiplMotM15//HznqY8+nLMFenLjCOQmfTXQexILcqttt3P9tfpwdQimhy3etHznScqX3OXRNO0XpVQW8IRS6iR6D+PaO4xhke3YW9GveTr6sM1j153DCuyxzYWLQv9MrtcX6IPe23YIvcd0nVKqqaZpJ9Dn2VXIV78CkJkvAcw9VzawQil1QCn1m6Zpv99kX9CTSiGEEIVIEjwhhHiAKKUC0OeJPQn8pGmaWSn1G7Zhfrb3K9GHCZ4G1mmalvtL+FFgoqZpE29xirxf9m93LvR5Xf759q2R7/ujwGWgckFJUQHmAnuAbpqmZSil3kRfLCT3WF1vst9RoA6w77ryi7avbsAF2/e+19WxXvd+kq2ssaZp55RS/wQ+zHeemkopx5tczxL0YZqngC9sSdJtaZpmQe8lfBdAKdUROG57FcQR/XoL0gT9806wvd9kSzhbo696uR99fuGvtu3/sJXdjBP6Yiq/59t3Zb59T2uaJr13QghRyCTBE0KIB4s7ehKSO7+rD9DoujrLga/Qh85F5itfiL6q4rfov+S7AU8A8fmSwLs510pgkFJqPXpClTu8EE3TTiql4oDpSqnR6D1AgYC/pmnfcyMP9EQsUynVAHgt97zoc9+ibUnfXPRex4dtQ00/AsYrpf5E77VqDBzXNC1FKXUc6Gkb7tmLmydG+WM4D5xXSlUHhuXb9it6QjtZKfUu+pzB5vkeI7AMPRHKQB+ieUeUUpXQFzP5G33eXzQwTtM0i22xk1fQ73M68CjwBvo8wYLsACKVUjHoC8x0AOpzNfn9BBiilNqA/rkOxbZip1KqJfrvFL+iL1ATgT438Zd8+36slPoUOAGMQp9jKYQQopDJIitCCPEA0TTtT/RhfD+h99A1Bn68rs4v6AmXH/rCG7nlO9EThg/Rhx4eAnrfx7kWAnHAXvTetw3oc9lyV8l8CT0Z+9N2vi+AahTsLfT5ghm2436eL44MIBR96OIp4CDQzrY5Gj0BikNPEP+L/qgBbNc6DD3RbQhsv9m12kShLxZzHliPvmBNbgxm2/nroq9Kegx4Md/2o8Bu9MTph/wHVUplKqXa3uScldHv20X0z2qRpmkL8m1/Dn0IagZ6EhlDvscoXHfsT9AXfNlquxezgFc1TfvLtn0+8DXwB3rSt56r8yHLAbPR79Vx9PmTT9mGdqJp2ib0eYVbbNefxNX5iUIIIQqRwWq9foSJEEIIUfyUUp3Rl9wPsHcs9qCUWoS+cMsoe8cihBCi9JIhmkIIIexCKeWK3pMWhz6c712uLujyQFFK1QL+hf44CCGEEOKeyRBNIYQQ9mJAH9aYhj5E8wAwxq4R2YHtod/7gGmaph2xdzxCCCFKNxmiKYQQQgghhBBlhPTgCSGEEEIIIUQZURrn4JVDX+r5JFdXWhNCCCGEEEKIB4UD+srSO9CfG5unNCZ4j3LdEtJCCCGEEEII8QBqC2zLX1AaE7yTAGlpF7FYStb8QW/v8qSmZto7DFECSFsQuaQtiFzSFkR+0h5ELmkLItfdtAWj0YCXlzvYcqP8SmOCZwawWKwlLsEDSmRMwj6kLYhc0hZELmkLIj9pDyKXtAWR6x7awg1T1mSRFSGEEEIIIYQoIyTBE0IIIYQQQogyojQO0SyQ2WwiLS0FkynHbjGcOWPEYrHY7fwljdHogKtrecqXr4jBYLB3OEIIIYQQQpR5ZSbBS0tLwcXFDXd3X7slE46ORkwmSfAArFYrZrOJjIx00tJSqFSpqr1DEkIIIYQQoswrM0M0TaYc3N0rSE9RCWEwGHB0dMLT05ucnGx7hyOEEEIIIcQDocwkeIAkdyWQwWAEZGUoIYQQQgghikOZSvCEEEIIIYQQ4kEmCV4RiY/fSo8e/0efPt1JTk60dzg3yMjI4NNPl9x0e05ODkOGDOSpp57kqaeeLMbIhBBCCCGEEPdKErwismbNl/Tt25/Fi5dTs2atO97PbL7hWYVFIjMzg+XLP7npdqPRSLduPZk5c06xxCOEEEIIIURJsnHjerp06cAff/xu71DuSplZRbMkmTVrOnv37iE5OYnVq1cREzOfn3/ezvz5H2KxWPD09GLYsJH4+9dg9+6dfPDB+yj1EAkJGq+88hpNmjQlJmYGhw8fJCcnh6ZNWzBw4GAcHBxISTnDzJnTOHbsKAAdOoQRHt6HuLhNrFq1ApPpCgBvvPEmLVo8hsViITp6Krt378DJyRk3N1fmzl1EdPQUMjMz6d27Oy4uLsybt+iaa3B0dOTRR4M4efJEsd8/IYQQQggh7OXEieO8884wNm5cx8MPN8LHp5q9Q7orZTbB+/GPk2zbe7JIjt3mkWoEN775Bx0RMZSEBI1u3cIJDm5LWto5JkwYQ0zMAgIDa7Nu3VdERY1i4UJ9iOSRI38zbNhIGjV6BIDJk8fTpEkzRowYjcViISpqFOvXr+XZZ59j3LjRtGoVzMSJ0wBIT08HICioJaGhYRgMBpKTExk06HVWr97AoUMJ7Nmzk2XLVmE0Grlw4QIAQ4YMp1+/cD7+eHmR3CMhhBBCCCFKE7PZzKJFC5g0aTwWi5nRo8fRv/8bODk52Tu0u1JmE7ySZP/+fdSpU5/AwNoAdOnyLNOnTyEr6yIA/v418pI7gG3b4jlwYD+fffYpANnZ2VSt6kNWVhb79u1lxozZeXU9PT0BOH78GGPHRpKSkoKjoyPnzqWSmnoWPz9/TCYTkyePp1mzFrRu3ba4LlsIIYQQQohS4Y8/fmfo0Ah++20P7dt3YMqUaAICatk7rHtSZhO84Ma37mUrSVxd3a4rsTJp0vtUr+5/TWlWVtZNjzF2bCQDBgwmJOQJLBYLHTq0IScnB2/vyixdupI9e3axc+evzJ0bw6JFy4rgKoQQQgghhChdMjMzmTp1EgsWzMHbuzILFizm//2/f5Xqx6/JIivFoGHDxhw+nEBSUiIAGzeuo149hZube4H1g4NDWLZsSd6CK+np6Zw4cRw3NzcaNXqElSuvDqvMHaKZmZlJtWp+AKxfv5acnBwA0tLSyM7OJiioFf37D6B8+fKcOHEcd3dLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,759 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.LSTMCell(size_layer, state_is_tuple = False)\n",
" \n",
" with tf.variable_scope('forward', reuse = False):\n",
" rnn_cells_forward = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X_forward = tf.placeholder(tf.float32, (None, None, size))\n",
" drop_forward = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells_forward, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer_forward = tf.placeholder(\n",
" tf.float32, (None, num_layers * 2 * size_layer)\n",
" )\n",
" self.outputs_forward, self.last_state_forward = tf.nn.dynamic_rnn(\n",
" drop_forward,\n",
" self.X_forward,\n",
" initial_state = self.hidden_layer_forward,\n",
" dtype = tf.float32,\n",
" )\n",
"\n",
" with tf.variable_scope('backward', reuse = False):\n",
" rnn_cells_backward = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X_backward = tf.placeholder(tf.float32, (None, None, size))\n",
" drop_backward = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells_backward, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer_backward = tf.placeholder(\n",
" tf.float32, (None, num_layers * 2 * size_layer)\n",
" )\n",
" self.outputs_backward, self.last_state_backward = tf.nn.dynamic_rnn(\n",
" drop_backward,\n",
" self.X_backward,\n",
" initial_state = self.hidden_layer_backward,\n",
" dtype = tf.float32,\n",
" )\n",
"\n",
" self.outputs = self.outputs_backward - self.outputs_forward\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value_forward = np.zeros((1, num_layers * 2 * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * 2 * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x_forward = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_x_backward = np.expand_dims(\n",
" np.flip(df_train.iloc[k : index, :].values, axis = 0), axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state_forward, last_state_backward, _, loss = sess.run(\n",
" [\n",
" modelnn.logits,\n",
" modelnn.last_state_forward,\n",
" modelnn.last_state_backward,\n",
" modelnn.optimizer,\n",
" modelnn.cost,\n",
" ],\n",
" feed_dict = {\n",
" modelnn.X_forward: batch_x_forward,\n",
" modelnn.X_backward: batch_x_backward,\n",
" modelnn.Y: batch_y,\n",
" modelnn.hidden_layer_forward: init_value_forward,\n",
" modelnn.hidden_layer_backward: init_value_backward,\n",
" },\n",
" )\n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value_forward = np.zeros((1, num_layers * 2 * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * 2 * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" batch_x_forward = np.expand_dims(\n",
" df_train.iloc[k : k + timestamp, :], axis = 0\n",
" )\n",
" batch_x_backward = np.expand_dims(\n",
" np.flip(df_train.iloc[k : k + timestamp, :].values, axis = 0), axis = 0\n",
" )\n",
" out_logits, last_state_forward, last_state_backward = sess.run(\n",
" [\n",
" modelnn.logits,\n",
" modelnn.last_state_forward,\n",
" modelnn.last_state_backward,\n",
" ],\n",
" feed_dict = {\n",
" modelnn.X_forward: batch_x_forward,\n",
" modelnn.X_backward: batch_x_backward,\n",
" modelnn.hidden_layer_forward: init_value_forward,\n",
" modelnn.hidden_layer_backward: init_value_backward,\n",
" },\n",
" )\n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" output_predict[k + 1 : k + timestamp + 1, :] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" batch_x_forward = np.expand_dims(df_train.iloc[upper_b:, :], axis = 0)\n",
" batch_x_backward = np.expand_dims(\n",
" np.flip(df_train.iloc[upper_b:, :].values, axis = 0), axis = 0\n",
" )\n",
" out_logits, last_state_forward, last_state_backward = sess.run(\n",
" [modelnn.logits, modelnn.last_state_forward, modelnn.last_state_backward],\n",
" feed_dict = {\n",
" modelnn.X_forward: batch_x_forward,\n",
" modelnn.X_backward: batch_x_backward,\n",
" modelnn.hidden_layer_forward: init_value_forward,\n",
" modelnn.hidden_layer_backward: init_value_backward,\n",
" },\n",
" )\n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" o_f = np.flip(o, axis = 0)\n",
" out_logits, last_state_forward, last_state_backward = sess.run(\n",
" [\n",
" modelnn.logits,\n",
" modelnn.last_state_forward,\n",
" modelnn.last_state_backward,\n",
" ],\n",
" feed_dict = {\n",
" modelnn.X_forward: np.expand_dims(o, axis = 0),\n",
" modelnn.X_backward: np.expand_dims(o_f, axis = 0),\n",
" modelnn.hidden_layer_forward: init_value_forward,\n",
" modelnn.hidden_layer_backward: init_value_backward,\n",
" },\n",
" )\n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0812 16:41:29.569112 139847292135232 deprecation.py:323] From <ipython-input-6-2e28fdecec52>:12: LSTMCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.LSTMCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0812 16:41:29.570642 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f302d208da0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 16:41:29.571565 139847292135232 deprecation.py:323] From <ipython-input-6-2e28fdecec52>:17: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0812 16:41:29.886489 139847292135232 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0812 16:41:29.889781 139847292135232 deprecation.py:323] From <ipython-input-6-2e28fdecec52>:30: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0812 16:41:30.079713 139847292135232 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 16:41:30.086595 139847292135232 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:961: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 16:41:30.565006 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f302d1de6d8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 16:41:30.647609 139847292135232 deprecation.py:323] From <ipython-input-6-2e28fdecec52>:54: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [01:39<00:00, 3.02it/s, acc=97.7, cost=0.00132] \n",
"W0812 16:43:12.068012 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f3022b11fd0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 16:43:12.148377 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f30229e3080>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 3.02it/s, acc=97.4, cost=0.00157]\n",
"W0812 16:44:53.274921 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2fdc217b00>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 16:44:53.357845 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2fdc1bc0f0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.98it/s, acc=97.3, cost=0.00171]\n",
"W0812 16:46:35.140946 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f847c0240>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 16:46:35.223572 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f847c00b8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.00it/s, acc=96.5, cost=0.00334] \n",
"W0812 16:48:14.756632 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f843747f0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 16:48:14.838256 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f722b19e8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:41<00:00, 2.98it/s, acc=97.9, cost=0.00113]\n",
"W0812 16:49:56.968556 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f6bd75b70>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 16:49:57.051066 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f6bd755f8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 3.01it/s, acc=97.7, cost=0.00145]\n",
"W0812 16:51:38.877053 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f6a0db3c8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 16:51:38.959546 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f6976aef0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:41<00:00, 2.98it/s, acc=97.3, cost=0.00172]\n",
"W0812 16:53:21.123231 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f67bdbcc0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 16:53:21.205539 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f67258e10>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.06it/s, acc=97.8, cost=0.00117]\n",
"W0812 16:55:00.356067 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f65677da0>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 16:55:00.437367 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f65677898>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.05it/s, acc=97.7, cost=0.00127]\n",
"W0812 16:56:40.365346 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f628e67b8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0812 16:56:40.448274 139847292135232 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f2f628e6320>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.97it/s, acc=97.2, cost=0.00216]\n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzde3xMZ/7A8c9ccr83FSRBQjkUXdQKUtoSbLVrt7v761ZV6Zbdtirqtlq3CuonRai0Ve2upVXtsrtaS9tVUVRVW2V/SjlxS0ISROQekzFz5vfHmYkgQUgySXzfr9e8ZuY8zznne848Yr7zPOc5BofDgRBCCCGEEEKIhs/o7gCEEEIIIYQQQtQMSfCEEEIIIYQQopGQBE8IIYQQQgghGglJ8IQQQgghhBCikZAETwghhBBCCCEaCUnwhBBCCCGEEKKRkARPCCGEEEIIIRoJs7sDEEIIIcTNURTFAEwF/gQEA58Cf1RVtfCKencAKqCqqnrfNbbXGlgK3A+UAStUVf2zs2w10B/wA04Dr6mq+pcK6/YH3gRaAt8CI1VVTXeWeQHLgN8Bpc51k275BAghhLiK9OAJIYSocYqi3FY/ILrxeJ8ChgOxQDjgAyRXUi8ROHStDSmK4gl8AWwFmgGRwOoKVf4XiFJVNRAYAsxVFOVe57p3Av8CZgB3AHuAv1dYdxbQFmgFPAj8WVGUX1TjOIUQQtyg2+o/YCGEEKAoykvAaCAMOAlMU1V1vbOX5Qxwn6qqB5x1mwAZQCtVVc8qivIIMBeIAn4CnlVVdb+zbhp6L80w/a3iB0yqbF/O+ibgNWAEUAQsQk9OPFRVtSmKEgQkAYMBDfgb8IqqqvZKjqkH8DrQAbgA/BOYoKqq1VneEVgC3AtcBF5XVXWeM4YpwDPOGFOBXwMm4IQrFuc2tgGrVVX9i6IoI53H9R16krVMUZS/Ae8CPwMcwH+AMaqq5jvXb+GMsQ/6D6wfAhPQe8PuV1X1R2e9MCDNec5zrvlhwi+Bv6qqetK5biKwVVGU51RVLXUu6w10At5xHmdVRgJZV/Ss7Xe9UFX1YIXlDuejDfAD8BvgoKqq65z7nAWcUxSlvaqqh9E/45GqquYBeYqivOvc3+fXOT4hhBDVJD14Qghx+zmGnmQEAQnAakVRmquqWobeCzO0Qt3HgO3O5K4rsAJ9OGAosBzY4EwMXYYCDwPBzsSo0n05644GHgK6AN3QE6uKVgI24C6gKzAQGFXFMdmB8cCdQC/0oYTPAyiKEgBsQU8mwp3bS3GuN8EZ82AgEPgD+hDCGxEDHAeaAq8CBvRernD0RLMFes+VK5ndCKSjJ8cRwEfOBPQj4MkK2x0KpLiSO0VR8hVFqXJYpXO/FV97ofeWufb7BvACekJ2LT2BNEVRPlMU5ZyiKNsURelcsYKiKG8pilIKHAay0YeEAnQE/s9VT1XVEvTPvqOiKCFA84rlztcdrxOPEEKImyA9eEIIcZtx9bI4/V1RlJeBHsAnwBr0xG2as/wJ53uAPwLLVVX91vl+laIoU9ETg+3OZUtdvUk3sK/H0HvSTgEoijIfPTFDUZSm6ElXsKqqF4ASRVEWu2Ko5Jh+qPA2TVGU5ejXkS0BHgFOq6q6yFluQb9GDPSE8c+qqqrO9//n3H/AVSfualmqqrqGQ9qAo84HQI6iKEnAK873PdATv8muHkFgp/N5FbBOUZSXVFV1oA+5fK3CsQVfI4bP0Yc7rgXy0HsjAXydz/HAt6qq/nBlslaJSPThk0PQE+BxwCfOXjirM5bnFUUZi55EP4B+nR6AP3Blb2MBEOAsc72/skwIIUQNkwRPCCFuM4qiPIXecxXlXOSP3vMF8CXgqyhKDPpwzS7AemdZK2CE8wu+iyd64uJyssLr6+0r/Ir6FV+3AjyAbEVRXMuMV26/wn7aoQ/n7I6e3JjRhw6C3pN2rLL1rlN2PVcea1MuDcEMcMabV2E/6RWSu3Kqqn7r7BV7QFGUbPQexg03GMMK57a3oR/zIvRhm6cURQlHT/DuvcFtXQB2qqr6mfN4FgLT0XsjK/bO2YGdiqI8CTyHPilLMXoPaEWB6ENviyu8t1xRJoQQooZJgieEELcRRVFaoV8n1h/4RlVVu6Io/8U5zM/5fi36MMEzwEZVVV1fxE8Cr6qq+uo1dlE+DPB6+0If4hdZYd0WFV6fRO8durOypKgSy4B9wFBVVYsURXkRfcZG17Yer2K9k+jXkR24YnmJ89kXcM1I2eyKOlcOeZznXNZZVdXziqL8Gn14pGs/LRVFMVdxPKvQh2meBv6hqqqlkjpXUVVVQ+8lfAVAUZSBQKbzMQR9aORPziTZB/BRFOU0EFHJtYz70SdruVFm9HMHcBD9Ojuccfg5yw6qqprnTFx/hj6JC87XFa/pE0IIUUMkwRNCiNuLH3oS4rq+62n0CTgqWgN8DORyaagm6MnaekVRtqBPLuKLPkxvR4UksDr7WguMUxRlE3pC5RpeiKqq2YqibAYWKYoyA70XKBqIVFV1O1cLQE/EihVFaY/es+QaMrgRSHImfcvQex3vdg41/QswR1GUn9CHV3YGMlVVzVEUJRN40jnccwSXkpmqBKAPPSxQFCUCmFyh7Dv0hHa+oiivoF8zeK+qql87y1ej95IVoQ/RvCHO2x+EoF8L2AG9F3O2qqqaoiifcannFOD36ENuf1XZRDXOGCYqihKH3pMbD5wDDjknfumHfi4vAHHoPwK4rtdcDyxQFOW3wCZgJrDfOcEKwHvAdEVR9qBfszgaePpGj1MIIcSNk0lWhBDiNqKq6k/ow/i+Qe+h6wx8fUWdb9ETrnDgswrL96B/MX8DfejhUfSZEG92X+8Cm9F7jvahT9hhQ09+QJ+d0hN9ts484B/oPVKVmYSevBQ5t1s+Rb8z+RyAPnTxNHAE/Voz0BOitc44CoG/ovd04TzWyeiJbkdgV1XH6pSAPllMAXqS868KMdid+78LfVbSU+gJl6v8JLAXPSH+quJGFUUpVhSlTxX7vBP9vJWgf1YrVFV9x7nNMlVVT7sezrguOl+jKEpL57ZbOuur6L2Ib6Of718BQ5zX3znQk+ZTzrKFwIuqqm5wrpsD/BZ9spk89AloKvaavoI+FDYd/XrNBaqqygyaQghRCwwOx/Um1RJCCCFqn6IoDwFvq6rayt2xuIOiKCvQJ26Z7u5YhBBCNFwyRFMIIYRbKIrig96Tthl92N4rXJrQ5baiKEoU+r3kuro5FCGEEA2cDNEUQgjhLgb0YY156EM0D6Ffu3VbURRlDvokLwtUVT3h7niEEEI0bDJEUwghhBBCCCEaCenBE0IIIYQQQohGoiFeg+cF/Bx9uunKpnkWQgghhBBCiMbMhD6z9Pfo940t1xATvJ9zxRTSQgghhBBCCHEb6gPsrLigISZ42QB5eSVoWv26fjA01J/c3GJ3hyHqAWkLwkXagnCRtiAqkvYgXKQtCJfqtAWj0UBIiB84c6OKGmKCZwfQNEe9S/CAehmTcA9pC8JF2oJwkbYgKpL2IFykLQiXm2gLV12yJpOsCCGEEEIIIUQjIQmeEEIIIYQQQjQSDXGIZqXsdht5eTnYbFa3xXD2rBFN09y2//rGaDTh4+OPv38QBoPB3eEIIYQQQgjR6DWaBC8vLwdvb1/8/Jq5LZkwm43YbJLgATgcDux2G0VF+eTl5XDHHWHuDkkIIYQQQohGr9EM0bTZrPj5BUpPUT1hMBgwmz0IDg7FarW4OxwhhBBCCCFuC40mwQMkuauHDAYjIDNDCSGEEEIIURcaVYInhBBCCCGEELczSfBqyY4d2xg27Hc8/fQTZGSkuTucqxQVFfHBB6uqLLdarUyYMJaHH+7Pww/3r8PIhBBCCCGEEDdLErxa8skn/+KZZ57lb39bQ8uWUTe8nt1+1b0Ka0VxcRFr1rxXZbnRaGTo0CdZsuStOolHCCGEEEKI+iQlZTMPPzyAH3/8P3eHUi2NZhbN+mTp0kXs37+PjIx01q9fR3Lycnbv3sXy5W+gaRrBwSFMnjyVyMgW7N27h9dfX4iidCA1VWX06Ofo0qUrycmLOXbsCFarla5duzN27HhMJhM5OWdZsmQBp06dBCAubhDDhz/N5s2fs27dh9hsFwEYM+ZFunfvgaZpJCW9xt693+Ph4Ymvrw/Llq0gKSmR4uJiRo58Am9vb95+e8Vlx2A2m/n5z2PIzs6q8/MnhBBCCCGEu+Tn5zFjxsv8/e9raN++A2FhzdwdUrU02gTv6x+z2bk/u1a2fd89zYnt3LzK8vj4iaSmqgwdOpzY2D7k5Z1n7tyZJCe/Q3R0azZu/JiEhOm8+64+RPLEieNMnjyVTp3uAWD+/Dl06dKNl16agaZpJCRMZ9OmDQwZ8iizZ8+gV69YXn11AQD5+fkAxMT0ZMCAQRgMBjIy0hg37nnWr/+Uo0dT2bdvD6tXr8NoNFJYWAjAhAlTGDVqOCtXrqmVcySEEEIIIURD85//fMakSeM4dy6HCRMmM378n/Hy8nJ3WNXSaBO8+uTgwQO0adOO6OjWAAwePIRFixIpLS0BIDKyRXlyB7Bz5w4OHTrIRx99AIDFYiEsrCmlpaUcOLCfxYvfLK8bHBwMQGbmKWbNmkZOTg5ms5nz53PJzT1HeHgkNpuN+fPn0K1bd3r37lNXhy2EEEIIIUSDcP58LtOmTeGf/1zL3Xd34oMP1nLPPV3cHdZNabQJXmzna/ey1Sc+Pr5XLHEwb95CIiIiL1taWlpa5TZmzZrGCy+Mp2/fB9A0jbi4+7BarYSG3sn7769l374f2LPnO5YtS2bFitW1cBRCCCGEEEI0PBs3bmDKlAnk5Z1n8uSXGTduIp6enu4O66bJJCt1oGPHzhw7lkp6ehoAn322kbZtFXx9/SqtHxvbl9WrV5VPuJKfn09WVia+vr506nQPa9deGlbpGqJZXFxM8+bhAGzatAGr1QpLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,675 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 hours, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.GRUCell(size_layer)\n",
"\n",
" rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer = tf.placeholder(\n",
" tf.float32, (None, num_layers * size_layer)\n",
" )\n",
" self.outputs, self.last_state = tf.nn.dynamic_rnn(\n",
" drop, self.X, initial_state = self.hidden_layer, dtype = tf.float32\n",
" )\n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value = np.zeros((1, num_layers * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" ) \n",
" init_value = last_state\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value = np.zeros((1, num_layers * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value = last_state\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0811 22:46:29.978655 140681713489728 deprecation.py:323] From <ipython-input-6-1b755385b006>:12: GRUCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.GRUCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0811 22:46:29.981659 140681713489728 deprecation.py:323] From <ipython-input-6-1b755385b006>:16: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0811 22:46:31.758260 140681713489728 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0811 22:46:31.762153 140681713489728 deprecation.py:323] From <ipython-input-6-1b755385b006>:27: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0811 22:46:32.109607 140681713489728 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0811 22:46:32.121295 140681713489728 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:564: calling Constant.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0811 22:46:32.136707 140681713489728 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:574: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0811 22:46:32.395149 140681713489728 deprecation.py:323] From <ipython-input-6-1b755385b006>:29: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [02:10<00:00, 2.45it/s, acc=97.1, cost=0.00211]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:09<00:00, 2.48it/s, acc=96.2, cost=0.00432]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:08<00:00, 1.88it/s, acc=96.7, cost=0.00239]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:05<00:00, 2.12it/s, acc=96.4, cost=0.00307]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:12<00:00, 1.95it/s, acc=96.9, cost=0.00227]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:15<00:00, 2.64it/s, acc=96.4, cost=0.00318]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:10<00:00, 2.59it/s, acc=96.9, cost=0.00256]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:11<00:00, 2.19it/s, acc=97.2, cost=0.00188]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:14<00:00, 2.48it/s, acc=96.9, cost=0.00226]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [02:06<00:00, 2.48it/s, acc=97.5, cost=0.00167]\n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdd5wddb3/8dfpdespW7PZ1EmnJ1QRFEQQpIgKioKiXtEr6LVdFQUBpYngRURR7P5UmhSRLkhNSCEkW2aTbDbbezu7e+rM/P6Yc042yW7qJrvZfJ6Px3nMnJk5Z77n7DeT857vd75jMQwDIYQQQgghhBCHP+tkF0AIIYQQQgghxMSQgCeEEEIIIYQQ04QEPCGEEEIIIYSYJiTgCSGEEEIIIcQ0IQFPCCGEEEIIIaYJCXhCCCGEEEIIMU1IwBNCCCGEEEKIacI+2QUQQgghxP5RFMUCfAf4ApAPPA18XlXVwZ22KwRUQFVV9dTdvN9s4GfA6UAceFBV1W+m1/0JeB/gA9qB21VV/XV63YnATcBxgAa8DHxFVdW29PpvAJ8GZgLdwH2qqt4xAV+BEEKInUgLnhBCiAmnKMoRdQJxEj/vp4ArgFOAUsAD/N8Y290G1OzujRRFcQLPAy8BxUA58KdRm/wYqFRVNRe4ALhZUZTj0usKgF8BlZghLgL8dtRrLemyFgDnAF9WFOXje/shhRBC7L0j6j9gIYQQoCjKt4HPAWGgCfiuqqqPKYriAjqAU1VV3ZjeNgQ0AjNVVe1UFOVDwM2YP+Srgf9SVfXd9LYNwC+AT5hPFR/w9bH2ld7eBtyO2bITAX6CGU4cqqqmFEXJA+4CzgV0zMDwA1VVtTE+03LgHmAhEAUeAb6mqmoivX4xcDdmC1MSuEdV1R+ly/At4LPpMtYBFwI2YGumLOn3eBn4k6qqv1YU5cr051qFGVx+oSjKb4EHgKMAA3gW+JKqqv3p189Il/E0zBOs/w/4GmZr2Omqqm5IbxcGGtLfeddu/5hwPvAbVVWb0q+9DXhJUZQvqqo6kl52MrAEM4B9djfvdSXQqqrqXaOWvZuZUVW1atRyI/2YA6xRVfVfo99IUZR7gVdGvfb2UatVRVEexwylf93D5xNCCLGPpAVPCCGOPFswQ0YecCPwJ0VRSlRVjQOPApeN2vajwCvpcHcM8CBmd8AA8EvgiXQwzLgMOA/ITwejMfeV3vZzwAeBo4FjMYPVaL8DUsBc4BjgbODqcT6TBnwVCAInYXYlvAZAUZQc4AXgGcxWrrnAi+nXfS1d5nOBXOAzwMg4+9jZCqAeKAJuwWyl+nF6HwuBGcAN6TLYgKeAbZjhuAz4azqA/hX45Kj3vQx4MRPuFEXpVxRl3G6V6f2OnncB80bt917gy5iBbHdOBBoURfmXoijdiqK8rCjK0tEbKIpyn6IoI0At0IbZJXQs7wGqxlqR7lZ62njrhRBCHBhpwRNCiCOMqqoPjXr6N0VR/hdYDjwO/AUzuH03vf7y9HOAzwO/VFV1Zfr57xVF+Q5mMMi01vws05q0F/v6KGZLWjOAoii3YgYzFEUpwgxd+aqqRoFhRVF+minDGJ9pzainDYqi/BLzOrK7gQ8B7aqq/iS9PgZkPsPVwDdVVVXTz9en95+zyxe3q1ZVVTPdIVPA5vQDoEtRlLuAH6SfL8cMft/ItAgCr6WnvwceUhTl26qqGphdLrMtXqqq5u+mDM8A31QU5e9AH2ZrJIA3Pf0KsFJV1TU7h7UxlANnYHa/fBG4FnhcUZQFmZZQVVWvURTlvzFD9Hsxr9PbgaIoy4DvAx8eZz83YJ5g/u0464UQQhwACXhCCHGEURTlU5gtV5XpRX7Mli+AfwNeRVFWYHbXPBp4LL1uJvDp9A/8DCdmcMloGjW/p32V7rT96PmZgANoUxQls8y68/uP2s98zO6cx2OGGzuQCX0zMFsSx7K7dXuy82ctYnsXzJx0eftG7WfbqHCXparqynSr2HsVRWnDbGF8Yi/L8GD6vV/G/Mw/wey22awoSilmwDtu3FfvKAq8luluqSjKncD3MFsj148qrwa8pijKJ4EvYg7KQvo1c4F/AdeqqvrqzjtQFOXLmF1aT0u3GAshhJhgEvCEEOIIoijKTMzrxN4HvKmqqqYoyjuku/mln/8ds5tgB/CUqqqR9MubgFtUVb1lN7vIdgPc074wu/iVj3rtjFHzTZitQ8GxQtEYfgGsAy5TVTWiKMp1wEdGvdd4A3o0YV5HtnGn5cPpqRfIjEhZvNM2O3d5/FF62VJVVXsVRbkQs3tkZj8ViqLYx/k8v8fsptkOPKyqamyc8u5AVVUds5XwBwCKopwNtKQfFwAlQHU6JHsAj6Io7UDZGNcyvot5XdzesmN+d6T3PROzK+xNqqr+ceeNFUX5DPBt4D2ZVlshhBATTwKeEEIcWXyYISRzfddVmANwjPYX4B9AD9u7aoIZ1h5TFOUFzMFFvJjd9P4zKgTuy77+DlyrKMo/MQNVpnshqqq2KYryHPATRVGuB4aAWUC5qqqvsKsczCA2pCjKAsyWpcwAJU8Bd6VD3y8wWx0Xpbua/hq4SVGUaszulUuBFlVVuxRFaQE+me7u+WlGhZlx5AADwICiKGXAN0atW4UZaG9VFOUHmNcMHqeq6uvp9X/CbCWLYHbR3Cvp2x8UYF4LuBCzFfOHqqrqiqL8i+0tpwAfw+xy++GxBqpJl+F/FEV5P2ZL7lcwb2lQkx745UzM7zIKvB/zJMBl6XKUYY6+ea+qqvePUc5PYAbgM1RVrd/bzyeEEGLfySArQghxBFFVtRqzG9+bmC10S4HXd9pmJWbgKsXsbpdZvhpzYJR7MbsebsYceXF/9/UA8Bxmy9E6zAE7UpjhB8yufE7M0Tr7gIcxW6TG8nXM8BJJv+/fRpUjApyF2XWxHdiEea0ZmIHo7+lyDAK/wWzpIv1Zv4EZdBcDb4z3WdNuxBwsZgD4J+aANZkyaOn9z8UclbQZM3Bl1jcBazED8Q5dGxVFGVIU5bRx9hnE/N6GMf9WD6qq+qv0e8ZVVW3PPNLlSqbnURSlIv3eFentVcxWxPsxv+8PAxekr78zMENzc3rdncB1qqpmupJeDcwGbki/55CiKEOjynkz5sA8b49av0sQFEIIceAshrGnQbWEEEKIg09RlA8C96uqOnOyyzIZFEV5EHPglu9NdlmEEEIcvqSLphBCiEmhKIoHsyXtOcxbDfyA7QO6HFEURakELsa8HYQQQgix36SLphBCiMliwezW2IfZRbMGc3j9I4qiKDdhDvJyh6qqWye7PEIIIQ5v0kVTCCGEEEIIIaaJw7EFz445Kph0LxVCCCGEEEIcicbNRIdjSJqJOXLbaZijeQkhhBBCCCHEkaQcc9TlucCW0SsOx4CXGSL71d1uJYQQQgghhBDTWwnTIOC1AfT1DaPrU+v6wUDAT0/P0J43FNOe1AWRIXVBZEhdEKNJfRAZUhdExr7UBavVQkGBD9LZaLTDMeBpALpuTLmAB0zJMonJIXVBZEhdEBlSF8RoUh9EhtQFkbEfdUHbecHhOMiKEEIIIYQQQogxSMATQgghhBBCiGlCAp4QQgghhBBCTBMS8IQQQgghhBBimpCAJ4QQQgghhBDThAQ8IYQQQgghhJgmJOAJIYQQQgghxDQhAU8IIYQQQgghpgkJeEIIIYQQQgiRZhgGGzdu4Ic//D5nn3067777zmQXaZ/YJ7sAQgghhBBCCDHZtm1r4LHHHuaRR/6OqtZit9s544z3UVJSNtlF2ycS8IQQQgghhBBHpK6uLp544jEeeeTvrF69CoAVK07i9tt/yvnnX0ggEJjkEu47CXhCCCGEEEKII8bQUISnn36KRx99iFde+TeaprFo0RK+970bueiiS5gxo2Kyi3hAJOAJIYQQQgghprVEIsFLL73Ao4/+nWef/RfRaJQZMyr48pev4+KLL2XhwkWTXcQJIwFPCCGEEEIIMe3ous5bb73BI488xJNPPkZ/fz+BQIDLLvskF1/8UU44YTkWi2WyiznhJOAJIYQQQggxjW3YsJ5Nm+oIBkPZRyAQwGazTXbRJlxmBMxHHvk7//jHI7S2tuD1+jj33A9xySWX8p73nIHD4ZjsYh5UEvCEEEIIIYSYpgYG+rnkkvPp7+/fYbnFYiEQCIwKfcEdAuDo5aFQCL8/Z0q3djU0bOXRRx/i0Ucfoq5OxW638773ncUPfnATZ5/9QXw+32QX8ZCRgCeEEEIIIcQ09X//dzcDAwP85S8P4fP56e7uoquri+7uzKOb7u4uNm7cQHd31y5BMMPlcu0mDJohMBgMkZubh8PhwG534HDYcTgc2Gz29DL7hIbEzs5OnnjiUR555CHWrHkbgJNOOoU77vgi55//YQoLD78RMCeCBDwhhBBCCCGmoba2Vh544BdcfPGlvP/9H9ir1yQSCXp7e8YMgaMfdXUqXV2dxGKxfSqT3T469NnTQdAMhHa7bYdwaC7bHg63b2enr6+PN954FU3TWLx4Kd///k1cdNEllJWV789XNa1IwBNCCCGEEGIauvPOW0mlUnz729/b69c4nU6Ki0soLi7Z47aGYTA8PLxDEOzv70PTNJLJJKlUkmQyRTKZRNNS6WWp7DpzPpXeLrNs9GvNbRKJBCMjw9n3SqWSOBxOvvKVr3LRRZeyYMHCA/maph0JeEIIIYQQQkwzmzbV8ec//4Grr/4CM2dWHpR9WCwW/H4/fr+fyspZB2UfYt9ZJ7sAQgghhBBCiIl1yy034vX6uO66b0x2UcQhJgFPCCGEEEKIaWT16lU8/fSTfOlLXyEYDE52ccQhJgFPCCGEEEKIacIwDH74w+8TCoX5whe+NNnFEZNAAp4QQgghhBDTxAsvPMtbb73B17/+bfx+/2QXR0wCCXhCCCGEEEJMA5qmcfPNNzJr1mw++clPT3ZxxCSRUTSFEEIIIYSLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,704 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.GRUCell(size_layer)\n",
"\n",
" backward_rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" forward_rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop_backward = tf.contrib.rnn.DropoutWrapper(\n",
" backward_rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" forward_backward = tf.contrib.rnn.DropoutWrapper(\n",
" forward_rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.backward_hidden_layer = tf.placeholder(\n",
" tf.float32, shape = (None, num_layers * size_layer)\n",
" )\n",
" self.forward_hidden_layer = tf.placeholder(\n",
" tf.float32, shape = (None, num_layers * size_layer)\n",
" )\n",
" self.outputs, self.last_state = tf.nn.bidirectional_dynamic_rnn(\n",
" forward_backward,\n",
" drop_backward,\n",
" self.X,\n",
" initial_state_fw = self.forward_hidden_layer,\n",
" initial_state_bw = self.backward_hidden_layer,\n",
" dtype = tf.float32,\n",
" )\n",
" self.outputs = tf.concat(self.outputs, 2)\n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value_forward = np.zeros((1, num_layers * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" ) \n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value_forward = np.zeros((1, num_layers * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0812 17:04:18.991346 140383403915072 deprecation.py:323] From <ipython-input-6-5c392a5d20ef>:12: GRUCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.GRUCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0812 17:04:18.995361 140383403915072 deprecation.py:323] From <ipython-input-6-5c392a5d20ef>:16: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0812 17:04:19.316777 140383403915072 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0812 17:04:19.322190 140383403915072 deprecation.py:323] From <ipython-input-6-5c392a5d20ef>:42: bidirectional_dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.Bidirectional(keras.layers.RNN(cell))`, which is equivalent to this API\n",
"W0812 17:04:19.322940 140383403915072 deprecation.py:323] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn.py:464: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0812 17:04:19.515542 140383403915072 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 17:04:19.522486 140383403915072 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:564: calling Constant.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 17:04:19.531559 140383403915072 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:574: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 17:04:19.763414 140383403915072 deprecation.py:323] From <ipython-input-6-5c392a5d20ef>:45: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.98it/s, acc=97.1, cost=0.00199]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:39<00:00, 3.02it/s, acc=76.2, cost=0.139] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 3.00it/s, acc=97.1, cost=0.00205]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.99it/s, acc=95.3, cost=0.00587]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.97it/s, acc=96.2, cost=0.00386]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.99it/s, acc=97.1, cost=0.00196]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.98it/s, acc=96.7, cost=0.0032] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:39<00:00, 3.00it/s, acc=85.2, cost=0.0599] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.99it/s, acc=97.6, cost=0.00142]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.03it/s, acc=97.7, cost=0.00138]\n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdd3wcd5n48c/Ubeqy5CL3Nu52bMd24rjEaaSQQgmEQGgJ3B0Hd3C/cEe5C8cd9e64g4MDDo6QEEJIL04lCUmcOHbcS8q4F8myLMmSVbbMTvn9MStZ7k32SvLzfr32Naud2Zln5a9X++zzLUoQBAghhBBCCCGE6P3UfAcghBBCCCGEEKJ7SIInhBBCCCGEEH2EJHhCCCGEEEII0UdIgieEEEIIIYQQfYQkeEIIIYQQQgjRR0iCJ4QQQgghhBB9hCR4QgghhBBCCNFH6PkOQAghhBCnx7IsBfg68HmgBHgG+Jxt2y25/f8O3AAMAGqA79q2fe8xzrUQeBlIdnn4C7Zt39PlmI8CdwFDgb3Ap2zbXmJZ1q3AL7s8TwViwEzbtldZlvVl4ItAP6AN+CNwp23b7pn9BoQQQhxOKnhCCCG6nWVZ59UXiHl8vbcBnwDmAoMIk6r/7rK/HXg/UAx8EvixZVkXH+d8e2zbLuhy65rcXQH8APg0UAjMB7YB2Lb9+67PA/4qt2917ulPAtNt2y4CJgFTgS+d2UsXQghxNOfVH2AhhBBgWdY/AHcAlcBu4Bu2bT9mWVYEqAMusW17Y+7YCmAXMMy27X2WZV0H/CswHHgH+Avbttfnjt0B/By4NfzRSgD/72jXyh2vAT8kTDxagf8gTE4M27Zdy7KKgR8B1wA+cDdwl23b3lFe0yzgx8B4IAU8AnzFtm0nt38i8F/ADCAL/Ni27e/mYvh74LO5GDcBNwIasL0jltw5XgHus23715ZlfSr3ut4iTLJ+blnW3cCvCJOXAHiesALWnHv+kFyM8wi/YP0D8BXCStgC27Y35I6rBHbkfuf1x/3HDJO3/7Nte3fuuT8AXrYs6y9t207atn1Xl2OXW5a1BLgIWHqC8x7NPwPftm17We7nmuMc+0ngXtu2AwDbtrd22acQ/nuOPo0YhBBCnIBU8IQQ4vyzlTDJKCb80H6fZVkDbdvOAI8Ct3Q59mbg1VxydwHwG8LugOWEXfKezCWGHW4BrgVKconRUa+VO/YO4GpgGjCdMLHq6reAS5gIXABcCdx+jNfkAV8m7AJ4EXAZYRUJy7IKgReB5wirXKOBl3LP+0ou5muAIuAzHNpF8XhmE1ap+gPfIUxcvpe7xnhgCPCtXAwasBjYSZgcVwEP5BLQB4CPdznvLcBLHcmdZVnNlmVdcpw4lMPuR4Axhx9kWVYMuBB4+zjnqrQsq86yrO2WZf1nLknviH8mUGFZ1hbLsqoty/pp7pyHX2cYYXXv3sMe/5hlWS1AA2ES/MvDnyuEEOLMSQVPCCHOM7ZtP9Tlxz9alvU1YBbwBHA/4Qfvb+T2f4yDH8Q/B/zStu3luZ/vsSzr68Ac4NXcYz/pqCadxLVuJqykVQNYlvV9wsQMy7L6EyZdJbZtp4B2y7L+syOGo7ymVV1+3GFZ1i+BBYRVu+uAvbZt/0dufxroeA23A1+1bdvO/bwud/3CI35xR9pj23ZHd0gX2JK7AdRblvUjwvFq5F7zIA4dd/Z6bnsP8JBlWf+Qq3h9grCy2fHaSo4Tw3PAVy3LehBoIqxGAsSPcuwvcq/v+WOc6z3CZPs9YFgurh8RJvT9AQP4EGHCniX8N/wmB9tKh9uAJbZtb+/6oG3b9wP3W5Y1JndM3XFelxBCiNMkCZ4QQpxnLMu6jbByNTz3UAFh5Qvgz0DcsqzZhB/ApwGP5fYNAz5pWdYXu5zOJExcOuzucv9E1xp02PFd7w8jTChqLcvqeEw9/PxdrjOWMBmZSZjc6EBH0jeEsJJ4NMfbdyKHv9b+HOyCWZiLt6nLdXYebVIR27aXW5aVBBZallVLWGF88iRj+E3u3K8Qvub/IOy2WX1YbP9GOPbt0o5uk0eJYy9hd1GA7ZZlfZWw6vh5wm6vAP9t23Zt7pw/4tgJ3nePFbBt25sty3ob+B/gAyf1KoUQQpw0SfCEEOI8kus+9yvCStmbtm17lmWtJdfNL/fzg4TdBOuAxbZtt+aevhv4jm3b3znOJTqThxNdC6gFBnd57pAu93cDGaDfSc60+HNgDXCLbdutlmX9LWG1qeNcHz3G83YDo4CNhz3entvGgZbc/QGHHXN4ovTd3GOTbdveb1nWjcBPu1xnqGVZ+jFezz2E3TT3Ag/btp0+RryHsG3bJ6wS3gVgWdaVhGPjOsfHWZb1z4RdYRd0zK55kgJyQzls226yLKuaQ1/zEYmiZVkdk708fIJz64S/dyGEEN1MEjwhhDi/JAg/mHeM7/o0YWWnq/uBx4FGDq3O/Ap4zLKsFwknF4kDC4HXuiSBp3KtB4G/sSzracKEqqN7IbZt11qW9QLwH5Zl/SPh1PojgMG2bb/KkQoJE7E2y7LGAX/ZcV3CKtSPcknfzwmrjhNyXU1/DfyLZVnvEHavnAzU2LZdb1lWDfDxXHfPT3LihKQQOAAcsCyrCrizy763CBPa71uWdRfhmMEZtm2/kdt/H2H3yVbCLponxbKsMqCUcCzgeMIq5rdziR+5LrEfA+bZtt14gnNdmjvPLsLE+/uE3TA73A180bKs5wi7aH6Z8Hfb1SeBRw5vD5Zl3Q48mRvLOQH4GsfuKiqEEOIMyCQrQghxHrFt+x3CbnxvElboJgNvHHbMcsKEaxDwbJfHVxJOjPJTwq6HW4BPncG1fgW8AKwnrL49QziWrWOWzNsIk7F3ctd7GBjI0f0/wkSmNXfeP3aJoxW4grDr4l5gM3BpbvePCBPNFwgTxP8jXGqA3Gu9kzDRnciJZ578Z8LJYg4ATxNOWNMRg5e7/mjCBKoa+EiX/bsJlxQIgCVdT2pZVptlWfOOcc1+hL+3dsJ/q9/Ytv2/XfZ/l3DNui2587Tlxk0e7dwX5F5je267gUOXMvgXYAXhTKPvEv6bfafLuaKE4yrv4UhzgQ2WZbXn4n2GcP0+IYQQ3UwJgqN2xRdCCCHOKcuyrgZ+Ydv2sHzHkg+WZf2GcOKWb+Y7FiGEEL2XdNEUQgiRF7kp9i8lrJ71JxxH9thxn9RHWZY1nHDCkQvyHIoQQoheTrpoCiGEyBeFsFtjE2F3v3eBf8prRHlgWda/EE7y8m+HLy0ghBBCnCrpoimEEEIIIYQQfYRU8IQQQgghhBCij+iNY/AiwIWE0017JzhWCCGEEEIIIfoajXBm6RWE68Z26o0J3oUcNoW0EEIIIYQQQpyH5gGvd32gNyZ4tQBNTe34fs8aP1heXkBjY1u+wxA9gLQF0UHaguggbUF0Je1BdJC2IDqcSltQVYXS0gTkcqOuemOC5wH4ftDjEjygR8Yk8kPaguggbUF0kLYgupL2IDpIWxAdTqMtHDFkTSZZEUIIIYQQQog+QhI8IYQQQgghhOgjJMETQgghhBBCiD5CEjwhhBBCCCGE6CMkwRNCCCGEEEKIPkISPCGEEEIIIYToIyTBE0IIIYQQQog+QhI8IYQQQgghhOgjJMETQgghhBDiHAvSbSQX/4B2+618hyL6GEnwehmvqQa/pT7fYQghhBBCiDOQXvp7vD3v0vDsLwgy7fkOR/Qher4DECfHa9iJs+px3J1rQNUwp9+AOe1aFFXLd2hCCCGEEOIUZHeswt3yJvqo2bjbVpBZ8SjRSz6R77BEHyEJXg/nNe4KE7sdq8GMY864Cb+pBmflo7g71xBdeAda6aB8hymEEEIIIU6Cn24ls+Qe1PKhRC+9A7W0jJZVz2FY89Aqhuc7PNEHSILXQ3n7d+OsfBx3xyowYpgzbsScdAVKJAFAdutMMq/fS/LRfyJy4QcxJl2FokqPWyGEEEKInizzxn0EmXZi19yJouqULriF1rffIP36vcRv/CaKIp/nxJmRBK+H8fbX4Kx+HHfbCjCimNOvx5x8VWdi18EYNQtt4FgyS+4hs+yPuNtXE114O2px/zxFLoQQQgghjie7bQXu1uWYMz+AVj4EAC2aIDL7I6Rf+RXZ917DHL8wv0GKXk8SvB7Ca9qDs/oJ3K1vgRHBvOD9YWIXLTjmc9R4CdErv4S7eSnppffR/sg/Epl1M8bERfLtjxBCCCFED+KnWsi8fi9qv+GY0649ZJ8+5mI0+zUybz2EPmIGarQwT1GKvkASvDzzm2vJrH4Cd8ty0E3MaddiTnnfcRO7rhRFwRg7F23QeNKv/YbM0vtwd6wiuuAzqIUVZzl6IYQQQghxMjJv/I7ASRFbeMcRk+QpikJk7m0kH/knnOUPEV3wmTxFKfoCSfDyxD+wl8yqJ3C3LgPNwJx6NcaU96HGik7rfGpBGbGr/47se6+SWfYA7Q//I5GLbsGw5qMoSjdH3/sEQYBfvx2loAw1XpLvcIQQQghxHslufQt32wrMWR9CK6s66jFaWRXG5CvIrs9NuDJgzDmOUvQVkuCdY37LvrBit3kpqAbG5Kswp15z2oldV4qiYI5fiD54IulXf0Pmtbtxt68iOv/TqInSboi+d/Lqt5N58w94ezeBoqANHIc+eg7GiJlHjG0UQgghhOhOfvJA2DWzYiTmlKuPe2xkxo24W5eTfuNe4jd9S5bDEqdFErxzJEzsnsLd/AaoGsakK8PELl7c7ddSCyuIXXsn2bdfJvPWg7Q/9A2iF9+KPubi86qa57c1knnrYdwtb6JEC4lcfCtBupXsluVkXrubzOv3og2ejDF6DvqwC1CMSL5DFkIIIUQfEgQBmdfvJXDTxBbefsKETTGiRC76GOkXf0b2nZcxJ11xjiIVfYkkeGeZ31qPs/opspveAFXBmHg55rRrzno3QUVRMSddjj5kMulXfk36lV+Line truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,742 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.GRUCell(size_layer)\n",
" \n",
" with tf.variable_scope('forward', reuse = False):\n",
" rnn_cells_forward = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X_forward = tf.placeholder(tf.float32, (None, None, size))\n",
" drop_forward = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells_forward, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer_forward = tf.placeholder(\n",
" tf.float32, (None, num_layers * size_layer)\n",
" )\n",
" self.outputs_forward, self.last_state_forward = tf.nn.dynamic_rnn(\n",
" drop_forward,\n",
" self.X_forward,\n",
" initial_state = self.hidden_layer_forward,\n",
" dtype = tf.float32,\n",
" )\n",
"\n",
" with tf.variable_scope('backward', reuse = False):\n",
" rnn_cells_backward = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X_backward = tf.placeholder(tf.float32, (None, None, size))\n",
" drop_backward = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells_backward, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer_backward = tf.placeholder(\n",
" tf.float32, (None, num_layers * size_layer)\n",
" )\n",
" self.outputs_backward, self.last_state_backward = tf.nn.dynamic_rnn(\n",
" drop_backward,\n",
" self.X_backward,\n",
" initial_state = self.hidden_layer_backward,\n",
" dtype = tf.float32,\n",
" )\n",
"\n",
" self.outputs = self.outputs_backward - self.outputs_forward\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value_forward = np.zeros((1, num_layers * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x_forward = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_x_backward = np.expand_dims(\n",
" np.flip(df_train.iloc[k : index, :].values, axis = 0), axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state_forward, last_state_backward, _, loss = sess.run(\n",
" [\n",
" modelnn.logits,\n",
" modelnn.last_state_forward,\n",
" modelnn.last_state_backward,\n",
" modelnn.optimizer,\n",
" modelnn.cost,\n",
" ],\n",
" feed_dict = {\n",
" modelnn.X_forward: batch_x_forward,\n",
" modelnn.X_backward: batch_x_backward,\n",
" modelnn.Y: batch_y,\n",
" modelnn.hidden_layer_forward: init_value_forward,\n",
" modelnn.hidden_layer_backward: init_value_backward,\n",
" },\n",
" )\n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value_forward = np.zeros((1, num_layers * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" batch_x_forward = np.expand_dims(\n",
" df_train.iloc[k : k + timestamp, :], axis = 0\n",
" )\n",
" batch_x_backward = np.expand_dims(\n",
" np.flip(df_train.iloc[k : k + timestamp, :].values, axis = 0), axis = 0\n",
" )\n",
" out_logits, last_state_forward, last_state_backward = sess.run(\n",
" [\n",
" modelnn.logits,\n",
" modelnn.last_state_forward,\n",
" modelnn.last_state_backward,\n",
" ],\n",
" feed_dict = {\n",
" modelnn.X_forward: batch_x_forward,\n",
" modelnn.X_backward: batch_x_backward,\n",
" modelnn.hidden_layer_forward: init_value_forward,\n",
" modelnn.hidden_layer_backward: init_value_backward,\n",
" },\n",
" )\n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" output_predict[k + 1 : k + timestamp + 1, :] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" batch_x_forward = np.expand_dims(df_train.iloc[upper_b:, :], axis = 0)\n",
" batch_x_backward = np.expand_dims(\n",
" np.flip(df_train.iloc[upper_b:, :].values, axis = 0), axis = 0\n",
" )\n",
" out_logits, last_state_forward, last_state_backward = sess.run(\n",
" [modelnn.logits, modelnn.last_state_forward, modelnn.last_state_backward],\n",
" feed_dict = {\n",
" modelnn.X_forward: batch_x_forward,\n",
" modelnn.X_backward: batch_x_backward,\n",
" modelnn.hidden_layer_forward: init_value_forward,\n",
" modelnn.hidden_layer_backward: init_value_backward,\n",
" },\n",
" )\n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" o_f = np.flip(o, axis = 0)\n",
" out_logits, last_state_forward, last_state_backward = sess.run(\n",
" [\n",
" modelnn.logits,\n",
" modelnn.last_state_forward,\n",
" modelnn.last_state_backward,\n",
" ],\n",
" feed_dict = {\n",
" modelnn.X_forward: np.expand_dims(o, axis = 0),\n",
" modelnn.X_backward: np.expand_dims(o_f, axis = 0),\n",
" modelnn.hidden_layer_forward: init_value_forward,\n",
" modelnn.hidden_layer_backward: init_value_backward,\n",
" },\n",
" )\n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0812 17:35:02.847837 140485361571648 deprecation.py:323] From <ipython-input-6-0ad5fabb14ba>:12: GRUCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.GRUCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0812 17:35:02.849749 140485361571648 deprecation.py:323] From <ipython-input-6-0ad5fabb14ba>:17: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0812 17:35:03.166620 140485361571648 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0812 17:35:03.169869 140485361571648 deprecation.py:323] From <ipython-input-6-0ad5fabb14ba>:30: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0812 17:35:03.361743 140485361571648 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 17:35:03.368463 140485361571648 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:564: calling Constant.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 17:35:03.377521 140485361571648 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:574: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 17:35:03.612136 140485361571648 deprecation.py:323] From <ipython-input-6-0ad5fabb14ba>:54: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.04it/s, acc=97.5, cost=0.00174]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:39<00:00, 3.01it/s, acc=96.7, cost=0.00259]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:39<00:00, 2.99it/s, acc=97.4, cost=0.00178]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.05it/s, acc=96.2, cost=0.00314]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 2.99it/s, acc=97.4, cost=0.00166]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.02it/s, acc=97.3, cost=0.00162]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 3.00it/s, acc=96.1, cost=0.00347]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:40<00:00, 3.01it/s, acc=96.1, cost=0.004] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.05it/s, acc=95.3, cost=0.00537]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:38<00:00, 3.05it/s, acc=97.1, cost=0.00213]\n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOy9d5xtWVnn/d355MpVN4e+Yd++oSONHWhs0AYR5sVBB2RG0ddXB+OoqIg6M8jrKyIqY8BBxTgvBhCVURHoFhqEDjQ0dHPvbTjdN6fK6eSd1po/1j6nTtWtujnf9f18dq2406l99tm//TzrWYaUEo1Go9FoNBqNRqPRXP+YV/sANBqNRqPRaDQajUZzadACT6PRaDQajUaj0WhuELTA02g0Go1Go9FoNJobBC3wNBqNRqPRaDQajeYGQQs8jUaj0Wg0Go1Go7lB0AJPo9FoNBqNRqPRaG4QtMDTaDQajUaj0Wg0mhsE+2ofgEaj0Wg0mgvD930D+EXgrUAv8C/Afy6Xy5W0/b3Am4EeYBb4w3K5/O4VtvVa4BeA3UAL+Gfgp8vlcjVtfyPwU8AdwNPlcvmhJev/EfDNwDbgB8rl8p93tf0B8D1d3R0gLJfLxYs4fY1Go9Esg7bgaTQajeaS4/v+TfUC8Sqe71uA7wUeANYAWeD3utr/BNhRLpdLwP3Af/J9/w0rbKsH+P/S7dwKrAV+o6t9Bvht4D0rrP8c8KPAV5Y2lMvlHy6Xy4X2Avw18LfndIYajUajOS9uqh9gjUaj0YDv++8AfggYBo4Dv1Qul//B930PGAdeVi6X96V9h4BjwMZyuTzh+/7rUCJgE/A88MPlcvlrad8jwAeA/6SKfh742eX2lfa3gPcC3wdUgd9CiROnXC7Hvu/3AO8Dvh0QwJ8B7yyXy8ky5/RS4HdQwqQJ/B3wtnK5HKbtu1Di5G4gAn6nXC6/Oz2Gnwf+n/QYXwC+A7CAw+1jSbfxWeBD5XL5j33f//70vJ5GiawP+L7/Z8AHgdsBCXwK+LFyuTyXrr8+PcYHUS9Y/xp4GzAGfHO5XN6b9hsGjqSf+eQZ/5nw74A/KZfLx9N1fx34jO/7P1Iulxvlcrm8pL8Ati63oXK5/FddxYbv+x8E3tXV/q/pPn5whfV/P21vnemA0+viO4HXnamfRqPRaC4MbcHTaDSam4+DKJHRg3qA/5Dv+6vL5XIA/D3Kpa/NG4HPpeLuTuBPUe6AA8AfAv+YCsM2bwZeC/SmwmjZfaV9fwh4Dcrl7y6UsOrmz4EYJUjuBF4FLCsugAT4aWAQuA/4FpQ1Cd/3i8C/Ap9EWae2Ap9O13tbeszfDpSAHwAaK+xjKd8EHAJGgF8FDODXWLCArQd+OT0GC+XyeBQljtcCf5MK0L9hsfvim4FPt8Wd7/tzvu+/7AzHYSzJeyg3SdL13+H7fg04AeSBv+LceDmw/xz7ng/fCUwC/3YZtq3RaDQ3PdqCp9FoNDcZ5XK52zXuw77v/wLwUuB/ox7+/xD4pbT9P6ZlgP+MGsP1xbT8F77v/yJwL/C5tO5329akc9jXG1GWtBMAvu+/ByXM8H1/BCW6esvlchOo+77/P9rHsMw5PdNVPOL7/h+ixoP9NspSNFYul38rbW8B7XP4QeDtXZau59L9n8vYsFPlcrntDhkDB9IFYNL3/fcB70zLL0UJv59rWwSBL6TpXwB/6/v+O8rlskS5XL6369x6z3AMnwTe7vv+R1Bj7H4+rc91rf+e1LJ3B0pEz5/txHzffxhlWf2ms/W9AL4P+F/puWo0Go3mEqMFnkaj0dxk+L7/FpTlalNaVUBZvgAeA3K+738Tyl3zDuAf0raNwPf5vv8TXZtzUcKlzfGu/Nn2tWZJ/+78RlQgjlHf99t15tLtd+1nO8qd8yUocWMDbdG3HmVJXI4ztZ2Npec6woILZjE93tmu/RztEncdyuXyF33fbwAP+b4/irIw/uM5HsOfptv+LOqcfwvltnliyT4k8FXf91+NsqS+baUN+r5/L0rof1e5XH7hHI/jnPB9fwPwEMp6q9FoNJrLgBZ4Go1GcxPh+/5G1DixbwGeLJfLie/7z5K6+aXlj6DcBMeBf25HUUQJml8tl8u/eoZddKwyZ9sXMAqs61p3fVf+OBAAg8uJomX4APBV4M3lcrnq+/5PAd/Vta3vXmG948AWYN+S+nqa5oBKml+1pM9SC9S707o95XJ5xvf97wDe37WfDb7v2yucz1+g3DTHgI+Wy+UzjmNrUy6XBcpK+E4A3/dfBZxMl+WwUee7LKkb7j+iomB+eqV+F8H3Ao+Xy+VDl2HbGo1Go0ELPI1Go7nZyKNESHt81/+NCovfzV8BHwOmWXDVBCXW/sH3/X9FBRfJoawx/9YlAs9nXx8BftL3/Y+jBFXbvZByuTzq+/4jwG/5vv/fgBqwGVhXLpc/x+kUUUKs5vv+DuBH2vtFjX17Xyr6PoCyOu5MXU3/GPgV3/efR7lX7gFOlsvlSd/3TwLfk7p7fh9nEEZdxzAPzPu+vxb4ua62p1GC9j2+778TNWbw7nK5/Hja/iGUe2gVJYLOCd/3+4E+1FjAW1FWzP+3XC4L3/dNlKXsI8AccA/wY6hxgsttazfK5fMnyuXyPy3TbqGsqjZg+r6fAZJyuRyl7S7KamkATtoepiK0zVuAXz/X89NoNBrN+aODrGg0Gs1NRLlcfh7lxvckykK3B3h8SZ8vogTXGuATXfVfRgmG96NcDw8A338R+/og8AjwNZT17V9QY9naUTLfghJjz6f7+yiwmuX5WdR4wWq63Q93HUcVeBjlujgGvAi8Im1+H0oAPYISiH+CmmqA9Fx/DiV0dwFPrHSuKe9CBYuZBz6OCljTPoYk3f9WVFTSE8CbutqPo6YXkMDnuzfq+37N9/0HV9jnIOpzq6P+V39aLpf/qKv936NcUKsoEfl7dE2jsGTbPwMMAX+S1td83+8OsvK9qAilH0C5oTZRn3WbR9K6+4E/SvMv79rXfSiLrZ4eQaPRaC4jhpR6jLNGo9Forj6+778G+INyubzxah/L1cD3/T9FBW75r1f7WDQajUZz/aJdNDUajUZzVfB9P4uypD2CmmrgnSwEdLmp8H1/E/AG1HQQGo1Go9FcMNpFU6PRaDRXCwPl1jiLctH8OvDfr+oRXQV83/8VVJCX3yiXy4ev9vFoNBqN5vpGu2hqNBqNRqPRaDQazQ2CtuBpNBqNRqPRaDQazQ3C9TgGz0OFeh5lIdKaRqPRaDQajUaj0dwsWKjI0l9CzRvb4XoUePewJIS0RqPRaDQajUaj0dyEPAh8obviehR4owCzs3WEuLbGDw4MFJierl3tw9BcA+hrQdNGXwuaNvpa0HSjrwdNG30taNqcz7VgmgZ9fXlItVE316PASwCEkNecwAOuyWPSXB30taBpo68FTRt9LWi60deDpo2+FjRtLuBaOG3Img6yotFoNBqNRqPRaDQ3CFrgaTQajUaj0Wg0Gs0NghZ4Go1Go9FoNBqNRnODoAWeRqPRaDQajUaj0dwgaIGn0Wg0Go1Go9FoNDcIWuBpNBqNRqPRaDQazQ2CFngajUaj0Wg0Go1Gc4OgBZ5Go9FoNBqNRqPR3CBcjxOdazQajUaj0Wg0Gs0lpR4lTLRCJpshR6emeeHQISanp/jp17+Wjb3Fq31454wWeBqNRqPRaDQajeamQEpJJYoZrTUpHzvONw4e5ODhw5w8doTpUyeojp2gOnaSVmWus85rNnyMja945VU86vNDCzyNRqPRaDQajUZzQyGk5MT0DM+9eKAj4o4fP8r48WPMjZ6gNnEKEced/qZlM7xmLZs2bOSWl97Dts2b2bzpFrZs2crOnbuu4pmcP1rgaTQajUaj0Wg0Nyijo6f4+Z/+cV44dBAvmyOT8chksnieSjMZD8/LLMln0rS770K952XIZjOdvqpe9XVdF9M0MQzjsp9bkiQcP3mCvS8e5OuHDnLg8CGOHT3K+IljTJ88tsgKB5Atlhhet4Hb9tzG5k2vZ8eWW9i++RY2bdrM2rXrsO0bQxrdGGeh0Wg0Go1Go9FoAEhqNRrP7+d//81f8ssf+WuCJOaB1WswPA/R00sYRVSrFSYnJ2m1mgRBQKvVotVqEQQtoii6JMdhGAamaS5aDMPAMLrrjE6/dv3S9dQ6BtIwEKilFYbMTYySdB2rYVqURlYzvHY9277l29i8+Rb8W25h99atbL9lM729fZfkvK51tMDTaDQajUaj0WiuY6QQtI4cprFvL/V9e5l5oczv7HuOjx87ys41a3nfz/8SqysVql96Gqu3l8F//12U7rsfw1w+oH6SJB2xFwQBzaYSgUHQ6hKCAa1Wc0k+IAwDpJQIIZBSpHlVbi9SSqRcXFZ5VR8nCY0ooR5F1KOYRhjTiGJEuj0pBCXH5vZXvJpNGzex/ZZb2LNtK7s2bybnuVf407/20AJPo9FoNBqNRqO5zojn5qjv30tj/z7q+/ch6nUwDF7wPN75pSc5OT3NT/3kz/Bzb/9FHMcBoPeVDzPx4b9i/M/+mLnHPs3QG7+b3Hb/tG1blkU+nyefz1/286hGMaONoGsJmWqF5IEhwDNNVuVcVuU81uQ8VmU9RrIurqVne1sJLfA0Go1Go9FoNJprHBnHNA+8SH3fXhr79xIcPw6AVSpRuP0OHP9WPviZR/nt3/8d1q1bz8c+9iHuvfe+RdvIbtvGhl/8b1S/+BRTf/9RTrz31yjc/RIGv+uNuEPDl/X4EymZaoUdETfWCDjVCKjHSadPr2uzOuexu7/A6qzH6pxLn+dgXoHxfDcSZxV4vu//JvCdwCZgT7lc3pfWbwf+AhgApoG3lMvlFy+mTaPRaDQ3NzJJaHzj61S/9EVEs4mVz2Pm8otSlc916sxs9ooM5tdoNJorTTQ5SX2/crtsfP3ryKAFlkV2y1YG3/Bd5HbvwVu3nsNHDvNLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,672 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 hours, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.BasicRNNCell(size_layer)\n",
"\n",
" rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer = tf.placeholder(\n",
" tf.float32, (None, num_layers * size_layer)\n",
" )\n",
" self.outputs, self.last_state = tf.nn.dynamic_rnn(\n",
" drop, self.X, initial_state = self.hidden_layer, dtype = tf.float32\n",
" )\n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value = np.zeros((1, num_layers * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" ) \n",
" init_value = last_state\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value = np.zeros((1, num_layers * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value = last_state\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0812 22:23:09.113305 140115879184192 deprecation.py:323] From <ipython-input-6-35b560f11ea1>:12: BasicRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.SimpleRNNCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0812 22:23:09.116059 140115879184192 deprecation.py:323] From <ipython-input-6-35b560f11ea1>:16: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0812 22:23:09.445186 140115879184192 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0812 22:23:09.448605 140115879184192 deprecation.py:323] From <ipython-input-6-35b560f11ea1>:27: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0812 22:23:09.640925 140115879184192 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 22:23:09.647897 140115879184192 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:459: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0812 22:23:09.740828 140115879184192 deprecation.py:323] From <ipython-input-6-35b560f11ea1>:29: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [00:52<00:00, 5.77it/s, acc=76.9, cost=0.129] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:49<00:00, 6.03it/s, acc=78, cost=0.11] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:52<00:00, 5.77it/s, acc=73.7, cost=0.154] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:50<00:00, 6.11it/s, acc=82.7, cost=0.075] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:52<00:00, 5.65it/s, acc=76.3, cost=0.12] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:48<00:00, 6.15it/s, acc=77, cost=0.123] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:52<00:00, 5.65it/s, acc=80.2, cost=0.0896]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:50<00:00, 5.99it/s, acc=75.1, cost=0.15] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:52<00:00, 5.58it/s, acc=87.2, cost=0.0367]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [00:52<00:00, 5.69it/s, acc=76.7, cost=0.121] \n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdd3wU1fr48c9uNr0npBACJCQwQAICAqEIAoKKBctVFAXFq96vehWvBfVio1jAgih6EfVHUUAFlSIIYqFKhyAkwqRQQnrvbdvvj93EhARSSIXn/Xrlxe6cmTPPTIbNPnPOnKMxm80IIYQQQgghhGj/tK0dgBBCCCGEEEKIpiEJnhBCCCGEEEJcJiTBE0IIIYQQQojLhCR4QgghhBBCCHGZkARPCCGEEEIIIS4TkuAJIYQQQgghxGVCEjwhhBBCCCGEuEzoWjsAIYQQQjSOoigaYAbwf4AH8BPwL1VV863lE4H/AP2AA6qqjqpnvUuAh4DuqqrGVVl+L/A60AVIBaaqqrqryr5mAYHAOWCGqqrrqmzbDfgIuBYoA5aoqvpCow9eCCFEraQFTwghRJNTFOWKuoHYisf7ADAFGA4EAI7Awirl2cACYG59K1QU5RogpJbl44B5WBI/V2AkcMpa1glYATwLuAHTgVWKovhay+2AX4DfAX8sSeCK+h+mEEKI+rqi/gALIYQARVFeAh4FfLG0tLysqupaRVHsgTTgGlVVo6zr+gAJQFdVVdMVRbkFeAMIAv4CHlNV9Zh13TPAIuB+y1vFGXi+tn1Z17cB3gEeBAqA97EkJ7aqqhoURXEH5gM3ASZgKfC6qqrGWo5pMPAh0AsoAb4HnlVVtdxaHoYl0bka0AMfqqr6ljWGF4GHrTHGALcDNsDpilisdWwHVqiq+oWiKFOtx3UAS5K1SFGUpcDnwFWAGfgZ+LeqqrnW7TtbYxyB5Qbr11gSolTgWlVVj1vX8wXOWM95xkV/mXAr8P9UVT1n3XYe8LuiKI+rqlqsquqv1uWP1FFPxXnUYfkdPAj8eV7xLGC2qqr7rO+TqpQFArmqqm62vt+kKEoRlkQxHZgKJKuqOr/KNsfqE5MQQoiGkRY8IYS48sRjSTLcsXxpX6EoSkdVVcuAH4BJVdadCOywJnf9gSVYugN6A4uBDdbEsMIk4GbAw5oY1bov67qPAuOxdB8cgCWxqmoZYABCgf7A9cCFEhUj8AzQARgKXAc8AaAoiivwK7AFSytXKPCbdbtnrTHfhKXl6Z9A8QX2cb4ILC1YfsCbgAZ427qPXkBnYKY1BhtgI3AWS3LcCfjGmoB+A0yuUu8k4LeK5E5RlFxrq9qFaM57bQ90r+cxnO8ZYGdF0l7BGv9AwEdRlDhFURIVRflYURRH6yqHgBOKokxQFMVGUZTbsXTDrKhnCHBGUZTNiqJkKoqyXVGUPo2MUQghxEVIC54QQlxhVFVdU+Xtt4qi/BcYDKwHVmFJ3F62lt9nfQ/wL2Cxqqr7re+XK4oyA8uX9x3WZR9VtCbVY18TsbSkJQIoijIXS2KGoih+WJIuD1VVS4AiRVE+qIihlmM6XOXtGUVRFmN51msBcAuQqqrq+9byUqDiGB4BXlBVVbW+/9O6f9caJ66mZFVVK7pDGoA46w9AhqIo87E8r4b1mAOA6RUtgsBu67/LgTWKorykqqoZS5fLd6ocm8dFYtgCvKAoymogB0trJIBTPeKvxtrC+H9YWjnP5wfYAndhSdj1WH6Hr2BplTUqivIlluvHASgH7lZVtci6fSAwGpiAJbl+GlivKErPilZWIYQQTUMSPCGEuMIoivIAlparIOsiFywtXwDbACdFUSKwdNfsB6y1lnUFHlQU5akq1dlhSVwqnKvyuq59BZy3ftXXXbEkFCmKolQs055ff5X99MDSnXMgluRGB1QkfZ2xtCTW5mJldTn/WP34uwumqzXenCr7OVsluaukqup+RVGKgVGKoqRgaWHcUM8Ylljr3o7lmN/H0m0zsaEHgyUZnq2qal4tZSXWfxeqqpoCYE1gXwFeVhRlLJakdBRwBEuSuEFRlPGqqh61br+7ogunoijvWbftRc2uoEIIIS6BJHhCCHEFURSlK5bnxK4D9lpbXo5i7eZnfb8aSzfBNGCjqqoF1s3PAW+qqvrmRXZhru++gBQsLTsVOld5fQ5LF78OtSVFtVgERAKTVFUtUBTlP1hamyrquvcC253D8pxY1HnLK1qenIB862v/89Yxn/f+LeuyPqqqZlu7KX5cZT9dFEXRXeB4lmPpppkKfKeqaukF4q1GVVUTllbC1wEURbkey7NxSRfb7gKuA65RFOWdKsv2KorytKqqqxRFSaT6MVd93Q9L185D1vcHFUXZD4wFjmLpqjm8ETEJIYRoIEnwhBDiyuKM5Yt5xfNdDwHh562zClgHZPF3V02wJGtrFUX5FcvgIk5YWmx2VkkCG7Kv1cDTiqJswpJQVXQvRFXVFEVRtgLvK4ryKlAIBAOBqqruoCZXLIlYoaIoPYHHK/aL5dm3+dakbxGWVsfe1q6mXwBzFEX5C0v3yj5AkqqqGYqiJAGTrd09H6SWkSVriSEPyLOOKjm9StkBLAntXEVRXsfyzODVqqr+YS1fgaUlqwBLF816URTFC/DE8ixgLyytmLOtiV/Fs3O2WP7eaxVFcQCMqqrqa6muB9WfzU/B0hpY0cK2FHhKUZQtWLpoPoPl3AIcBF5SFKWfqqpHrc9rjgD+V+X4nrO29G0DpgGZwIn6HqsQQoj6kUFWhBDiCqKq6l9YuvHtxdJC1wf447x19mNJuAKAzVWWH8IyMMrHWLoexmEZHbGx+/oc2IqldScSyxxuBizJD1hGp7TDMlpnDvAd0JHaPY/lecECa73fVomjABiHJVlJBWKxPA8GloRotTWOfOD/YZlqAOuxTseS6IYBey50rFazsAwWkwdswjJgTUUMRuv+Q7GMSpoI3FOl/ByWro1mYFfVShVFKVQUZcQF9tkBy3krwvK7WqKq6mdVyqdg6R65CEvCVYLl/NSoW1XVdFVVUyt+rKtkWp+BBJiDJZGLwZKYRWIZXAZr0j0T+E5RlAIso5i+parqVmu5iqWF8lMsv8vbgAny/J0QQjQ9jdl8fg8TIYQQouUpijIe+FRV1a6tHUtrsE4unqyq6iutHYsQQoj2S7poCiGEaBXWIfZHY2k988PyHNnai250mVIUJQi4E8t0EEIIIUSjSRdNIYQQrUWDpVtjDpbufieA11o1olagKMocLIO8vKuq6unWjkcIIUT7Jl00hRBCCCGEEOIyIS14QgghhBBCCHGZaI/P4NkDg7AM32ysY10hhBBCCCGEuNzYYBlZ+iCWeWMrtccEbxDnDSEthBBCCCGEEFegEcDuqgvaY4KXApCTU4TJ1LaeH/T2diErq7C1wxBtgFwLooJcC6KCXAuiKrkeRAW5FkSFhlwLWq0GT09nsOZGVbXHBM8IYDKZ21yCB7TJmETrkGtBVJBrQVSQa0FUJdeDqCDXgqjQiGuhxiNrMsiKEEIIIYQQQlwmJMETQgghhBBCiMtEe+yiKYQQQgghhGhiRqOBnJwMDIby1g7lipSersVkMlVbptPZ4enpg41N/dM2SfCEEEIIIYQQ5ORk4ODghLOzPxqNprXDueLodFoMhr8TPLPZTFFRPjk5GXTo0LHe9UgXTSGEEEIIIQQGQznOzm6S3LURGo0GZ2e3BreoSoInhBBCCCGEAJDkro1pzO9DEjwhhBBCCCGEuExIgieEEEIIIYRoc3bu3M7999/FQw/dR0LCmdYOp4aCggJWrlx+wfLy8nKeffYpbr75Om6++boWi0sSPCGakMlsZslPJ/hhW1xrhyKEEEII0a6tX/8DDz/8GEuXrqJLl6B6b2c01pj7u9HM5gtPPF5YWMCqVV9esFyr1TJp0mQWLPhfk8VTHzKKphBN6Ke9Z9l9LIXImAyG9OyArc6mtUMSQgghhGh3PvrofY4diyQh4Sxr165h4cLF7Nu3h8WLP8ZkMuHh4cn06TMIDOzMkSOH+PDD91CUXsTEqDz66OP069efhQs/ID4+lvLycvr3H8hTTz2DjY0NGRnpLFjwLomJ5wAYO/YGpkx5iK1bt7Bmzdfo9XpMJiN33z2Jrl2DCQoK4rPPFnHkyEFsbe1wcnJk0aIlzJ8/j8LCQqZOvQ8HBwc+/XRJtWPQ6XQMGhRBSkpyi547SfCEaCIx53JZu+sUgT4uJGYUcjQui0E9fVs7LCGEEEKIBvvjeAq7j6U0S93X9O3I8D4XH/Z/2rTniIlRmTRpCsOHjyAnJ5s33niNhQs/Izi4Gxs3rmPWrFf4/HNLF8nTp08xffoMwsP7AjB37hz69RvASy+9islkYtasV9i0aQMTJtzB7NmvMnTocN58810AcnNzMZvN9O4dxssvzyQ/P5+UlCQWLvyAzz5bRnJyMpGRh1ixYg1arZb8/HwAnn32RR55ZArLlq1qlvPUWJLgCdEECkv0LN4QjY+HIy/d35/Xlhxkb1SqJHhCCCGEEE0gOjqKkJAeBAd3A+Cmmybw/vvzKC4uAiAwsHNlcgewe/dOTpyI5ptvVgJQWlqKr68fxcXFREUd44MPPsFsNlNcXExhYQGJieeIi4th06YN5OfnY29vR0FBPvb29gQGdsZgMDB37hwGDBjIsGEjWv4ENIAkeEJcIrPZzP/b+BcFxeW8PGUgTg62XDsgkA0748kvLsfNya61QxRCCCGEaJDhfepuZWtLHB2dzlti5q233qNTp8BqS4uLiwFITU0hPz+f8vIytFotbm7urFixjKeeeoZrrx2DyWRi7NhrKC8vx9u7A199tZrIyMMcOnSARYsWsmTJihY6soaTQVaEuES/HDzHn/FZTBwdSld/VwBGXx2I0WTm4In0Vo5OCCGEEKL9CwvrQ3x8DGfPngFg8+aNdO+u4OTkXOv6w4ePZMWLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,701 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.BasicRNNCell(size_layer)\n",
"\n",
" backward_rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" forward_rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop_backward = tf.contrib.rnn.DropoutWrapper(\n",
" backward_rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" forward_backward = tf.contrib.rnn.DropoutWrapper(\n",
" forward_rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.backward_hidden_layer = tf.placeholder(\n",
" tf.float32, shape = (None, num_layers * size_layer)\n",
" )\n",
" self.forward_hidden_layer = tf.placeholder(\n",
" tf.float32, shape = (None, num_layers * size_layer)\n",
" )\n",
" self.outputs, self.last_state = tf.nn.bidirectional_dynamic_rnn(\n",
" forward_backward,\n",
" drop_backward,\n",
" self.X,\n",
" initial_state_fw = self.forward_hidden_layer,\n",
" initial_state_bw = self.backward_hidden_layer,\n",
" dtype = tf.float32,\n",
" )\n",
" self.outputs = tf.concat(self.outputs, 2)\n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value_forward = np.zeros((1, num_layers * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" ) \n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value_forward = np.zeros((1, num_layers * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.backward_hidden_layer: init_value_backward,\n",
" modelnn.forward_hidden_layer: init_value_forward,\n",
" },\n",
" )\n",
" init_value_forward = last_state[0]\n",
" init_value_backward = last_state[1]\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0813 00:50:36.840563 140096227489600 deprecation.py:323] From <ipython-input-6-c6fe3802893b>:12: BasicRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.SimpleRNNCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0813 00:50:36.842919 140096227489600 deprecation.py:323] From <ipython-input-6-c6fe3802893b>:16: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0813 00:50:37.159364 140096227489600 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0813 00:50:37.164350 140096227489600 deprecation.py:323] From <ipython-input-6-c6fe3802893b>:42: bidirectional_dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.Bidirectional(keras.layers.RNN(cell))`, which is equivalent to this API\n",
"W0813 00:50:37.165004 140096227489600 deprecation.py:323] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn.py:464: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0813 00:50:37.355312 140096227489600 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0813 00:50:37.362000 140096227489600 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:459: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0813 00:50:37.520977 140096227489600 deprecation.py:323] From <ipython-input-6-c6fe3802893b>:45: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [01:04<00:00, 4.68it/s, acc=71.9, cost=0.169]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:06<00:00, 4.45it/s, acc=77.7, cost=0.11] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:04<00:00, 4.67it/s, acc=68.9, cost=0.211] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:06<00:00, 4.45it/s, acc=78.9, cost=0.104] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:06<00:00, 4.53it/s, acc=70.2, cost=0.193]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:06<00:00, 4.57it/s, acc=70.6, cost=0.189]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:07<00:00, 4.43it/s, acc=66.1, cost=0.253]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:05<00:00, 4.53it/s, acc=80.6, cost=0.0892]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:07<00:00, 4.51it/s, acc=63.5, cost=0.287] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:03<00:00, 4.71it/s, acc=72.8, cost=0.167]\n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdd1wc9534/9c2FpZd+lKFQAJp1BACoWa523F34hLHdqzYcRKnx5fcpV4ul8svl7sk3ziX5iQ+XxLLJS5yL3Hv6qIJIaQBFYooYum7CyywO78/ZpGRhCTKAov0fj4ePNidmZ35DAzLvufz/rw/Bk3TEEIIIYQQQggx+xlnugFCCCGEEEIIIUJDAjwhhBBCCCGEOEtIgCeEEEIIIYQQZwkJ8IQQQgghhBDiLCEBnhBCCCGEEEKcJSTAE0IIIYQQQoizhAR4QgghhBBCCHGWMM90A4QQQggxcYqifAP4ZyARqAa+qarq5uA6K/Bb4EbAAmwBvqyqauMp9nU98N9ANlABfEFV1aoR+/o5cCsQBTwO/JOqqoPBdX8ELgcSgIPAD1RVfTX42juAB0YcyhjcR5GqqiWh+UkIIYQA6cETQggxBRRFOaduIM7U+SqKsgY96PokEAv8BXhOURRTcJN/AtYBy4F0oBP4/Sn2tQB4DPgyEAe8BLw44ty+DxQBy4CFQCHwb8F1ZqABuCjYjn8DnlIUJRtAVdXHVFW1D38BXwUOAaWT/ykIIYQY6Zz6ByyEEAIURfk+cA+QjP6h/Ieqqj4X7IU5CpyvqmplcFsnUA9kqaraqijKdcB/ovfwVKH3BlUEt60F/gTcoT9VooFvj3as4PYm4JfAXYAbuA89+LCoqjqkKEos8GvgGiAA/A34saqq/lHOaTV6T9VioA94BvhnVVUHguuXAr8BVgKDwG9VVf2vYBu+B3w+2MZq4AbABBwebktwH+8Bj6qq+n+Konw2eF47gTuBPymK8jfgQSAf0IDXga+pqtoVfH1msI0XoN9gfRy9560FuEhV1T3B7ZKB2uDP3HXaX6b+e9g73AumKMrD6D1pyUAzMA94XVXVo8H1TwZ/pqO5EvhwRO/fL4B/Rw/a3gauB36hqmpHcP3vgF+g/068wH+M2NfLiqIcDv68a0c51l3Aw6qqamc4PyGEEOMkPXhCCHHuOYgeZMQCPwEeVRQlTVVVH/AscPuIbT8FvB8M7gqAvwJfQk8HfAC9h8c6YvvbgWuBuGBgNOqxgtveA1wNrEDvDbrhhHY+BAwBuUABcAXwhVOckx/4FpCE3mN1GXovEYqiOIC3gNfQe7Fy0QMW0AOs29GDyBjgc0DvKY5xojXovVApwM8AA3p6Yzp6oJlJMOgJBpIvA3XoQVkG8EQwAH0C2DBiv7cDbw8Hd4qidCmKcv4p2vAqYFIUZU3wGJ8DytGDRtB79NYripKuKIoNPfh+9TTnZDjhsQG9x+5U6+cEA/HjKIqSgt7Lt3eUdVnAhcDDp2mHEEKICZIePCGEOMeoqrppxNMnFUX5AbAaeAH4O3rg9sPg+k/z0dipLwIPqKq6I/h8o6Io/wqsBd4PLvudqqoNYzzWp9B70o4AKIryc/TAbDhAuAY9UOwDvIqi/M9wG0Y5p5HjuGoVRXkAvefpN8B1QIuqqvcF1/cDw+fwBeC7qqqqwee7g8d3nPSDO1mTqqrD6Y5DwIHgF4BLUZRfAz8OPl+NHvh9Z7hHENgc/L4R2KQoyveDPVqfQe/ZHD63uNO0wY3eW7kZPeDqAq4e0TNWg95z2ogeBO8Bvn6Kfb0F/EJRlIuBreg9mxGALbj+NeCfFEV5F72H897gchvQPbwTRVEs6KmeG1VV3T/Kce5E7yk8fJrzEkIIMUES4AkhxDlGUZQ70XuusoOL7Og9XwDvArbg2K6j6L1rzwXXZQF3BYt6DItAD1yGNYx4fKZjpZ+w/cjHWehFQZoVRRleZjxx/yOOsxA99bAIPeAwA8NBXyZ6T+JoTrfuTE481xQ+SsF0BNvbOeI4dSOCu2NUVd2hKEovcLGiKM3oPYwvjrENnwfuBpaiB5dXoKdHFqiq2gTcD1jRe1y9wHfRe/DWjNKO/Yqi3AX8AUgDHkVPwz0S3ORn6GPzygEfejpqAfp1MvwzMAKPAAOcOpC8E/ivMZ6fEEKIcZIATwghziHB9LgH0XvKtqmq6lcUpZxg6l3w+VPoaYJHgZdVVXUHX94A/ExV1Z+d5hDHxlSd6VjoY8TmjHht5ojHDehBRNJoQdEo/gSUAberqupWFOWb6IVHhvd12yle1wDkAJUnLPcGv9uAnuDj1BO2OXH82H8Fl+WpqtqhKMoN6MHS8HHmKopiPsX5bERP02wBnlZVtf8U7T3RCvTfUXXw+WvBIPE84Ong+h+OGDf3e+D/UxQlSVXVthN3pqrq08HXoShKHHoAuSu4rg89aPt6cP0XgRJVVQPB5wb0lNAU4BpVVQdP3L+iKOvRA/unx3h+QgghxkkCPCGEOLdEowchw+O77ub4MVagp2k+D7TzUaom6MHac4qivIVeXMQGXAx8MCIIHM+xnkJP+XsFPaD63vAKVVWbFUV5A7hPUZQfAR70giFzVFV9n5M50AMxj6Ioi4CvDB8Xfezbr4NB35/Qex2XBFNN/w/4qaIoVeg9YHlAo6qqLkVRGoENwXTPu9ADwdNxoKcqdiuKkgF8Z8S6negB7c8VRfkxerrkSlVVtwTXP4qeHupGT9Ecq13AD4OB22H0aQoW8lHAugu4M1ggphd9XGLTaMEdgKIoK9F76BLQe/9eHE6zDJ6TFjyPNcCP0APAYX9CH3t4eTAYHM1dwDOnuF6EEEKEgBRZEUKIc0hwTrP7gG3oPXR56HOjjdxmB3rAlc6IghyqqhajF0b5A3rq4QHgs5M41oPAG+jzrZUB/0AfyzZcJfNO9GCsKni8p9FTB0fzbfTxgu7gfp8c0Q438DH0KpAt6OPSLgmu/jV6oPkGeoD4F/T52Qie63fQA92l6OPSTucn6MViuoFX0AvWDLfBHzx+LnpV0iPo88kNr29AnzJAAz4cuVNFUTyKolxwimM+jF6k5b1g+38HfGnE2Ldvo485rEEPeK9BnxNveN+vBsdRDvst+jg+Ff1nfs+IdTnoPwMveo/j91VVfSO4nyz04jsrgJZgmz3B+e+GjxWJPu5y4ynORQghRAgYNE0qFAshhJh5iqJcDfxZVdWsmW7LTFAU5a/ovWv/dsaNhRBCiFOQFE0hhBAzQlGUKPSetDfQx239mI8KupxTghOC34RetEQIIYSYMEnRFEIIMVMM6GmNnegpmvvQJ9Y+pyiK8lP0MXP/T6YOEEIIMVmSoimEEEIIIYQQZwnpwRNCCCGEEEKIs8RsHINnBVahl2n2n2FbIYQQQgghhDjbmNArS+9Cnzf2mNkY4K3ihBLSQgghhBBCCHEOugDYPHLBbAzwmgE6O70EAuE1fjAx0U57u2emmyHCgFwLYphcC2KYXAtiJLkexDC5FsSw8VwLRqOB+PhoCMZGI83GAM8PEAhoYRfgAWHZJjEz5FoQw+RaEMPkWhAjyfUghsm1IIZN4Fo4aciaFFkRQgghhBBCiLPEGXvwFEX5FXAzkA3kqapaebrlwXULgY1AItAO3Kmqas2Z1gkhhBBCCCGEmLix9OA9D1wI1I1xOcCfgftVVV0I3A88MMZ1QgghhBBCCCEm6IwBnqqqm1VVbRjrckVRkoFC4PHgoseBQkVRnKdbN9ETEEIIIYQQQgihm4oiK5lAo6qqfgBVVf2KojQFlxtOs841noMkJtpD2+oQcTodM90EESbkWhDD5FoQw+RaECPJ9SCGybUghoXiWpiNVTQBaG/3hF3FIafTgcvlnulmiDAg14IYJteCGCbXghhJrgcxTK4FMWw814LRaDhlh9dUVNFsADIURTEBBL+nB5efbp0QQgghhBBCiEkIeYCnqmorUA7cHlx0O1CmqqrrdOtC3Q4hhBBCCCGEONecMcBTFOV3iqIcAeYAbymKsvd0y4O+DHxDUZRq4BvB52NZJ4QQQgghhBCnVNfi5qcbi+npHZjppoSlM47BU1X1XuDesS4PrtsPrBnvOiGEEEIIIYQ4nbdKGjjc3MOufa1ctnLOTDcn7EzFGDwhhBBCCCGECDnfoJ/i4Oiu4v2tM9ya8CQBnhBCCCGEEGJWKKt24Rvws2huHNUNXXR7JU3zRBLgCSGEEEIIIWaFrZUtJMZEcvvlC9GAUlV68U4kAZ4QQgghhBAi7HV5fOyt7WDdshTmOKNJTbAdS9cUH5EATwghhBBCCBH2tu89iqbBuqWpGAwGihYls7++kx5J0zyOBHhCCCGEEEKIsLdtbwvz0mJIS4wGoEhxomlQWiO9eCNJgCeEEEIIIYQIaw2tHhpaPZy3LPXYssxkOynxUVJN8wQS4AkhhBBCCCHC2rbKFkxGA6sXJx9bdixNs64Lt0x6fowEeEIIIYQQQoiwFQhobKtqYXlOIg5bxHHripRkAppGWU3bDLUu/EiAJ4QQQgghhAhbVXUddHsGWLc09aR1c1PsOOMi2SVpmsdIgCeEEEIIIYQIW1srW7BZzeTnJp20bjhNc19tJ56+wRloXfiRAE8IIYQQQggRlvp8Q5RWu1i9OBmLefTQ5ViaZrVU0wQJ8IQQQgghhBBhqrTaxcBggHXLTk7PHJad6iApNlImPQ+SAE8IIYQQQggRlrZWtuCMiyQ3I/aU2xgMBoqUZKpqO/D2S5qmBHhCCCGEEEKIsNPR08/+uk7WLU3FYDCcdtuiRcn4AxrlUk1TAjwhhBBCCCFE+NledRQNjpvc/FTmpTlIjLFKNU0kwBNCCCGEEEKEGU3T2FrZQm5GLMnxtjNubzAYWKkks/dwB739Q9PQwvAlAZ4QQgghhBAirNQf9dDU5h1T792wVcNpmgfO7WIrEuAJIYQQQgghwsqWymbMJgOrFieP+TXLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,739 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Split train and test\n",
"\n",
"I will cut the dataset to train and test datasets,\n",
"\n",
"1. Train dataset derived from starting timestamp until last 30 days\n",
"2. Test dataset derived from last 30 days until end of the dataset\n",
"\n",
"So we will let the model do forecasting based on last 30 days, and we will going to repeat the experiment for 10 times. You can increase it locally if you want, and tuning parameters will help you by a lot."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (222, 1), (30, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"test_size = 30\n",
"simulation_size = 10\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.BasicRNNCell(size_layer)\n",
" \n",
" with tf.variable_scope('forward', reuse = False):\n",
" rnn_cells_forward = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X_forward = tf.placeholder(tf.float32, (None, None, size))\n",
" drop_forward = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells_forward, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer_forward = tf.placeholder(\n",
" tf.float32, (None, num_layers * size_layer)\n",
" )\n",
" self.outputs_forward, self.last_state_forward = tf.nn.dynamic_rnn(\n",
" drop_forward,\n",
" self.X_forward,\n",
" initial_state = self.hidden_layer_forward,\n",
" dtype = tf.float32,\n",
" )\n",
"\n",
" with tf.variable_scope('backward', reuse = False):\n",
" rnn_cells_backward = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X_backward = tf.placeholder(tf.float32, (None, None, size))\n",
" drop_backward = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells_backward, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer_backward = tf.placeholder(\n",
" tf.float32, (None, num_layers * size_layer)\n",
" )\n",
" self.outputs_backward, self.last_state_backward = tf.nn.dynamic_rnn(\n",
" drop_backward,\n",
" self.X_backward,\n",
" initial_state = self.hidden_layer_backward,\n",
" dtype = tf.float32,\n",
" )\n",
"\n",
" self.outputs = self.outputs_backward - self.outputs_forward\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"future_day = test_size\n",
"learning_rate = 0.01"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value_forward = np.zeros((1, num_layers * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x_forward = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_x_backward = np.expand_dims(\n",
" np.flip(df_train.iloc[k : index, :].values, axis = 0), axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state_forward, last_state_backward, _, loss = sess.run(\n",
" [\n",
" modelnn.logits,\n",
" modelnn.last_state_forward,\n",
" modelnn.last_state_backward,\n",
" modelnn.optimizer,\n",
" modelnn.cost,\n",
" ],\n",
" feed_dict = {\n",
" modelnn.X_forward: batch_x_forward,\n",
" modelnn.X_backward: batch_x_backward,\n",
" modelnn.Y: batch_y,\n",
" modelnn.hidden_layer_forward: init_value_forward,\n",
" modelnn.hidden_layer_backward: init_value_backward,\n",
" },\n",
" )\n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value_forward = np.zeros((1, num_layers * size_layer))\n",
" init_value_backward = np.zeros((1, num_layers * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" batch_x_forward = np.expand_dims(\n",
" df_train.iloc[k : k + timestamp, :], axis = 0\n",
" )\n",
" batch_x_backward = np.expand_dims(\n",
" np.flip(df_train.iloc[k : k + timestamp, :].values, axis = 0), axis = 0\n",
" )\n",
" out_logits, last_state_forward, last_state_backward = sess.run(\n",
" [\n",
" modelnn.logits,\n",
" modelnn.last_state_forward,\n",
" modelnn.last_state_backward,\n",
" ],\n",
" feed_dict = {\n",
" modelnn.X_forward: batch_x_forward,\n",
" modelnn.X_backward: batch_x_backward,\n",
" modelnn.hidden_layer_forward: init_value_forward,\n",
" modelnn.hidden_layer_backward: init_value_backward,\n",
" },\n",
" )\n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" output_predict[k + 1 : k + timestamp + 1, :] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" batch_x_forward = np.expand_dims(df_train.iloc[upper_b:, :], axis = 0)\n",
" batch_x_backward = np.expand_dims(\n",
" np.flip(df_train.iloc[upper_b:, :].values, axis = 0), axis = 0\n",
" )\n",
" out_logits, last_state_forward, last_state_backward = sess.run(\n",
" [modelnn.logits, modelnn.last_state_forward, modelnn.last_state_backward],\n",
" feed_dict = {\n",
" modelnn.X_forward: batch_x_forward,\n",
" modelnn.X_backward: batch_x_backward,\n",
" modelnn.hidden_layer_forward: init_value_forward,\n",
" modelnn.hidden_layer_backward: init_value_backward,\n",
" },\n",
" )\n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" o_f = np.flip(o, axis = 0)\n",
" out_logits, last_state_forward, last_state_backward = sess.run(\n",
" [\n",
" modelnn.logits,\n",
" modelnn.last_state_forward,\n",
" modelnn.last_state_backward,\n",
" ],\n",
" feed_dict = {\n",
" modelnn.X_forward: np.expand_dims(o, axis = 0),\n",
" modelnn.X_backward: np.expand_dims(o_f, axis = 0),\n",
" modelnn.hidden_layer_forward: init_value_forward,\n",
" modelnn.hidden_layer_backward: init_value_backward,\n",
" },\n",
" )\n",
" init_value_forward = last_state_forward\n",
" init_value_backward = last_state_backward\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.3)\n",
" \n",
" return deep_future[-test_size:]"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0813 01:22:40.361487 140690772289344 deprecation.py:323] From <ipython-input-6-04b2b1d463f4>:12: BasicRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.SimpleRNNCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0813 01:22:40.364087 140690772289344 deprecation.py:323] From <ipython-input-6-04b2b1d463f4>:17: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0813 01:22:40.688215 140690772289344 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0813 01:22:40.691791 140690772289344 deprecation.py:323] From <ipython-input-6-04b2b1d463f4>:30: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0813 01:22:40.883351 140690772289344 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0813 01:22:40.890208 140690772289344 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:459: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0813 01:22:41.052329 140690772289344 deprecation.py:323] From <ipython-input-6-04b2b1d463f4>:54: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [01:08<00:00, 4.43it/s, acc=73.8, cost=0.155] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:09<00:00, 4.32it/s, acc=75, cost=0.151] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:08<00:00, 4.41it/s, acc=75.2, cost=0.14] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:08<00:00, 4.36it/s, acc=71, cost=0.186] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:09<00:00, 4.33it/s, acc=84, cost=0.0574] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:09<00:00, 4.34it/s, acc=72.3, cost=0.166] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:08<00:00, 4.44it/s, acc=80.4, cost=0.0918]\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:07<00:00, 4.45it/s, acc=79.7, cost=0.101] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:05<00:00, 4.61it/s, acc=80.6, cost=0.088] \n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:08<00:00, 4.40it/s, acc=70.8, cost=0.194] \n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdd3hcV5n48e9UaWbURxr1bum6yXGN7diJ4/QG/AiBJBAIsAuEuuwuuxsWdmHpLBCWLAnJUgMJSYAQCCGkOMXdsRPLtmzLV7Z6H2k00qiMpCn398eMbEmWbJVR9ft5Hj0zc8u558rXmvvec857dJqmIYQQQgghhBBi4dPPdQWEEEIIIYQQQkSGBHhCCCGEEEIIsUhIgCeEEEIIIYQQi4QEeEIIIYQQQgixSEiAJ4QQQgghhBCLhAR4QgghhBBCCLFISIAnhBBCCCGEEIuEca4rIIQQQoipURRFB/w78AkgAXgB+Liqqp7w+vcBnwdWAwdVVb36AmVdDbwG9A1b/GlVVR8btV0RUAb8QVXVe4Ytfz/wbSAZeAX4qKqqHcPW3wV8BcgBWoAPq6q6e0onLoQQYlzSgieEECLiFEW5pB4gzuH5fgj4ILAFyAAswP8OW98B/A/wnQmW16Sqasywn8fG2OYh4NDwBYqirAAeDdcllVCQ+PCw9dcD3wU+AsQCVwFVE6yTEEKISbikvoCFEEKAoij3Ax8DHEA98CVVVZ9VFCUKaAW2qqp6PLxtClAH5Kqq6lQU5TbgG0AecBK4T1XVY+Fta4CfAB8IfVRswBfGOlZ4ewPw38C9QDfwA0LBiUlVVb+iKPHAA8AtQBD4JfAVVVUDY5zT5cCPgGWAF3gG+CdVVQfD61cQCnTWAT7gR6qqfitch38D/i5cxwrg/wEGoHqoLuEy3gAeV1X1Z4qifDh8XgcJBVk/URTll8BPgcsADXiJUAtYZ3j/7HAdryT0gPVJ4J8ItWZtU1W1LLydA6gJ/87bLviPCe8Afq6qan143+8CrymK8klVVftUVd0RXv73FylnQsKtcJ3APmDJsFUfAP6iququ8Hb/AZQrihKrqmo38F/A11RVPRDevjES9RFCCHE+acETQohLTyWhICOe0I3344qipKuqOgD8Ebh72LbvA3aGg7s1wC8IdQe0E2qxeS4cGA65G7gVSAgHRmMeK7ztx4CbCXUfXEsosBruV4CfUCCxBrgBGC9QCQD/SKh74GbgWuBTAIqixAI7gBcJtXItAV4N7/dP4TrfAsQBH2VkF8UL2UioFSoV+CagI9RFMYNQoJkNfDVcBwPwPFBLKDjOBJ4KB6BPAfcMK/du4NWh4E5RlE5FUbZeoB66Ue+jgKIJnsNoDkVRWhVFqVYU5YfhIJ1wPeKArxH6nY22Ajg69EFV1UpgECgOn/t6IEVRlDOKojQoivJjRVEsU6yjEEKIC5AWPCGEuMSoqvr7YR+fVhTli8DlwJ+B3xIK3L4UXv/+8GeAjwOPqqr6ZvjzY4qi/DuwCdgZXvbgUGvSBI71PkItaQ0AiqJ8h1BghqIoqYSCrgRVVb1Ar6IoPxyqwxjn9PawjzWKojwKbCPUancb0KKq6g/C6/uBoXP4e+BfVVVVw5+Pho8fe94v7nxNqqoOdYf0A2fCPwBtiqI8QGjMGeFzzgD+ZahFENgTfn0M+L2iKPerqqoR6ub438POLeECdXgR+FdFUX4HuAm1RgJYJ1D/0U4RCrZPAbnhej1AKKAH+Dqh1sIGRVFG7xsDdI1a1kWoO2YqYALuIBTs+wj9+3+Zc9eZEEKICJEATwghLjGKonyIUCtMXnhRDKGWL4DXAauiKBsJdddcDTwbXpcL3KsoymeHFWcmFLgMqR/2/mLHyhi1/fD3uYSCguZhwYR+dPnDjlNMKBhZTyi4MQJDQV82oZbEsVxo3cWMPtdUznXBjA3X1z3sOLXDgruzVFV9U1GUPuBqRVGaCbUwPjfBOvwiXPYbhM75B4S6bTZM9mRUVW0h1F0UoFpRlH8l1Or4CUVRVgPXEWpJHUsPoRbQ4eIIdb31hj//r6qqzQDh4FcCPCGEmAES4AkhxCVEUZRcQuPErgX2q6oaUBTlCOFufuHPvyPUTbAVeD48hgpCAc03VVX95gUOoU30WEAzkDVs3+xh7+uBASB5rKBoDD8BSoG7VVXtVhTl84RajIbKumuc/eqBQuD4qOW94Vcr4Am/Txu1jTbq87fCy0pUVe1QFOX/AT8edpwcRVGM45zPY4S6abYQyk7ZP059R1BVNUiolfArAIqi3EBofFskxrhpnBvKcTWhIL0uHHDHAAZFUZarqroWOEFo7CHhehQQ6ipaEf73aGDk72v0704IIUSESIAnhBCXFhuhm+uh8V0fAVaO2ua3wJ8AFyNbWH4KPKsoyg5CyUWshG78dw0LAidzrN8B/6Aoyl8JBVRD3QtRVbVZUZSXgR+EE3b0APlAlqqqOzlfLKFArEdRlKXAJ4eOS6gV6oFw0PcTQq2Oy8NdTX8GfF1RlJOEuleWAI2qqrYpitII3BPu7nkvoUDwQmIJdUvsUhQlE/iXYesOEgpov6MoylcIjRlcp6rq3vD6xwl1D+0m1EVzQhRFSQISCY0FXEaoFfNr4cBvaOyfidD3vV5RlGggoKqqb4yytofLqSMUeH+HUFdKgP8jNFZwyBcIBXyfDH9+AtivKMqVwGFCY/X+OOy6+CXwWUVRXiTURfMfCf27CCGEiDBJsiKEEJcQVVVPEurGt59QC10JsHfUNm8SCrgygL8NW/4WocQoPybU9fAM8OFpHOunwMvAMUKtby8QGss2lCXzQ4SCsZPh4/0BSGdsXyA0XrA7XO7Tw+rRDVxPqOtiC3Aa2B5e/QChQPNlQgHizwlNNUD4XP+FUKC7glDmyAv5L0LJYrqAvxJKWDNUh0D4+EsIBVANwJ3D1tcTCow0YMTccIqi9IQDp7EkE/q99RL6t/qFqqr/N2z9Bwl1kfwJoa6jXkK/n7HKXhM+x97waxnwuXD9+lRVbRn6IRRw9w8lglFV9QRwH6FAz0ko2P3UsHp8ndDUChVAOaF/7wu1BAshhJginaZJLwkhhBBzT1GUm4FHVFXNneu6zAVFUX5BKHHLl+e6LkIIIRYu6aIphBBiToTT5G8n1HqWSmgc2bMX3GmRUhQlD7id8ZOYCCGEEBMiXTSFEELMFR2hbo1uQl32yoH/nNMazQFFUb5OKMnL91RVrZ7r+gghhFjYpIumEEIIIYQQQiwSC7EFz0goc5d0LxVCCCGEEEJcisaNiRZikJRLKHPblUxhIlchhBBCCCGEWOCyCGVdXgJUDl+xEAO8oRTZuy+4lRBCCCGEEEIsbuksggCvGcDt7iUYnF/jB+32GFyunrmuhpgH5FoQQ+RaEEPkWhDDyfUghsi1IIZM5lrQ63UkJtogHBsNtxADvABAMKjNuwAPmJd1EnNDrgUxRK4FMUSuBTGcXA9iiFwLYsgUroXA6AULMcmKEEIIIYQQQogxSIAnhBBCCCGEEIuEBHhCCCGEEEIIsUhIgCeEEEIIIYQQi4QEeEIIIYQQQgixSEiAJ4QQQgghhBCLhAR4QgghhBBCCLFISIAnhBBCCCGEEIuEBHhCiAvqbGtix5MPMODtneuqCCGEEEKIi5AATwhxQQ1njuJqrqGlpnyuqyKEEEIIIS5CAjwhxAW5mmsAaK2vmNN6CCGEEEKIi5MATwgxLk3T6GiuBcBZf3qOayOEEEIIIS5GAjwhxLh6OtsYHOgjISWT3i4XPV2uua6SEBFV39NPeWfPXFdDCCGEiBgJ8IQQ4xrqnrlsw3UAOKWbplhk/lrfxm/PtNA54JvrqgghhBARIQGeEGJcruYajKYosopXE2WNlW6aYlEZCARp6O0noGnsaJLWaSGEEIuDBHhCiHG5WmpJSstBrzfgyC7CWXcaTdPmulpCRERNt5egBlm2KErbu2n1Dsx1lYQQQohpM05kI0VRvg+8B8gDSlRVPR5eXgw8BtgBF/AhVVVPh9fVAP3hH4B/U1X1pfC6TcCjgAWoAe5RVdUZiRMSQkSG3zdIZ1sjS9dfC0BqdhH16mG63U7iklLnuHZCTF9VtxeDTsf7C9P50Yk6Xm5w8cGijLmulhBCCDEtE23B+xNwFVA7avkjwEOqqhYDDxEK2oa7Q1XV1eGfoeBODzwOfDq83y7gO1M9ASHEzHA7G9CCQexpuQA4sosBGYcnFo8qTx85MdEkRJm4Ki2R8s5earu9c10tIYQQYlomFOCpqrpHVdX64csURXEAa4Enw4ueBNYqipJykeLWAf2qqu4Jf34EeN/EqyyEmA0dLTUAJKXnARCTkIw1NlHG4YlFwesP0NQ3QEGsBYAtqQnEGA281NAu3ZCFEEIsaNMZg5cNNKqqGgAIvzaFlw95QlGUY4qiPKwoSkJ4WQ7DWgJVVW0H9IqiJE2jLkKICHM112CNS8JiiwNAp9OFxuHVn0bTgnNbOSGmqbrbiwYUxFkBMBv0XJOZRE1PPxVdfXNbOSGEEGIaJjQGb4quVFW1XlGUKOB/gB8D90SqcLs9JlJFRVRKSuxcV0HMEwv9Wuh01pORWzjiPAqXl1Bz8iD6gIfk9OwL7C2GW+jXwmLU5OzErNexNi8Zoz70rPNmewz72zy82uLmiiWp6HW6iB9XrgUxnFwPYohcC5Mz6B/kraZjbMpei163uHJGRuJamE6AVw9kKopiUFU1oCiKAcgIL2eoS6eqqgOKojwMPBferw7IHSpEUZRkIKiqasdkDu5y9RAMzq9uNCkpsbS1dc91NcQ8sNCvBW9PF92dLgovu2rEeVgSQkHdqWOlKMaE8XYXwyz0a2GxOuHsIjfGgtvVO2L5NWmJPF3VwqtqE6vtcRE9plwLYji5HsQQuRYm783mt/l1+dPc1xugJHn5XFcnYiZzLej1unEbvKYc8oazXh4B7g4vuhsoVVWLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].iloc[-test_size:].values, r) for r in results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'].iloc[-test_size:].values, label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,318 @@
# Copyright 2017 Google Inc.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# ==============================================================================
"""DNC access modules."""
from __future__ import absolute_import
from __future__ import division
from __future__ import print_function
import collections
import sonnet as snt
import tensorflow as tf
import addressing
import util
AccessState = collections.namedtuple('AccessState', (
'memory', 'read_weights', 'write_weights', 'linkage', 'usage'))
def _erase_and_write(memory, address, reset_weights, values):
"""Module to erase and write in the external memory.
Erase operation:
M_t'(i) = M_{t-1}(i) * (1 - w_t(i) * e_t)
Add operation:
M_t(i) = M_t'(i) + w_t(i) * a_t
where e are the reset_weights, w the write weights and a the values.
Args:
memory: 3-D tensor of shape `[batch_size, memory_size, word_size]`.
address: 3-D tensor `[batch_size, num_writes, memory_size]`.
reset_weights: 3-D tensor `[batch_size, num_writes, word_size]`.
values: 3-D tensor `[batch_size, num_writes, word_size]`.
Returns:
3-D tensor of shape `[batch_size, num_writes, word_size]`.
"""
with tf.name_scope('erase_memory', values=[memory, address, reset_weights]):
expand_address = tf.expand_dims(address, 3)
reset_weights = tf.expand_dims(reset_weights, 2)
weighted_resets = expand_address * reset_weights
reset_gate = tf.reduce_prod(1 - weighted_resets, [1])
memory *= reset_gate
with tf.name_scope('additive_write', values=[memory, address, values]):
add_matrix = tf.matmul(address, values, adjoint_a=True)
memory += add_matrix
return memory
class MemoryAccess(snt.RNNCore):
"""Access module of the Differentiable Neural Computer.
This memory module supports multiple read and write heads. It makes use of:
* `addressing.TemporalLinkage` to track the temporal ordering of writes in
memory for each write head.
* `addressing.FreenessAllocator` for keeping track of memory usage, where
usage increase when a memory location is written to, and decreases when
memory is read from that the controller says can be freed.
Write-address selection is done by an interpolation between content-based
lookup and using unused memory.
Read-address selection is done by an interpolation of content-based lookup
and following the link graph in the forward or backwards read direction.
"""
def __init__(self,
memory_size=128,
word_size=20,
num_reads=1,
num_writes=1,
name='memory_access'):
"""Creates a MemoryAccess module.
Args:
memory_size: The number of memory slots (N in the DNC paper).
word_size: The width of each memory slot (W in the DNC paper)
num_reads: The number of read heads (R in the DNC paper).
num_writes: The number of write heads (fixed at 1 in the paper).
name: The name of the module.
"""
super(MemoryAccess, self).__init__(name=name)
self._memory_size = memory_size
self._word_size = word_size
self._num_reads = num_reads
self._num_writes = num_writes
self._write_content_weights_mod = addressing.CosineWeights(
num_writes, word_size, name='write_content_weights')
self._read_content_weights_mod = addressing.CosineWeights(
num_reads, word_size, name='read_content_weights')
self._linkage = addressing.TemporalLinkage(memory_size, num_writes)
self._freeness = addressing.Freeness(memory_size)
def _build(self, inputs, prev_state):
"""Connects the MemoryAccess module into the graph.
Args:
inputs: tensor of shape `[batch_size, input_size]`. This is used to
control this access module.
prev_state: Instance of `AccessState` containing the previous state.
Returns:
A tuple `(output, next_state)`, where `output` is a tensor of shape
`[batch_size, num_reads, word_size]`, and `next_state` is the new
`AccessState` named tuple at the current time t.
"""
inputs = self._read_inputs(inputs)
# Update usage using inputs['free_gate'] and previous read & write weights.
usage = self._freeness(
write_weights=prev_state.write_weights,
free_gate=inputs['free_gate'],
read_weights=prev_state.read_weights,
prev_usage=prev_state.usage)
# Write to memory.
write_weights = self._write_weights(inputs, prev_state.memory, usage)
memory = _erase_and_write(
prev_state.memory,
address=write_weights,
reset_weights=inputs['erase_vectors'],
values=inputs['write_vectors'])
linkage_state = self._linkage(write_weights, prev_state.linkage)
# Read from memory.
read_weights = self._read_weights(
inputs,
memory=memory,
prev_read_weights=prev_state.read_weights,
link=linkage_state.link)
read_words = tf.matmul(read_weights, memory)
return (read_words, AccessState(
memory=memory,
read_weights=read_weights,
write_weights=write_weights,
linkage=linkage_state,
usage=usage))
def _read_inputs(self, inputs):
"""Applies transformations to `inputs` to get control for this module."""
def _linear(first_dim, second_dim, name, activation=None):
"""Returns a linear transformation of `inputs`, followed by a reshape."""
linear = snt.Linear(first_dim * second_dim, name=name)(inputs)
if activation is not None:
linear = activation(linear, name=name + '_activation')
return tf.reshape(linear, [-1, first_dim, second_dim])
# v_t^i - The vectors to write to memory, for each write head `i`.
write_vectors = _linear(self._num_writes, self._word_size, 'write_vectors')
# e_t^i - Amount to erase the memory by before writing, for each write head.
erase_vectors = _linear(self._num_writes, self._word_size, 'erase_vectors',
tf.sigmoid)
# f_t^j - Amount that the memory at the locations read from at the previous
# time step can be declared unused, for each read head `j`.
free_gate = tf.sigmoid(
snt.Linear(self._num_reads, name='free_gate')(inputs))
# g_t^{a, i} - Interpolation between writing to unallocated memory and
# content-based lookup, for each write head `i`. Note: `a` is simply used to
# identify this gate with allocation vs writing (as defined below).
allocation_gate = tf.sigmoid(
snt.Linear(self._num_writes, name='allocation_gate')(inputs))
# g_t^{w, i} - Overall gating of write amount for each write head.
write_gate = tf.sigmoid(
snt.Linear(self._num_writes, name='write_gate')(inputs))
# \pi_t^j - Mixing between "backwards" and "forwards" positions (for
# each write head), and content-based lookup, for each read head.
num_read_modes = 1 + 2 * self._num_writes
read_mode = snt.BatchApply(tf.nn.softmax)(
_linear(self._num_reads, num_read_modes, name='read_mode'))
# Parameters for the (read / write) "weights by content matching" modules.
write_keys = _linear(self._num_writes, self._word_size, 'write_keys')
write_strengths = snt.Linear(self._num_writes, name='write_strengths')(
inputs)
read_keys = _linear(self._num_reads, self._word_size, 'read_keys')
read_strengths = snt.Linear(self._num_reads, name='read_strengths')(inputs)
result = {
'read_content_keys': read_keys,
'read_content_strengths': read_strengths,
'write_content_keys': write_keys,
'write_content_strengths': write_strengths,
'write_vectors': write_vectors,
'erase_vectors': erase_vectors,
'free_gate': free_gate,
'allocation_gate': allocation_gate,
'write_gate': write_gate,
'read_mode': read_mode,
}
return result
def _write_weights(self, inputs, memory, usage):
"""Calculates the memory locations to write to.
This uses a combination of content-based lookup and finding an unused
location in memory, for each write head.
Args:
inputs: Collection of inputs to the access module, including controls for
how to chose memory writing, such as the content to look-up and the
weighting between content-based and allocation-based addressing.
memory: A tensor of shape `[batch_size, memory_size, word_size]`
containing the current memory contents.
usage: Current memory usage, which is a tensor of shape `[batch_size,
memory_size]`, used for allocation-based addressing.
Returns:
tensor of shape `[batch_size, num_writes, memory_size]` indicating where
to write to (if anywhere) for each write head.
"""
with tf.name_scope('write_weights', values=[inputs, memory, usage]):
# c_t^{w, i} - The content-based weights for each write head.
write_content_weights = self._write_content_weights_mod(
memory, inputs['write_content_keys'],
inputs['write_content_strengths'])
# a_t^i - The allocation weights for each write head.
write_allocation_weights = self._freeness.write_allocation_weights(
usage=usage,
write_gates=(inputs['allocation_gate'] * inputs['write_gate']),
num_writes=self._num_writes)
# Expands gates over memory locations.
allocation_gate = tf.expand_dims(inputs['allocation_gate'], -1)
write_gate = tf.expand_dims(inputs['write_gate'], -1)
# w_t^{w, i} - The write weightings for each write head.
return write_gate * (allocation_gate * write_allocation_weights +
(1 - allocation_gate) * write_content_weights)
def _read_weights(self, inputs, memory, prev_read_weights, link):
"""Calculates read weights for each read head.
The read weights are a combination of following the link graphs in the
forward or backward directions from the previous read position, and doing
content-based lookup. The interpolation between these different modes is
done by `inputs['read_mode']`.
Args:
inputs: Controls for this access module. This contains the content-based
keys to lookup, and the weightings for the different read modes.
memory: A tensor of shape `[batch_size, memory_size, word_size]`
containing the current memory contents to do content-based lookup.
prev_read_weights: A tensor of shape `[batch_size, num_reads,
memory_size]` containing the previous read locations.
link: A tensor of shape `[batch_size, num_writes, memory_size,
memory_size]` containing the temporal write transition graphs.
Returns:
A tensor of shape `[batch_size, num_reads, memory_size]` containing the
read weights for each read head.
"""
with tf.name_scope(
'read_weights', values=[inputs, memory, prev_read_weights, link]):
# c_t^{r, i} - The content weightings for each read head.
content_weights = self._read_content_weights_mod(
memory, inputs['read_content_keys'], inputs['read_content_strengths'])
# Calculates f_t^i and b_t^i.
forward_weights = self._linkage.directional_read_weights(
link, prev_read_weights, forward=True)
backward_weights = self._linkage.directional_read_weights(
link, prev_read_weights, forward=False)
backward_mode = inputs['read_mode'][:, :, :self._num_writes]
forward_mode = (
inputs['read_mode'][:, :, self._num_writes:2 * self._num_writes])
content_mode = inputs['read_mode'][:, :, 2 * self._num_writes]
read_weights = (
tf.expand_dims(content_mode, 2) * content_weights + tf.reduce_sum(
tf.expand_dims(forward_mode, 3) * forward_weights, 2) +
tf.reduce_sum(tf.expand_dims(backward_mode, 3) * backward_weights, 2))
return read_weights
@property
def state_size(self):
"""Returns a tuple of the shape of the state tensors."""
return AccessState(
memory=tf.TensorShape([self._memory_size, self._word_size]),
read_weights=tf.TensorShape([self._num_reads, self._memory_size]),
write_weights=tf.TensorShape([self._num_writes, self._memory_size]),
linkage=self._linkage.state_size,
usage=self._freeness.state_size)
@property
def output_size(self):
"""Returns the output shape."""
return tf.TensorShape([self._num_reads, self._word_size])
@@ -0,0 +1,410 @@
# Copyright 2017 Google Inc.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# ==============================================================================
"""DNC addressing modules."""
from __future__ import absolute_import
from __future__ import division
from __future__ import print_function
import collections
import sonnet as snt
import tensorflow as tf
import util
# Ensure values are greater than epsilon to avoid numerical instability.
_EPSILON = 1e-6
TemporalLinkageState = collections.namedtuple('TemporalLinkageState',
('link', 'precedence_weights'))
def _vector_norms(m):
squared_norms = tf.reduce_sum(m * m, axis=2, keep_dims=True)
return tf.sqrt(squared_norms + _EPSILON)
def weighted_softmax(activations, strengths, strengths_op):
"""Returns softmax over activations multiplied by positive strengths.
Args:
activations: A tensor of shape `[batch_size, num_heads, memory_size]`, of
activations to be transformed. Softmax is taken over the last dimension.
strengths: A tensor of shape `[batch_size, num_heads]` containing strengths to
multiply by the activations prior to the softmax.
strengths_op: An operation to transform strengths before softmax.
Returns:
A tensor of same shape as `activations` with weighted softmax applied.
"""
transformed_strengths = tf.expand_dims(strengths_op(strengths), -1)
sharp_activations = activations * transformed_strengths
softmax = snt.BatchApply(module_or_op=tf.nn.softmax)
return softmax(sharp_activations)
class CosineWeights(snt.AbstractModule):
"""Cosine-weighted attention.
Calculates the cosine similarity between a query and each word in memory, then
applies a weighted softmax to return a sharp distribution.
"""
def __init__(self,
num_heads,
word_size,
strength_op=tf.nn.softplus,
name='cosine_weights'):
"""Initializes the CosineWeights module.
Args:
num_heads: number of memory heads.
word_size: memory word size.
strength_op: operation to apply to strengths (default is tf.nn.softplus).
name: module name (default 'cosine_weights')
"""
super(CosineWeights, self).__init__(name=name)
self._num_heads = num_heads
self._word_size = word_size
self._strength_op = strength_op
def _build(self, memory, keys, strengths):
"""Connects the CosineWeights module into the graph.
Args:
memory: A 3-D tensor of shape `[batch_size, memory_size, word_size]`.
keys: A 3-D tensor of shape `[batch_size, num_heads, word_size]`.
strengths: A 2-D tensor of shape `[batch_size, num_heads]`.
Returns:
Weights tensor of shape `[batch_size, num_heads, memory_size]`.
"""
# Calculates the inner product between the query vector and words in memory.
dot = tf.matmul(keys, memory, adjoint_b=True)
# Outer product to compute denominator (euclidean norm of query and memory).
memory_norms = _vector_norms(memory)
key_norms = _vector_norms(keys)
norm = tf.matmul(key_norms, memory_norms, adjoint_b=True)
# Calculates cosine similarity between the query vector and words in memory.
similarity = dot / (norm + _EPSILON)
return weighted_softmax(similarity, strengths, self._strength_op)
class TemporalLinkage(snt.RNNCore):
"""Keeps track of write order for forward and backward addressing.
This is a pseudo-RNNCore module, whose state is a pair `(link,
precedence_weights)`, where `link` is a (collection of) graphs for (possibly
multiple) write heads (represented by a tensor with values in the range
[0, 1]), and `precedence_weights` records the "previous write locations" used
to build the link graphs.
The function `directional_read_weights` computes addresses following the
forward and backward directions in the link graphs.
"""
def __init__(self, memory_size, num_writes, name='temporal_linkage'):
"""Construct a TemporalLinkage module.
Args:
memory_size: The number of memory slots.
num_writes: The number of write heads.
name: Name of the module.
"""
super(TemporalLinkage, self).__init__(name=name)
self._memory_size = memory_size
self._num_writes = num_writes
def _build(self, write_weights, prev_state):
"""Calculate the updated linkage state given the write weights.
Args:
write_weights: A tensor of shape `[batch_size, num_writes, memory_size]`
containing the memory addresses of the different write heads.
prev_state: `TemporalLinkageState` tuple containg a tensor `link` of
shape `[batch_size, num_writes, memory_size, memory_size]`, and a
tensor `precedence_weights` of shape `[batch_size, num_writes,
memory_size]` containing the aggregated history of recent writes.
Returns:
A `TemporalLinkageState` tuple `next_state`, which contains the updated
link and precedence weights.
"""
link = self._link(prev_state.link, prev_state.precedence_weights,
write_weights)
precedence_weights = self._precedence_weights(prev_state.precedence_weights,
write_weights)
return TemporalLinkageState(
link=link, precedence_weights=precedence_weights)
def directional_read_weights(self, link, prev_read_weights, forward):
"""Calculates the forward or the backward read weights.
For each read head (at a given address), there are `num_writes` link graphs
to follow. Thus this function computes a read address for each of the
`num_reads * num_writes` pairs of read and write heads.
Args:
link: tensor of shape `[batch_size, num_writes, memory_size,
memory_size]` representing the link graphs L_t.
prev_read_weights: tensor of shape `[batch_size, num_reads,
memory_size]` containing the previous read weights w_{t-1}^r.
forward: Boolean indicating whether to follow the "future" direction in
the link graph (True) or the "past" direction (False).
Returns:
tensor of shape `[batch_size, num_reads, num_writes, memory_size]`
"""
with tf.name_scope('directional_read_weights'):
# We calculate the forward and backward directions for each pair of
# read and write heads; hence we need to tile the read weights and do a
# sort of "outer product" to get this.
expanded_read_weights = tf.stack([prev_read_weights] * self._num_writes,
1)
result = tf.matmul(expanded_read_weights, link, adjoint_b=forward)
# Swap dimensions 1, 2 so order is [batch, reads, writes, memory]:
return tf.transpose(result, perm=[0, 2, 1, 3])
def _link(self, prev_link, prev_precedence_weights, write_weights):
"""Calculates the new link graphs.
For each write head, the link is a directed graph (represented by a matrix
with entries in range [0, 1]) whose vertices are the memory locations, and
an edge indicates temporal ordering of writes.
Args:
prev_link: A tensor of shape `[batch_size, num_writes, memory_size,
memory_size]` representing the previous link graphs for each write
head.
prev_precedence_weights: A tensor of shape `[batch_size, num_writes,
memory_size]` which is the previous "aggregated" write weights for
each write head.
write_weights: A tensor of shape `[batch_size, num_writes, memory_size]`
containing the new locations in memory written to.
Returns:
A tensor of shape `[batch_size, num_writes, memory_size, memory_size]`
containing the new link graphs for each write head.
"""
with tf.name_scope('link'):
batch_size = prev_link.get_shape()[0].value
write_weights_i = tf.expand_dims(write_weights, 3)
write_weights_j = tf.expand_dims(write_weights, 2)
prev_precedence_weights_j = tf.expand_dims(prev_precedence_weights, 2)
prev_link_scale = 1 - write_weights_i - write_weights_j
new_link = write_weights_i * prev_precedence_weights_j
link = prev_link_scale * prev_link + new_link
# Return the link with the diagonal set to zero, to remove self-looping
# edges.
return tf.matrix_set_diag(
link,
tf.zeros(
[batch_size, self._num_writes, self._memory_size],
dtype=link.dtype))
def _precedence_weights(self, prev_precedence_weights, write_weights):
"""Calculates the new precedence weights given the current write weights.
The precedence weights are the "aggregated write weights" for each write
head, where write weights with sum close to zero will leave the precedence
weights unchanged, but with sum close to one will replace the precedence
weights.
Args:
prev_precedence_weights: A tensor of shape `[batch_size, num_writes,
memory_size]` containing the previous precedence weights.
write_weights: A tensor of shape `[batch_size, num_writes, memory_size]`
containing the new write weights.
Returns:
A tensor of shape `[batch_size, num_writes, memory_size]` containing the
new precedence weights.
"""
with tf.name_scope('precedence_weights'):
write_sum = tf.reduce_sum(write_weights, 2, keep_dims=True)
return (1 - write_sum) * prev_precedence_weights + write_weights
@property
def state_size(self):
"""Returns a `TemporalLinkageState` tuple of the state tensors' shapes."""
return TemporalLinkageState(
link=tf.TensorShape(
[self._num_writes, self._memory_size, self._memory_size]),
precedence_weights=tf.TensorShape([self._num_writes,
self._memory_size]),)
class Freeness(snt.RNNCore):
"""Memory usage that is increased by writing and decreased by reading.
This module is a pseudo-RNNCore whose state is a tensor with values in
the range [0, 1] indicating the usage of each of `memory_size` memory slots.
The usage is:
* Increased by writing, where usage is increased towards 1 at the write
addresses.
* Decreased by reading, where usage is decreased after reading from a
location when free_gate is close to 1.
The function `write_allocation_weights` can be invoked to get free locations
to write to for a number of write heads.
"""
def __init__(self, memory_size, name='freeness'):
"""Creates a Freeness module.
Args:
memory_size: Number of memory slots.
name: Name of the module.
"""
super(Freeness, self).__init__(name=name)
self._memory_size = memory_size
def _build(self, write_weights, free_gate, read_weights, prev_usage):
"""Calculates the new memory usage u_t.
Memory that was written to in the previous time step will have its usage
increased; memory that was read from and the controller says can be "freed"
will have its usage decreased.
Args:
write_weights: tensor of shape `[batch_size, num_writes,
memory_size]` giving write weights at previous time step.
free_gate: tensor of shape `[batch_size, num_reads]` which indicates
which read heads read memory that can now be freed.
read_weights: tensor of shape `[batch_size, num_reads,
memory_size]` giving read weights at previous time step.
prev_usage: tensor of shape `[batch_size, memory_size]` giving
usage u_{t - 1} at the previous time step, with entries in range
[0, 1].
Returns:
tensor of shape `[batch_size, memory_size]` representing updated memory
usage.
"""
# Calculation of usage is not differentiable with respect to write weights.
write_weights = tf.stop_gradient(write_weights)
usage = self._usage_after_write(prev_usage, write_weights)
usage = self._usage_after_read(usage, free_gate, read_weights)
return usage
def write_allocation_weights(self, usage, write_gates, num_writes):
"""Calculates freeness-based locations for writing to.
This finds unused memory by ranking the memory locations by usage, for each
write head. (For more than one write head, we use a "simulated new usage"
which takes into account the fact that the previous write head will increase
the usage in that area of the memory.)
Args:
usage: A tensor of shape `[batch_size, memory_size]` representing
current memory usage.
write_gates: A tensor of shape `[batch_size, num_writes]` with values in
the range [0, 1] indicating how much each write head does writing
based on the address returned here (and hence how much usage
increases).
num_writes: The number of write heads to calculate write weights for.
Returns:
tensor of shape `[batch_size, num_writes, memory_size]` containing the
freeness-based write locations. Note that this isn't scaled by
`write_gate`; this scaling must be applied externally.
"""
with tf.name_scope('write_allocation_weights'):
# expand gatings over memory locations
write_gates = tf.expand_dims(write_gates, -1)
allocation_weights = []
for i in range(num_writes):
allocation_weights.append(self._allocation(usage))
# update usage to take into account writing to this new allocation
usage += ((1 - usage) * write_gates[:, i, :] * allocation_weights[i])
# Pack the allocation weights for the write heads into one tensor.
return tf.stack(allocation_weights, axis=1)
def _usage_after_write(self, prev_usage, write_weights):
"""Calcualtes the new usage after writing to memory.
Args:
prev_usage: tensor of shape `[batch_size, memory_size]`.
write_weights: tensor of shape `[batch_size, num_writes, memory_size]`.
Returns:
New usage, a tensor of shape `[batch_size, memory_size]`.
"""
with tf.name_scope('usage_after_write'):
# Calculate the aggregated effect of all write heads
write_weights = 1 - tf.reduce_prod(1 - write_weights, [1])
return prev_usage + (1 - prev_usage) * write_weights
def _usage_after_read(self, prev_usage, free_gate, read_weights):
"""Calcualtes the new usage after reading and freeing from memory.
Args:
prev_usage: tensor of shape `[batch_size, memory_size]`.
free_gate: tensor of shape `[batch_size, num_reads]` with entries in the
range [0, 1] indicating the amount that locations read from can be
freed.
read_weights: tensor of shape `[batch_size, num_reads, memory_size]`.
Returns:
New usage, a tensor of shape `[batch_size, memory_size]`.
"""
with tf.name_scope('usage_after_read'):
free_gate = tf.expand_dims(free_gate, -1)
free_read_weights = free_gate * read_weights
phi = tf.reduce_prod(1 - free_read_weights, [1], name='phi')
return prev_usage * phi
def _allocation(self, usage):
r"""Computes allocation by sorting `usage`.
This corresponds to the value a = a_t[\phi_t[j]] in the paper.
Args:
usage: tensor of shape `[batch_size, memory_size]` indicating current
memory usage. This is equal to u_t in the paper when we only have one
write head, but for multiple write heads, one should update the usage
while iterating through the write heads to take into account the
allocation returned by this function.
Returns:
Tensor of shape `[batch_size, memory_size]` corresponding to allocation.
"""
with tf.name_scope('allocation'):
# Ensure values are not too small prior to cumprod.
usage = _EPSILON + (1 - _EPSILON) * usage
nonusage = 1 - usage
sorted_nonusage, indices = tf.nn.top_k(
nonusage, k=self._memory_size, name='sort')
sorted_usage = 1 - sorted_nonusage
prod_sorted_usage = tf.cumprod(sorted_usage, axis=1, exclusive=True)
sorted_allocation = sorted_nonusage * prod_sorted_usage
inverse_indices = util.batch_invert_permutation(indices)
# This final line "unsorts" sorted_allocation, so that the indexing
# corresponds to the original indexing of `usage`.
return util.batch_gather(sorted_allocation, inverse_indices)
@property
def state_size(self):
"""Returns the shape of the state tensor."""
return tf.TensorShape([self._memory_size])
@@ -0,0 +1,41 @@
import tensorflow as tf
import numpy as np
import time
def reducedimension(input_, dimension = 2, learning_rate = 0.01, hidden_layer = 256, epoch = 20):
input_size = input_.shape[1]
X = tf.placeholder("float", [None, input_size])
weights = {
'encoder_h1': tf.Variable(tf.random_normal([input_size, hidden_layer])),
'encoder_h2': tf.Variable(tf.random_normal([hidden_layer, dimension])),
'decoder_h1': tf.Variable(tf.random_normal([dimension, hidden_layer])),
'decoder_h2': tf.Variable(tf.random_normal([hidden_layer, input_size])),
}
biases = {
'encoder_b1': tf.Variable(tf.random_normal([hidden_layer])),
'encoder_b2': tf.Variable(tf.random_normal([dimension])),
'decoder_b1': tf.Variable(tf.random_normal([hidden_layer])),
'decoder_b2': tf.Variable(tf.random_normal([input_size])),
}
first_layer_encoder = tf.nn.sigmoid(tf.add(tf.matmul(X, weights['encoder_h1']), biases['encoder_b1']))
second_layer_encoder = tf.nn.sigmoid(tf.add(tf.matmul(first_layer_encoder, weights['encoder_h2']), biases['encoder_b2']))
first_layer_decoder = tf.nn.sigmoid(tf.add(tf.matmul(second_layer_encoder, weights['decoder_h1']), biases['decoder_b1']))
second_layer_decoder = tf.nn.sigmoid(tf.add(tf.matmul(first_layer_decoder, weights['decoder_h2']), biases['decoder_b2']))
cost = tf.reduce_mean(tf.pow(X - second_layer_decoder, 2))
optimizer = tf.train.RMSPropOptimizer(learning_rate).minimize(cost)
sess = tf.InteractiveSession()
sess.run(tf.global_variables_initializer())
for i in range(epoch):
last_time = time.time()
_, loss = sess.run([optimizer, cost], feed_dict={X: input_})
if (i + 1) % 10 == 0:
print('epoch:', i + 1, 'loss:', loss, 'time:', time.time() - last_time)
vectors = sess.run(second_layer_encoder, feed_dict={X: input_})
tf.reset_default_graph()
return vectors
@@ -0,0 +1,142 @@
# Copyright 2017 Google Inc.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# ==============================================================================
"""DNC Cores.
These modules create a DNC core. They take input, pass parameters to the memory
access module, and integrate the output of memory to form an output.
"""
from __future__ import absolute_import
from __future__ import division
from __future__ import print_function
import collections
import numpy as np
import sonnet as snt
import tensorflow as tf
import access
DNCState = collections.namedtuple('DNCState', ('access_output', 'access_state',
'controller_state'))
class DNC(snt.RNNCore):
"""DNC core module.
Contains controller and memory access module.
"""
def __init__(self,
access_config,
controller_config,
output_size,
clip_value=None,
name='dnc'):
"""Initializes the DNC core.
Args:
access_config: dictionary of access module configurations.
controller_config: dictionary of controller (LSTM) module configurations.
output_size: output dimension size of core.
clip_value: clips controller and core output values to between
`[-clip_value, clip_value]` if specified.
name: module name (default 'dnc').
Raises:
TypeError: if direct_input_size is not None for any access module other
than KeyValueMemory.
"""
super(DNC, self).__init__(name=name)
with self._enter_variable_scope():
self._controller = snt.LSTM(**controller_config)
self._access = access.MemoryAccess(**access_config)
self._access_output_size = np.prod(self._access.output_size.as_list())
self._output_size = output_size
self._clip_value = clip_value or 0
self._output_size = tf.TensorShape([output_size])
self._state_size = DNCState(
access_output=self._access_output_size,
access_state=self._access.state_size,
controller_state=self._controller.state_size)
def _clip_if_enabled(self, x):
if self._clip_value > 0:
return tf.clip_by_value(x, -self._clip_value, self._clip_value)
else:
return x
def _build(self, inputs, prev_state):
"""Connects the DNC core into the graph.
Args:
inputs: Tensor input.
prev_state: A `DNCState` tuple containing the fields `access_output`,
`access_state` and `controller_state`. `access_state` is a 3-D Tensor
of shape `[batch_size, num_reads, word_size]` containing read words.
`access_state` is a tuple of the access module's state, and
`controller_state` is a tuple of controller module's state.
Returns:
A tuple `(output, next_state)` where `output` is a tensor and `next_state`
is a `DNCState` tuple containing the fields `access_output`,
`access_state`, and `controller_state`.
"""
prev_access_output = prev_state.access_output
prev_access_state = prev_state.access_state
prev_controller_state = prev_state.controller_state
batch_flatten = snt.BatchFlatten()
controller_input = tf.concat(
[batch_flatten(inputs), batch_flatten(prev_access_output)], 1)
controller_output, controller_state = self._controller(
controller_input, prev_controller_state)
controller_output = self._clip_if_enabled(controller_output)
controller_state = snt.nest.map(self._clip_if_enabled, controller_state)
access_output, access_state = self._access(controller_output,
prev_access_state)
output = tf.concat([controller_output, batch_flatten(access_output)], 1)
output = snt.Linear(
output_size=self._output_size.as_list()[0],
name='output_linear')(output)
output = self._clip_if_enabled(output)
return output, DNCState(
access_output=access_output,
access_state=access_state,
controller_state=controller_state)
def initial_state(self, batch_size, dtype=tf.float32):
return DNCState(
controller_state=self._controller.initial_state(batch_size, dtype),
access_state=self._access.initial_state(batch_size, dtype),
access_output=tf.zeros(
[batch_size] + self._access.output_size.as_list(), dtype))
@property
def state_size(self):
return self._state_size
@property
def output_size(self):
return self._output_size
@@ -0,0 +1,739 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import sys\n",
"import warnings\n",
"\n",
"if not sys.warnoptions:\n",
" warnings.simplefilter('ignore')"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"from tqdm import tqdm\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Date</th>\n",
" <th>Open</th>\n",
" <th>High</th>\n",
" <th>Low</th>\n",
" <th>Close</th>\n",
" <th>Adj Close</th>\n",
" <th>Volume</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2016-11-02</td>\n",
" <td>778.200012</td>\n",
" <td>781.650024</td>\n",
" <td>763.450012</td>\n",
" <td>768.700012</td>\n",
" <td>768.700012</td>\n",
" <td>1872400</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2016-11-03</td>\n",
" <td>767.250000</td>\n",
" <td>769.950012</td>\n",
" <td>759.030029</td>\n",
" <td>762.130005</td>\n",
" <td>762.130005</td>\n",
" <td>1943200</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2016-11-04</td>\n",
" <td>750.659973</td>\n",
" <td>770.359985</td>\n",
" <td>750.560974</td>\n",
" <td>762.020020</td>\n",
" <td>762.020020</td>\n",
" <td>2134800</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2016-11-07</td>\n",
" <td>774.500000</td>\n",
" <td>785.190002</td>\n",
" <td>772.549988</td>\n",
" <td>782.520020</td>\n",
" <td>782.520020</td>\n",
" <td>1585100</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2016-11-08</td>\n",
" <td>783.400024</td>\n",
" <td>795.632996</td>\n",
" <td>780.190002</td>\n",
" <td>790.510010</td>\n",
" <td>790.510010</td>\n",
" <td>1350800</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" Date Open High Low Close Adj Close \\\n",
"0 2016-11-02 778.200012 781.650024 763.450012 768.700012 768.700012 \n",
"1 2016-11-03 767.250000 769.950012 759.030029 762.130005 762.130005 \n",
"2 2016-11-04 750.659973 770.359985 750.560974 762.020020 762.020020 \n",
"3 2016-11-07 774.500000 785.190002 772.549988 782.520020 782.520020 \n",
"4 2016-11-08 783.400024 795.632996 780.190002 790.510010 790.510010 \n",
"\n",
" Volume \n",
"0 1872400 \n",
"1 1943200 \n",
"2 2134800 \n",
"3 1585100 \n",
"4 1350800 "
]
},
"execution_count": 3,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/GOOG-year.csv')\n",
"df.head()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.112708</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.090008</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.089628</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.160459</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>0.188066</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0\n",
"0 0.112708\n",
"1 0.090008\n",
"2 0.089628\n",
"3 0.160459\n",
"4 0.188066"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = minmax.transform(df.iloc[:, 4:5].astype('float32')) # Close index\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Forecast\n",
"\n",
"This example is using model 1.lstm, if you want to use another model, need to tweak a little bit, but I believe it is not that hard.\n",
"\n",
"I want to forecast 30 days ahead! So just change `test_size` to forecast `t + N` ahead.\n",
"\n",
"Also, I want to simulate 10 times, 10 variances of forecasted patterns. Just change `simulation_size`."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((252, 7), (252, 1))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"simulation_size = 10\n",
"num_layers = 1\n",
"size_layer = 128\n",
"timestamp = 5\n",
"epoch = 300\n",
"dropout_rate = 0.8\n",
"test_size = 30\n",
"learning_rate = 0.01\n",
"\n",
"df_train = df_log\n",
"df.shape, df_train.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" forget_bias = 0.1,\n",
" ):\n",
" def lstm_cell(size_layer):\n",
" return tf.nn.rnn_cell.LSTMCell(size_layer, state_is_tuple = False)\n",
"\n",
" rnn_cells = tf.nn.rnn_cell.MultiRNNCell(\n",
" [lstm_cell(size_layer) for _ in range(num_layers)],\n",
" state_is_tuple = False,\n",
" )\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
" drop = tf.contrib.rnn.DropoutWrapper(\n",
" rnn_cells, output_keep_prob = forget_bias\n",
" )\n",
" self.hidden_layer = tf.placeholder(\n",
" tf.float32, (None, num_layers * 2 * size_layer)\n",
" )\n",
" self.outputs, self.last_state = tf.nn.dynamic_rnn(\n",
" drop, self.X, initial_state = self.hidden_layer, dtype = tf.float32\n",
" )\n",
" self.logits = tf.layers.dense(self.outputs[-1], output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [],
"source": [
"def forecast():\n",
" tf.reset_default_graph()\n",
" modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], dropout_rate\n",
" )\n",
" sess = tf.InteractiveSession()\n",
" sess.run(tf.global_variables_initializer())\n",
" date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"\n",
" pbar = tqdm(range(epoch), desc = 'train loop')\n",
" for i in pbar:\n",
" init_value = np.zeros((1, num_layers * 2 * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, last_state, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.last_state, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {\n",
" modelnn.X: batch_x,\n",
" modelnn.Y: batch_y,\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" ) \n",
" init_value = last_state\n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))\n",
" \n",
" future_day = test_size\n",
"\n",
" output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
" output_predict[0] = df_train.iloc[0]\n",
" upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
" init_value = np.zeros((1, num_layers * 2 * size_layer))\n",
"\n",
" for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" ),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
" if upper_b != df_train.shape[0]:\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"\n",
" init_value = last_state\n",
" \n",
" for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i]\n",
" out_logits, last_state = sess.run(\n",
" [modelnn.logits, modelnn.last_state],\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(o, axis = 0),\n",
" modelnn.hidden_layer: init_value,\n",
" },\n",
" )\n",
" init_value = last_state\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
" \n",
" output_predict = minmax.inverse_transform(output_predict)\n",
" deep_future = anchor(output_predict[:, 0], 0.4)\n",
" \n",
" return deep_future"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0818 12:00:52.795618 140214804277056 deprecation.py:323] From <ipython-input-6-d01d21f09afe>:12: LSTMCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.LSTMCell, and will be replaced by that in Tensorflow 2.0.\n",
"W0818 12:00:52.799092 140214804277056 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f8644897400>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n",
"W0818 12:00:52.801252 140214804277056 deprecation.py:323] From <ipython-input-6-d01d21f09afe>:16: MultiRNNCell.__init__ (from tensorflow.python.ops.rnn_cell_impl) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"This class is equivalent as tf.keras.layers.StackedRNNCells, and will be replaced by that in Tensorflow 2.0.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 1\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"W0818 12:00:53.121960 140214804277056 lazy_loader.py:50] \n",
"The TensorFlow contrib module will not be included in TensorFlow 2.0.\n",
"For more information, please see:\n",
" * https://github.com/tensorflow/community/blob/master/rfcs/20180907-contrib-sunset.md\n",
" * https://github.com/tensorflow/addons\n",
" * https://github.com/tensorflow/io (for I/O related ops)\n",
"If you depend on functionality not listed there, please file an issue.\n",
"\n",
"W0818 12:00:53.125179 140214804277056 deprecation.py:323] From <ipython-input-6-d01d21f09afe>:27: dynamic_rnn (from tensorflow.python.ops.rnn) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `keras.layers.RNN(cell)`, which is equivalent to this API\n",
"W0818 12:00:53.314420 140214804277056 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0818 12:00:53.321002 140214804277056 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/rnn_cell_impl.py:961: calling Zeros.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0818 12:00:53.718872 140214804277056 deprecation.py:323] From <ipython-input-6-d01d21f09afe>:29: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"train loop: 100%|██████████| 300/300 [01:17<00:00, 3.90it/s, acc=95.9, cost=0.00437]\n",
"W0818 12:02:12.766668 140214804277056 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f85be966eb8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 2\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:18<00:00, 3.81it/s, acc=96.2, cost=0.00386]\n",
"W0818 12:03:31.524121 140214804277056 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f85b4c59dd8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 3\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:17<00:00, 3.86it/s, acc=95.9, cost=0.00421]\n",
"W0818 12:04:49.292782 140214804277056 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f85ac67f5f8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 4\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:17<00:00, 3.85it/s, acc=95.1, cost=0.00617]\n",
"W0818 12:06:07.690939 140214804277056 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f85209545f8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 5\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:18<00:00, 3.81it/s, acc=96.8, cost=0.00293]\n",
"W0818 12:07:26.842436 140214804277056 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f85089d1128>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 6\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:17<00:00, 3.82it/s, acc=97.3, cost=0.00178]\n",
"W0818 12:08:45.222193 140214804277056 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f85082c6160>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 7\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:16<00:00, 3.94it/s, acc=97.5, cost=0.00161]\n",
"W0818 12:10:01.933482 140214804277056 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f84fc7de208>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 8\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:17<00:00, 3.81it/s, acc=97.5, cost=0.00156]\n",
"W0818 12:11:20.348971 140214804277056 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f84fc7127b8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 9\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:18<00:00, 3.81it/s, acc=96.7, cost=0.00297]\n",
"W0818 12:12:39.812369 140214804277056 rnn_cell_impl.py:893] <tensorflow.python.ops.rnn_cell_impl.LSTMCell object at 0x7f84f6ed44a8>: Using a concatenated state is slower and will soon be deprecated. Use state_is_tuple=True.\n"
]
},
{
"name": "stdout",
"output_type": "stream",
"text": [
"simulation 10\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 300/300 [01:17<00:00, 3.98it/s, acc=97.5, cost=0.00179]\n"
]
}
],
"source": [
"results = []\n",
"for i in range(simulation_size):\n",
" print('simulation %d'%(i + 1))\n",
" results.append(forecast())"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"['2017-11-27', '2017-11-28', '2017-11-29', '2017-11-30', '2017-12-01']"
]
},
"execution_count": 10,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"date_ori = pd.to_datetime(df.iloc[:, 0]).tolist()\n",
"for i in range(test_size):\n",
" date_ori.append(date_ori[-1] + timedelta(days = 1))\n",
"date_ori = pd.Series(date_ori).dt.strftime(date_format = '%Y-%m-%d').tolist()\n",
"date_ori[-5:]"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Sanity check\n",
"\n",
"Some of our models might not have stable gradient, so forecasted trend might really hangwired. You can use many methods to filter out unstable models.\n",
"\n",
"This method is very simple,\n",
"1. If one of element in forecasted trend lower than min(original trend).\n",
"2. If one of element in forecasted trend bigger than max(original trend) * 2.\n",
"\n",
"If both are true, reject that trend."
]
},
{
"cell_type": "code",
"execution_count": 13,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"6"
]
},
"execution_count": 13,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"accepted_results = []\n",
"for r in results:\n",
" if (np.array(r[-test_size:]) < np.min(df['Close'])).sum() == 0 and \\\n",
" (np.array(r[-test_size:]) > np.max(df['Close']) * 2).sum() == 0:\n",
" accepted_results.append(r)\n",
"len(accepted_results)"
]
},
{
"cell_type": "code",
"execution_count": 14,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA3gAAAFBCAYAAAAlhA0CAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzdeXhV1b3/8fcZEjIPhCHMg8AWGQREw6DUKkirLb22VkVBsWJ/WDQoSvUiUhD1ggoiUZHaoiiihV5RC2qpWuVSRUWiCOIOQ0KABBIykBxCcnLOPr8/9iGEIUxJOCR8Xs/Dk5y99l77u77heciXtfbajkAggIiIiIiIiDR8zlAHICIiIiIiInVDBZ6IiIiIiEgjoQJPRERERESkkVCBJyIiIiIi0kiowBMREREREWkkVOCJiIiIiIg0EirwREREREREGgl3qAMQERGRM2MYhgOYDPw/IAF4H/i9aZolwfamwHxgKBAA/gncfaj9BP1OBaYDw0zT/Kja8aHAU4ABFAETTdNcGmy7CngG6ALsA2aapvnnYNt1wH8DPYFyYAVwv2mapXWQBhERqUYzeCIiUucMwziv/gMxhOO9DRgNDAZaA5FAWrX2x4FEoBNwAdASmHaiDg3DuAD4LZB71PGLgCXAI0A8cDHwTbAtDFgOLAi23QTMMQzj4uDl8cFYWgPdgTbA06c/XBEROZnz6h9gEREBwzAeBu4CWgA7gUdM01xuGEYTYC9wuWmaG4PnNgeygQ6maeYZhvEL7F/UOwI/AONM09wQPDcLe7boVvujEQ08eLx7Bc93Yc8G3Q6UArOxi5Mw0zR9hmHEA3OAawELeAX4k2ma/uOM6TLgOezi4SDwv9izS95gew9gLnAJUAk8Z5rmk8EYHgLuDMaYAfwX4AIyD8US7ONTYLFpmn8xDGNMcFxfYRdZ8w3DeAV4GbvwOTRbNt40zeLg9e2CMV6B/R+sbwITgT3AT0zT/D54XgsgK5jz/BP+MOGXwF9N09wZvHYW8IlhGHebplmGXdi9U21Gbzkw4iR9vhDMyYtHHZ8CLDBN84Pg54LgH4CmQBzwummaAeBrwzA2AxcB35mmuaRaP2WGYbyMPUMoIiJ1TDN4IiLnn23YRUY89i/Ziw3DaGWaZgXwNjCy2rk3Ap8Fi7u+wELs5YBJ2LM17wULw0NGAtcBCcHC6Lj3Cp57F/BzoA/QD7uwqu5VwIe95K8vcA0wtoYx+YH7gWbAQOBq4A8AhmHEAh8BH2LPIHUBPg5eNzEY87XYBcrvgLIa7nG0FGA79qzYE4AD+B8Oz1K1IzhbFiwkVwA7sIvjNsBbwQL0LWBUtX5HAh8fKu4Mwyg2DOPyE8ThOOr7JkDX4OcXgF8YhpFoGEYi8BvgA2pgGMZvgQrTNN8/TvOA4DnfG4aRaxjG4uASUEzT3ItdsN5hGIbLMIyBQAdgTQ23GgJsOsGYRETkDGkGT0TkPGOa5rJqH/9mGMZ/A5cB72IvwVuAvQwP4JbgZ4DfY8/gfBn8vMgwjMnYv/h/Fjw279Bs0inc60bsmbRdAIZhzMQuzDAMoyV20ZVgmuZB4IBhGM8eiuE4Y/qm2scswzAWAD/BnrX7BbDHNM3ZwfZy4NAYxgJ/NE3TDH7+Lnj/2GMSd6wc0zQPLYf0AVuDfwDyDcOYA/wp+Pky7MJv0qEZQQ4XP4uAZYZhPByc/RqNPbN5aGwJJ4jhQ+CPhmEsxX4m7qHg8ajg1/VAOIdn2j7m2Jk5oGrMTwLDarhX22Bs1wA5wbjTsGdswS7w/oI9Swn2s347j+7EMIxh2LO2KScYl4iInCEVeCIi5xnDMG7DnrnqGDwUgz3zBfBvIMowjBTs5Zp9sJ+tAntG5nbDMO6t1l04duFyyBG/0J/kXq2POr/69x2AMCDXMIxDx5xH91/tPt2wl3P2xy5u3ASfD8OeSdt2vOtO0nYyR4+1JYeXYMYG4y2qdp8d1Yq7KqZpfmkYRhlwpWEYudgzjO+dYgwLg31/ij3m2djLNncF25cCG4BfYc/uPQMsxi6ujzYNe4llVg33Ogi8YppmRnC8T2LPjGIYxoXYM5G/Bv6FPYO4wjCMHNM0Vx7qwDCMAdj/iXDDoX5ERKRuqcATETmPGIbRAfs5sauBL0zT9BuG8S3BZX7Bz0uxlwnuBVZU2+lwJ/CEaZpPnOAWgVO9F/YmHm2rXduu2vc7gQqg2fGKouOYD6QDI03TLDUM4z7ghmp93VzDdTuxNx/ZeNTxA8GvUcChHSeTjzoncNTnJ4PHepmmWWgYxn8Bz1e7T3vDMNw1jGcR9jLNPcDfTdMsryHeI5imaWHPEv4JwDCMa4DdwT9gF+jjTdM8EGx/iZqXTV4NtDUM4w/Bz82BpYZhzDJNcxZ2oVh9zNW/7wlkmKb5z0OhGYaxEnsJ7srgvftiF66/M03zY0REpF6owBMROb9EY/9ifuj5rjuwfzmvbgnwDvayvkeqHX8ZWG4YxkfYm4tEAVcCq2vY7v5k91oKTAgWAgc4vLwQ0zRzDcNYBcw2DONRwIO9YUhb0zQ/41ix2IWYJzibdPeh+2I/+zYnWPTNx551vCi41PQvwAzDMH7AXl7ZC9htmma+YRi7gVHB5Z63YxeCJxIL7Af2G4bRBphUre0r7IJ2pmEYf8J+ZvAS0zT/E2xfjL08tBR7GeQpCT4Dl4j9LGB37FnMx4KFH8DXwFjDMP4Y/Px77ELteK7GnjU95Gvs2ddDz+y9AjxqGMZi7EL0Yezcgl1cdw2+KuHfQGfspbFPBePsib2c9F7TNP9xquMTEZHTp01WRETOI6Zp/oC9jO8L7Bm6XsB/jjrnS+yCqzXVNuQwTXMd9sYoz2MvPdwKjKnFvV4GVmEXHOnY73DzYRc/YO9OGY69W2cR8HegFcf3IPbzgqXBfv9WLY5S7OfKfoldmGwBfhpsnoNdaK7CLhD/iv2qAYJjnYRd6PYAPq9prEHTsTeL2Y89a/V2tRj8wft3wd6VdBf2qwQOte/Efl4uAPxf9U4Nw/AYhnFFDfdshp23A9g/q4WH3j0X9Dvs5bG7sGf1OmMXq4f63mQYxq3BGApM09xz6A/2z6HINE1PsH0h8Br284s7sGdYU4Nt24L3moedx8+wdzL9S/BWD2DPCP41OB6PYRjaZEVEpB44AoGjV5iIiIicfYZh/Bx4yTTNDqGOJRQMw1iIvXHLlFDHIiIiDZeWaIqISEgYhhGJPZO2CvtVA3/i8IYu5xXDMDpib1DSN8ShiIhIA6clmiIiEioO7GWNRdhLNDcDU0MaUQgYhjEDe5OXp03TzAx1PCIi0rBpiaaIiIiIiEgjoRk8ERERERGRRqIhPoPXBLgUe7tp/0nOFRERERERaWxc2DtLf429q3GVhljgXcpRW0iLiIiIiIich64A1lQ/0BALvFyAoqIDWNa59fxgUlIMBQWeUIfRoCmHtacc1p5yWDeUx9pTDmtPOaw95bBuKI+1pxwe5nQ6SEyMhmBtVF1DLPD8AJYVOOcKPOCcjKmhUQ5rTzmsPeWwbiiPtacc1p5yWHvKYd1QHmtPOTzGMY+saZMVERERERGRRkIFnoiIiIiISCPREJdoHpff76OoKB+fzxuyGPLynFiWFbL7h4rbHU5iYnNcrkbz10lEREREpEFqNL+RFxXlExERRXR0Mg6HIyQxuN1OfL7zq8ALBAIcOFBCUVE+zZq1CnU4IiIiIiLntUazRNPn8xIdHRey4u585XA4iI6OC+nMqYiIiIiI2BpNgQeouAsR5V1ERERE5NzQqAo8ERERERGR85kKvHqyevWn3HrrDdxxxy1kZ2eFOpxjlJaW8sYbi2ps93q9TJx4L9dddzXXXXf1WYxMRERERETOlAq8evLuu29z553jeOWVJbRv3/GUr/P7j3lXYb3weEpZsuS1GtudTicjR45i7twXz0o8IiIiIiLnkkqfn9lvpZO1pyTUoZyWRrOL5rlk3rzZbNiQTnb2DpYvX0Za2gLWrv2cBQuex7IsEhISmTRpMm3btmP9+nU899wzGEZ3MjJM7rrrbvr06Uta2rNs27YFr9dL3779uffe+3G5XOTn5zF37tPs2rUTgKFDhzN69B2sWvUhy5a9ic9XCcD48ffRv/9lWJbFnDlPsX7914SFhRMVFcn8+QuZM2cWHo+HMWNuISIigpdeWnjEGNxuN5demkJubs5Zz5+IiIiISKjlFR1kU1YRlxcepGNyXKjDOWWNtsD7z/e5rNmQWy99X967FYN71fxKgNTUB8jIMBk5cjSDB19BUVEhjz8+lbS0P9OpU2dWrHiH6dOn8PLL9hLJzMztTJo0mZ49ewMwc+YM+vTpx8MPP4plWUyfPoWVK99jxIjreeyxRxk4cDBPPPE0AMXFxQCkpAxg2LDhOBwOsrOzmDDhDyxf/j5bt2aQnr6OxYuX4XQ6KSmx/wdi4sSHGDt2NK++uqReciQiIiIi0pAVe+xd4hNjm4Q4ktPTaAu8c8mmTRu54IJudOrUGYBrrx3B7NmzKCs7AEDbtu2qijuANWtWs3nzJt566w0AysvLadGiJWVlZWzcuIFnn32h6tyEhAQAdu/exbRpj5Cfn4/b7aawsICCgn20bt0Wn8/HzJkz6NevP4MGXXG2hi0iIiIi0mAVlVYAkBATHuJITk+jLfAG9zrxLNu5JDIy6qgjAZ588hnatGl7xNGysrIa+5g27RHuued+hgy5EsuyGDr0crxeL0lJzXj99aWkp3/DunVfMX9+GgsXLq6HUYiIiIiINB7FnkMFXsOawdMmK2dBjx692LYtgx07sgD44IMVdO1qEBUVfdzzBw8ewuLFi6o2XCkuLiYnZzdRUVH07NmbpUsPL6s8tETT4/HQqlVrAFaufA+v155SLioqory8nJSUgYwbdw8xMTHk5OwmOjqa8vJyfD5ffQ1bRERERKTBKvZUEB3hJjzMFepQTkujncE7lyQLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"accuracies = [calculate_accuracy(df['Close'].values, r[:-test_size]) for r in accepted_results]\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"for no, r in enumerate(accepted_results):\n",
" plt.plot(r, label = 'forecast %d'%(no + 1))\n",
"plt.plot(df['Close'], label = 'true trend', c = 'black')\n",
"plt.legend()\n",
"plt.title('average accuracy: %.4f'%(np.mean(accuracies)))\n",
"\n",
"x_range_future = np.arange(len(results[0]))\n",
"plt.xticks(x_range_future[::30], date_ori[::30])\n",
"\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,746 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 1,
"metadata": {},
"outputs": [],
"source": [
"import tensorflow as tf\n",
"import numpy as np\n",
"import matplotlib.pyplot as plt\n",
"import seaborn as sns\n",
"import pandas as pd\n",
"from sklearn.preprocessing import MinMaxScaler\n",
"from datetime import datetime\n",
"from datetime import timedelta\n",
"sns.set()\n",
"tf.compat.v1.random.set_random_seed(1234)"
]
},
{
"cell_type": "code",
"execution_count": 2,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>timestamp</th>\n",
" <th>close</th>\n",
" <th>positive</th>\n",
" <th>negative</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>2019-08-09T23:00:00</td>\n",
" <td>11860.074544</td>\n",
" <td>0.672896</td>\n",
" <td>0.327104</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>2019-08-09T23:20:00</td>\n",
" <td>11872.025879</td>\n",
" <td>0.595100</td>\n",
" <td>0.404900</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>2019-08-09T23:40:00</td>\n",
" <td>11880.504557</td>\n",
" <td>0.596702</td>\n",
" <td>0.403298</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>2019-08-10T00:00:00</td>\n",
" <td>11918.873481</td>\n",
" <td>0.577972</td>\n",
" <td>0.422028</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>2019-08-10T00:20:00</td>\n",
" <td>11937.581272</td>\n",
" <td>0.585342</td>\n",
" <td>0.414658</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" timestamp close positive negative\n",
"0 2019-08-09T23:00:00 11860.074544 0.672896 0.327104\n",
"1 2019-08-09T23:20:00 11872.025879 0.595100 0.404900\n",
"2 2019-08-09T23:40:00 11880.504557 0.596702 0.403298\n",
"3 2019-08-10T00:00:00 11918.873481 0.577972 0.422028\n",
"4 2019-08-10T00:20:00 11937.581272 0.585342 0.414658"
]
},
"execution_count": 2,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df = pd.read_csv('../dataset/BTC-sentiment.csv')\n",
"df.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## How we gather the data, provided by Bitcurate, bitcurate.com\n",
"\n",
"Because I don't have sentiment data related to stock market, so I will use crpytocurrency data, `BTC/USDT` from binance.\n",
"\n",
"1. close data came from CCXT, https://github.com/ccxt/ccxt, an open source cryptocurrency aggregator.\n",
"2. We gather from streaming twitter, crawling hardcoded crpyocurrency telegram groups and Reddit. And we store in Elasticsearch as a single index. We trained 1/4 layers BERT MULTILANGUAGE (200MB-ish, originally 700MB-ish) released by Google on most-possible-found sentiment data on the internet, leveraging sentiment on multilanguages, eg, english, korea, japan. **Actually, it is very hard to found negative sentiment related to bitcoin / btc in large volume.**\n",
"\n",
"How we request using elasticsearch-dsl, https://elasticsearch-dsl.readthedocs.io,\n",
"```python\n",
"# from index name\n",
"s = s.filter(\n",
" 'query_string',\n",
" default_field = 'text',\n",
" query = 'bitcoin OR btc',\n",
")\n",
"```\n",
"\n",
"We only do text query only contain `bitcoin` or `btc`."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Consensus introduction\n",
"\n",
"We have 2 questions here when saying about consensus, what happened,\n",
"\n",
"1. to future price if we assumed future sentiment is really positive, near to 1.0 . Eg, suddenly China want to adapt cryptocurrency and that can cause huge requested volumes.\n",
"2. to future price if we assumed future sentiment is really negative, near to 1.0 . Eg, suddenly hackers broke binance or any exchanges, or any news that caused wreck by negative sentiment.\n",
"\n",
"**We can use deep-learning to simulate for us!**"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA8EAAAF6CAYAAAA52lqfAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOy9eZwcZZ34/67qe2Z67p5kcoeQFEFMICBEDlEQFxAVlRW8UdRV96d7uKu7rsey6nqs+NVVl0sEBZdDXFAERAQChDMEAuTqHCSTZO7pOfruruP5/VFHd093z5WZScLU+/XiFabqqaeep+upp57P87kkIQQuLi4uLi4uLi4uLi4uLnMB+Ug3wMXFxcXFxcXFxcXFxcVltnCFYBcXFxcXFxcXFxcXF5c5gysEu7i4uLi4uLi4uLi4uMwZXCHYxcXFxcXFxcXFxcXFZc7gCsEuLi4uLi4uLi4uLi4ucwZXCHZxcXFxcXFxcXFxcXGZM3hn60aKovwQeD+wDHhjNBrdqihKC3ArsALIA7uBv4lGo/3WNeuB64EQsB/4SDQa7Tuccy4uLi4uLi4uLi4uLi5zl1kTgoF7gZ8ATxYdE8APotHoBgBFUf4L+B5wlaIoMnAbcGU0Gt2oKMrXrHOfnOq5CbYzALwJ6Ab0w+qxi4uLi4uLi4uLi8vRiAdoBzYBuSPclmMBL7AIOARoR7gth82sCcHRaHQjgKIoxccGgQ1FxZ4FPmf9/6lA1r4OuA5Tq/vJwzg3Ed5EqaDu4uLi4uLi4uLi4vL65Bxg47ilXBYB+4DlmLLVMc1saoLHxNLgfg74g3VoCdBhn49GowOKosiKojRP9ZwldI9HN8DQUArDEIfdr8OhpaWOWCx5RNtwJHD7PbeYi/2ei30Gt99zjbnY77nYZ3D7PZd4PfVZliWammrBWvu7zC2OGiEY+CmQBH52hNuhA/ZLccRpaak70k04Irj9nlvMxX7PxT6D2++5xlzs91zsM7j9nku8Dvvsuj/OQY4KIdgKmrUSeFc0GjWswweApUVlWgEjGo0OKooypXOTaVMsljzimuBIJEx/f+KItuFI4PZ7bjEX+z0X+wxuv+cac7Hfc7HP4PZ7LvF66rMsS69Hgd5lghzxFEmKovwnph/vpdFotNgpfTMQUhTlbOvvzwK/PcxzLi4uLi4uLi4uLi4uLnOY2UyR9N/A+4D5wF8URYkBHwD+FdgFPG0FzdoXjUbfG41GDUVRPgpcryhKECvVEcBUz7m4uLi4uLi4uLi4uEwWRVFWAb8CWoAY8LFoNLp7VJlfA2uKDq3BVPT9AZejitmMDv1F4IsVTkljXPM08MbpPOfi4uLi4uLi4uIy2+i6xtBQP5qWP9JNmTJ9fTKGYYxf8CjC6/XT1BTB4zlssec64OfRaPQ2RVE+AlwPnFdcIBqNfsz+f0VR1gKPAg8d7o1dpp+jwifYxcXFxcXFxcXF5fXM0FA/wWANtbXzkaSqOqCjGq9XRtOOHSFYCEEqFWdoqJ/W1vYp16MoShuwDrjAOnQ78DNFUSLRaLS/ymVXAb8Z5e7pcpTgCsEuLi4uLi4uLi4uM4ym5Y9pAfhYRJIkamvrSSaHq5a54YYbFl1zzTWjDw9Ho9HiixYDndFoVAeIRqO6oihd1vEyIVhRFD/wIeDth9kFlxnCFYKrcLREi4tEwke6CUcEt99zi7nY77nYZ3D7PdeYi/2ei30Gt98Toa9PxufzzGBrZgev94jH1Z00sixXfVZ33HHHkxUOXw38+2Hc8lLgQDQa3XIYdbjMIK4QXIVjIUXSUCJHQ50feYZ2FIUQiGQMOdw6I/VX4/UUfn8yuP0+OonnE4Q8QXweX8lxoeUACcnrn3SdR3ufZwq333OLudjvudhncPs9UQzDOKpMiS+77F34/X78/gD5fI61a09h9eo38H//ZyZV6e3tIRgM0tDQCMA///NXWbt2DQ88cD+3334ruVyOYDDE4sWL+dzn/o758+cDcMstvyAcDhMK1fD000/y7W//wLnnU089ye2338rPfnYDQghuuul6Nmx4FI/Hg65rXHLJe7jiio/Q3d3FFVe8l+XLV2AYOpqmsXbtKXziE5+mrW0e11zzfV599WUA9u9/jQULFuL3BwC46aZb8XhKNxsMwyh7VnaKpCuuuOKca6655tCon2e06vggsFBRFI+lBfYAC6zjlfgk8MsJPAaXI4QrBB+j7Do4zA/+9yUuWr+E95+7YkLX5F64B5FNEDz7Y+MXBrSOF8k+/HNqP3QNcm3T4TTX5TARhga6juQLHOmmzDn+64WfcWb76Vy0/PyS45kHf4QUqif09r89Qi1zcXFxcXE5PL797e9z3HHHo+s6f/u3n+a0007nllv+F4DvfOffOeGE1bz//Zc75X//+3u4/fbb+O53r2Hx4iUAvPjiCwwODjhC8MaNT/Cd7/yAzZs3jXnvxx57hM2bN3HTTbcSCATI5/N0dhZk0bq6Oqctqqryq1/dxGc/+0l+/es7+dKXvuKUu+yydzn9mAqf+cxnDn3mM5/ZP1aZaDTapyjKFuCDwG3Wvy9V8gdWFGURcI5VxuUo5dizZ3AhmVG54b5tGELw8AsHGUlNLMqg3rcXvXvXhO9jDB4CYSBSQ1Nt6oTRYwfROl6a8fscLQgtj5hEdMX8C/eQvvdbM9gil2ok80lSaqrkmJGMoXdH0WMHjlCrXFxcXFxcpo98Pk8+nyMcrh+z3E033cAXvvCPjgAMsG7daZx44kkA9PX1IoRg3rz5496zv7+XxsZG/H7Tosrv97N8+XEVy/p8Pj71qc8SibTx0EMPTLRb081ngS8oirIL+IL1N4qiPKAoymlF5T4O3BeNRmd+Ae0yZVxN8DGGEIJfPbiTkWSez7zrRG7843YefLaDK85fOf7Fho7QJx6WXyQGzH9zyak2d0IY2QSZB69BqFnqrvyfGb3X0YAQBqm7/hXfCecSWPfuCV2j9+7BGOlGCOEG1JhlDAQGpRsW2v4XARCJGMIwkGR3P3G60Ic6yT13F6HzP39UWz4Y6RHkmoYj3QwXFxeXw+JrX/sKfn+Azs5DnH76GZx++vqqZYeGBunr63UE3ko8+eTjnH32WyZ07/PP/yvuvfd3XHHFe1m79hROPfVNnH/+O/B6q4snq1e/gX37XptQ/dNNNBrdCZxR4fjFo/7+zqw1ymXKuELwMcZr3XE27+rn/ecex/o3zGfrvkEee6mTC89YQmPdOAtGQ4dJ5KYzHCE4NU7JqSOEIPf4LxFp0/VCJAagbXoXlplHr8d73Gn4lp06rfVOFSN2EJGMYQxWcyOpcM1Ql/n81Az4a2awdS6jMYSBEKXxAbR9m62TGiI9hFTXcgRa9vpE2/UU+oGXMYY68bRV1ggcaYzhHlJ3/SuhC/8O75KTK5bRuqPIje3IobG1Ki4uLnOTp17tZuMr3TNS99lr2jnrjRNLB2SbEedyOb72tS9z113/ywc+8KEp33vjxsf5/Of/DqDqpr19vLW1lVtvvYtt217llVe28Otf/5KHHnqQH/3op2Pc4cjG63F5/eCqL44xnt3Wi9cj87ZTFgHwrrOWYRiCb/3qBf686SB5Va96rTA0hK5O+F5GMmZel505IViLPonW8RLe482dR32gY1rrF4aGtucZcs/cgTCq/zazid69EwBjgmbmRiaOyJrBHER2ZrXyLuUIITAQZB65luyzd2Jk4ug9UeQ20xff3ixymR60zu3AxN4Pde9zR+SdMJIDgEDdWSmgqDnvZO7/L/Ivj2+yZyQHMUZ6p7mFLi4uLpMjEAhw5pnnsGnTc1XLNDU1E4m0sWPHtornk8kk3d3drFy5CoDGxkZGRkZKyoyMDNPU1Oz87fV6Wbv2FD760U/w059ez/PPP0M8XnpNMTt2bOe44yYWC8fFZSxcTfAxhG4YbNrRy9rjW6gJmo9uXlMNX7r8ZO7duI87HtnNS7v6+cfLT8ZXKXz9JDTBwjAQthA8hibYSA8jheqRpKntp6j7NyM3zCf4lk+Q3PscxjT7WIpc2vw30Y/22iZ8x1c38xG6BkJH8s6sCabeZQrBE/W1NoYLO8UiE4f6tqplhZolv+0v+NdcfNSZ6Aoh0Lt24FlwwpTHy2wjhEAgEMJA744i0sPmJoYQ+NdcSPYvP0fE+6Bdca5Rdz2F3rOb4FuuPHINnyBCGIhscla0ldqhrcQPJmDxm6u3J5vEsDbCxns/jPQw2UeuxX/aewmse8+0tnU87DlRO/AyIpdCCtSWnk8Ng6FhDPeMW1fu2dsxkoPUXvr1GWmri4tLOfpgJ9reZ/Gf9r4j5mJ01hsnrq2dDQzDYMuWzSW+vpX45Cc/xU9/+iO+970fsXChqZDZsuVF/H4/nZ2HWL/+TKfs6tUnsW/fa+zdu4cVK44nl8vywAP3ceGFpvXwzp07aGhooL19AQDR6E7C4Xrq6sKkUqVrT1VVufXWm+nv7+Md77hoOrvuMkdxheBjiB37h4inVdafWBps4ISlTfzL0iaeerWbm+7fwS8f2MFpSoR7ntzHmhUtfOBtVrQ8QwddnZBfqUgPm+Wp7hNsJAZI3fkVQm////AuO2XM+tRdG5FqGvEuKvUjEZk4Un0EyRtAbmxHj03cRHhCFAnw+Zfvx7vijKp9zz11G0aij5p3fnl621CEEAZajxmcTKSGJ+RPagx1Fq4fR+uldW4n//zdeBe+AU9k+RjtEOQ3/Q5kGU/7CXjaT5hxodno20vm/h8QeueX8S48cUbvNV0Iy+xKCAHCAI8fo38fUjiCd+kpgFSmCdb2v4jWue2oEoKFoYOulfnYavs2k330emov/y5yODJmHVp3FE9k+ZRSQgHkt/4FdaSL0OXVhWCtO4pt6ibSYwvBtguFMd1zxgRwNgYNDXXfC/hPOLfkvGNFkygLGlpeVyYOanba2zgR9MGDSL7guM/exWUm0Hp2k334Z9S8+6vIDfNLine truncated
"text/plain": [
"<Figure size 1224x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"from mpl_toolkits.axes_grid1 import host_subplot\n",
"import mpl_toolkits.axisartist as AA\n",
"\n",
"close = df['close'].tolist()\n",
"positive = df['positive'].tolist()\n",
"negative = df['negative'].tolist()\n",
"timestamp = df['timestamp'].tolist()\n",
"\n",
"plt.figure(figsize = (17, 5))\n",
"host = host_subplot(111)\n",
"plt.subplots_adjust(right = 0.75, top = 0.8)\n",
"par1 = host.twinx()\n",
"par2 = host.twinx()\n",
"\n",
"par2.spines['right'].set_position(('axes', 1.1))\n",
"par2.spines['bottom'].set_position(('axes', 0.9))\n",
"host.set_xlabel('timestamp')\n",
"host.set_ylabel('BTC/USDT')\n",
"par1.set_ylabel('positive')\n",
"par2.set_ylabel('negative')\n",
"\n",
"host.plot(close, label = 'BTC/USDT')\n",
"par1.plot(positive, label = 'positive')\n",
"par2.plot(negative, label = 'negative')\n",
"host.legend()\n",
"plt.xticks(\n",
" np.arange(len(timestamp))[::30], timestamp[::30], rotation = '45', ha = 'right'\n",
" )\n",
"plt.legend()\n",
"plt.show()"
]
},
{
"cell_type": "code",
"execution_count": 4,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"<div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>0</th>\n",
" <th>1</th>\n",
" <th>2</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>0.947020</td>\n",
" <td>0.672896</td>\n",
" <td>0.327104</td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>0.955190</td>\n",
" <td>0.595100</td>\n",
" <td>0.404900</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>0.960986</td>\n",
" <td>0.596702</td>\n",
" <td>0.403298</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>0.987212</td>\n",
" <td>0.577972</td>\n",
" <td>0.422028</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td>1.000000</td>\n",
" <td>0.585342</td>\n",
" <td>0.414658</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>"
],
"text/plain": [
" 0 1 2\n",
"0 0.947020 0.672896 0.327104\n",
"1 0.955190 0.595100 0.404900\n",
"2 0.960986 0.596702 0.403298\n",
"3 0.987212 0.577972 0.422028\n",
"4 1.000000 0.585342 0.414658"
]
},
"execution_count": 4,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"minmax = MinMaxScaler().fit(df.iloc[:, 1:2].astype('float32'))\n",
"df_log = minmax.transform(df.iloc[:, 1:2].astype('float32'))\n",
"df_log = pd.DataFrame(df_log)\n",
"df_log[1] = df['positive']\n",
"df_log[2] = df['negative']\n",
"df_log.head()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Model definition\n",
"\n",
"This example is using model 17.cnn-seq2seq, if you want to use another model, need to tweak a little bit, but I believe it is not that hard."
]
},
{
"cell_type": "code",
"execution_count": 5,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((339, 4), (309, 3), (30, 3))"
]
},
"execution_count": 5,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"num_layers = 1\n",
"size_layer = 128\n",
"epoch = 200\n",
"dropout_rate = 0.75\n",
"test_size = 3 * 10 # timestamp every 20 minutes, and I want to test on last 12 hours\n",
"learning_rate = 1e-3\n",
"timestamp = test_size\n",
"\n",
"df_train = df_log.iloc[:-test_size]\n",
"df_test = df_log.iloc[-test_size:]\n",
"df.shape, df_train.shape, df_test.shape"
]
},
{
"cell_type": "code",
"execution_count": 6,
"metadata": {},
"outputs": [],
"source": [
"def encoder_block(inp, n_hidden, filter_size):\n",
" inp = tf.expand_dims(inp, 2)\n",
" inp = tf.pad(\n",
" inp,\n",
" [\n",
" [0, 0],\n",
" [(filter_size[0] - 1) // 2, (filter_size[0] - 1) // 2],\n",
" [0, 0],\n",
" [0, 0],\n",
" ],\n",
" )\n",
" conv = tf.layers.conv2d(\n",
" inp, n_hidden, filter_size, padding = 'VALID', activation = None\n",
" )\n",
" conv = tf.squeeze(conv, 2)\n",
" return conv\n",
"\n",
"\n",
"def decoder_block(inp, n_hidden, filter_size):\n",
" inp = tf.expand_dims(inp, 2)\n",
" inp = tf.pad(inp, [[0, 0], [filter_size[0] - 1, 0], [0, 0], [0, 0]])\n",
" conv = tf.layers.conv2d(\n",
" inp, n_hidden, filter_size, padding = 'VALID', activation = None\n",
" )\n",
" conv = tf.squeeze(conv, 2)\n",
" return conv\n",
"\n",
"\n",
"def glu(x):\n",
" return tf.multiply(\n",
" x[:, :, : tf.shape(x)[2] // 2],\n",
" tf.sigmoid(x[:, :, tf.shape(x)[2] // 2 :]),\n",
" )\n",
"\n",
"\n",
"def layer(inp, conv_block, kernel_width, n_hidden, residual = None):\n",
" z = conv_block(inp, n_hidden, (kernel_width, 1))\n",
" return glu(z) + (residual if residual is not None else 0)\n",
"\n",
"class Model:\n",
" def __init__(\n",
" self,\n",
" learning_rate,\n",
" num_layers,\n",
" size,\n",
" size_layer,\n",
" output_size,\n",
" kernel_size = 3,\n",
" n_attn_heads = 16,\n",
" dropout = 0.9,\n",
" ):\n",
" self.X = tf.placeholder(tf.float32, (None, None, size))\n",
" self.Y = tf.placeholder(tf.float32, (None, output_size))\n",
"\n",
" encoder_embedded = tf.layers.dense(self.X, size_layer)\n",
"\n",
" e = tf.identity(encoder_embedded)\n",
" for i in range(num_layers):\n",
" z = layer(\n",
" encoder_embedded,\n",
" encoder_block,\n",
" kernel_size,\n",
" size_layer * 2,\n",
" encoder_embedded,\n",
" )\n",
" z = tf.nn.dropout(z, keep_prob = dropout)\n",
" encoder_embedded = z\n",
"\n",
" encoder_output, output_memory = z, z + e\n",
" g = tf.identity(encoder_embedded)\n",
"\n",
" for i in range(num_layers):\n",
" attn_res = h = layer(\n",
" encoder_embedded,\n",
" decoder_block,\n",
" kernel_size,\n",
" size_layer * 2,\n",
" residual = tf.zeros_like(encoder_embedded),\n",
" )\n",
" C = []\n",
" for j in range(n_attn_heads):\n",
" h_ = tf.layers.dense(h, size_layer // n_attn_heads)\n",
" g_ = tf.layers.dense(g, size_layer // n_attn_heads)\n",
" zu_ = tf.layers.dense(\n",
" encoder_output, size_layer // n_attn_heads\n",
" )\n",
" ze_ = tf.layers.dense(output_memory, size_layer // n_attn_heads)\n",
"\n",
" d = tf.layers.dense(h_, size_layer // n_attn_heads) + g_\n",
" dz = tf.matmul(d, tf.transpose(zu_, [0, 2, 1]))\n",
" a = tf.nn.softmax(dz)\n",
" c_ = tf.matmul(a, ze_)\n",
" C.append(c_)\n",
"\n",
" c = tf.concat(C, 2)\n",
" h = tf.layers.dense(attn_res + c, size_layer)\n",
" h = tf.nn.dropout(h, keep_prob = dropout)\n",
" encoder_embedded = h\n",
"\n",
" encoder_embedded = tf.sigmoid(encoder_embedded[-1])\n",
" self.logits = tf.layers.dense(encoder_embedded, output_size)\n",
" self.cost = tf.reduce_mean(tf.square(self.Y - self.logits))\n",
" self.optimizer = tf.train.AdamOptimizer(learning_rate).minimize(\n",
" self.cost\n",
" )\n",
" \n",
"def calculate_accuracy(real, predict):\n",
" real = np.array(real) + 1\n",
" predict = np.array(predict) + 1\n",
" percentage = 1 - np.sqrt(np.mean(np.square((real - predict) / real)))\n",
" return percentage * 100\n",
"\n",
"def anchor(signal, weight):\n",
" buffer = []\n",
" last = signal[0]\n",
" for i in signal:\n",
" smoothed_val = last * weight + (1 - weight) * i\n",
" buffer.append(smoothed_val)\n",
" last = smoothed_val\n",
" return buffer"
]
},
{
"cell_type": "code",
"execution_count": 7,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"WARNING: Logging before flag parsing goes to stderr.\n",
"W0818 16:34:24.824237 140007582447424 deprecation.py:323] From <ipython-input-6-6c0655f4345e>:55: dense (from tensorflow.python.layers.core) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use keras.layers.dense instead.\n",
"W0818 16:34:24.831443 140007582447424 deprecation.py:506] From /usr/local/lib/python3.6/dist-packages/tensorflow/python/ops/init_ops.py:1251: calling VarianceScaling.__init__ (from tensorflow.python.ops.init_ops) with dtype is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Call initializer instance with the dtype argument instead of passing it to the constructor\n",
"W0818 16:34:25.094202 140007582447424 deprecation.py:323] From <ipython-input-6-6c0655f4345e>:13: conv2d (from tensorflow.python.layers.convolutional) is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Use `tf.keras.layers.Conv2D` instead.\n",
"W0818 16:34:25.236837 140007582447424 deprecation.py:506] From <ipython-input-6-6c0655f4345e>:66: calling dropout (from tensorflow.python.ops.nn_ops) with keep_prob is deprecated and will be removed in a future version.\n",
"Instructions for updating:\n",
"Please use `rate` instead of `keep_prob`. Rate should be set to `rate = 1 - keep_prob`.\n"
]
}
],
"source": [
"tf.reset_default_graph()\n",
"modelnn = Model(\n",
" learning_rate, num_layers, df_log.shape[1], size_layer, df_log.shape[1], \n",
" dropout = dropout_rate\n",
")\n",
"sess = tf.InteractiveSession()\n",
"sess.run(tf.global_variables_initializer())"
]
},
{
"cell_type": "code",
"execution_count": 8,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"train loop: 100%|██████████| 200/200 [00:40<00:00, 5.17it/s, acc=98, cost=0.000637] \n"
]
}
],
"source": [
"from tqdm import tqdm\n",
"\n",
"pbar = tqdm(range(epoch), desc = 'train loop')\n",
"for i in pbar:\n",
" init_value = np.zeros((1, num_layers * 2 * size_layer))\n",
" total_loss, total_acc = [], []\n",
" for k in range(0, df_train.shape[0] - 1, timestamp):\n",
" index = min(k + timestamp, df_train.shape[0] - 1)\n",
" batch_x = np.expand_dims(\n",
" df_train.iloc[k : index, :].values, axis = 0\n",
" )\n",
" batch_y = df_train.iloc[k + 1 : index + 1, :].values\n",
" logits, _, loss = sess.run(\n",
" [modelnn.logits, modelnn.optimizer, modelnn.cost],\n",
" feed_dict = {modelnn.X: batch_x, modelnn.Y: batch_y},\n",
" ) \n",
" total_loss.append(loss)\n",
" total_acc.append(calculate_accuracy(batch_y[:, 0], logits[:, 0]))\n",
" pbar.set_postfix(cost = np.mean(total_loss), acc = np.mean(total_acc))"
]
},
{
"cell_type": "code",
"execution_count": 9,
"metadata": {},
"outputs": [],
"source": [
"future_day = test_size\n",
"\n",
"output_predict = np.zeros((df_train.shape[0] + future_day, df_train.shape[1]))\n",
"output_predict[0] = df_train.iloc[0]\n",
"upper_b = (df_train.shape[0] // timestamp) * timestamp\n",
"\n",
"for k in range(0, (df_train.shape[0] // timestamp) * timestamp, timestamp):\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(\n",
" df_train.iloc[k : k + timestamp], axis = 0\n",
" )\n",
" },\n",
" )\n",
" output_predict[k + 1 : k + timestamp + 1] = out_logits\n",
"\n",
"if upper_b != df_train.shape[0]:\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: np.expand_dims(df_train.iloc[upper_b:], axis = 0)\n",
" },\n",
" )\n",
" output_predict[upper_b + 1 : df_train.shape[0] + 1] = out_logits\n",
" future_day -= 1"
]
},
{
"cell_type": "code",
"execution_count": 10,
"metadata": {},
"outputs": [],
"source": [
"output_predict_negative = output_predict.copy()\n",
"output_predict_positive = output_predict.copy()"
]
},
{
"cell_type": "code",
"execution_count": 11,
"metadata": {},
"outputs": [],
"source": [
"for i in range(future_day):\n",
" o = output_predict[-future_day - timestamp + i:-future_day + i].copy()\n",
" o = np.expand_dims(o, axis = 0)\n",
" \n",
" o_negative = output_predict_negative[-future_day - timestamp + i:-future_day + i].copy()\n",
" o_negative = np.expand_dims(o_negative, axis = 0)\n",
" o_negative[:, :, 1] = 0.0\n",
" o_negative[:, :, 2] = 1.0\n",
" \n",
" o_positive = output_predict_positive[-future_day - timestamp + i:-future_day + i].copy()\n",
" o_positive = np.expand_dims(o_positive, axis = 0)\n",
" o_positive[:, :, 1] = 1.0\n",
" o_positive[:, :, 2] = 0.0\n",
" \n",
" # original without any consensus\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: o\n",
" },\n",
" )\n",
" output_predict[-future_day + i] = out_logits[-1]\n",
" \n",
" # negative consensus\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: o_negative\n",
" },\n",
" )\n",
" output_predict_negative[-future_day + i] = out_logits[-1]\n",
" \n",
" # positive consensus\n",
" out_logits = sess.run(\n",
" modelnn.logits,\n",
" feed_dict = {\n",
" modelnn.X: o_positive\n",
" },\n",
" )\n",
" output_predict_positive[-future_day + i] = out_logits[-1]"
]
},
{
"cell_type": "code",
"execution_count": 12,
"metadata": {},
"outputs": [],
"source": [
"output_predict_original = minmax.inverse_transform(output_predict[:,:1])\n",
"output_predict_negative = minmax.inverse_transform(output_predict_negative[:,:1])\n",
"output_predict_positive = minmax.inverse_transform(output_predict_positive[:,:1])"
]
},
{
"cell_type": "code",
"execution_count": 19,
"metadata": {},
"outputs": [],
"source": [
"deep_future = anchor(output_predict_original[:, 0], 0.7)\n",
"deep_future_negative = anchor(output_predict_negative[:, 0], 0.7)\n",
"deep_future_positive = anchor(output_predict_positive[:, 0], 0.7)"
]
},
{
"cell_type": "code",
"execution_count": 14,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"((339, 4), 339)"
]
},
"execution_count": 14,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"df.shape, len(deep_future_negative)"
]
},
{
"cell_type": "code",
"execution_count": 15,
"metadata": {},
"outputs": [],
"source": [
"df_train = minmax.inverse_transform(df_train)\n",
"df_test = minmax.inverse_transform(df_test)"
]
},
{
"cell_type": "code",
"execution_count": 20,
"metadata": {},
"outputs": [
{
"data": {
"image/png": "iVBORw0KGgoAAAANSUhEUgAAA4kAAAGFCAYAAABQc7YHAAAABHNCSVQICAgIfAhkiAAAAAlwSFlzAAALEgAACxIB0t1+/AAAADh0RVh0U29mdHdhcmUAbWF0cGxvdGxpYiB2ZXJzaW9uMy4xLjEsIGh0dHA6Ly9tYXRwbG90bGliLm9yZy8QZhcZAAAgAElEQVR4nOzde1zO5//A8dd93xU6IinlVA43E0vlkDYbY8yMmY0xTRszxwxj5pTmMOdThFFO5fybUznv8HWYIZmzm+QwDBWhbpXuu98f5U46CKnk/Xw89lh9rutzfd6f933n0bvr+ly3IjU1FSGEEEIIIYQQAkBZ2AEIIYQQQgghhCg6pEgUQgghhBBCCGEgRaIQQgghhBBCCAMpEoUQQgghhBBCGEiRKIQQQgghhBDCQIpEIYQQQgghhBAGRvk1kFqtngZ0BKoCdTUazUm1Wm0NrACqAcnAeeBbjUYTnX5OY2AhUAq4BHTTaDS3XqRNCCGEEEIIIcTzy7ciEdgIzAb2PnYsFZii0Wj+BFCr1VOBSUAPtVqtBIIBb41Gs0+tVo9Kb/v6edvyGGcJoAHwH6B7oTsWQgghhBBCiFePCqgAHAaSnmzMtyJRo9HsA1Cr1Y8fuw38+Vi3v4E+6V+7AYmPzgMWkDYr+PULtOVFAzIXskIIIYQQQgjxOnob2PfkwQJ7JjF9BrAPsDn9UGXg8qN2jUYTAyjVanXZF2jLi/9e5D6EEEIIIYQQopjItjbKz+WmT+MPxANzC/Ca2dEBxMbGo9enFnIomdnYWBAdfb+wwygSJBeZST4yk3xkkFxkJvnIILnITPKRQXKRmeQjg+Qis+KcD6VSgbW1OeTw+F2BzCSmb2pTA+is0Wj06YevAFUe61MO0KcvUX3eNiGEEEIIIYQQL+ClF4lqtXoiac8RfqzRaB5/KPIIUEqtVr+V/n1vYN0LtgkhhBBCCCGEeAH5+REYc4BPADtgt1qtjgU6AT8C54C/0je1uajRaDpoNBq9Wq32Ahaq1eqSpH+UBcDztgkhhBBCiKJFp0vhzp1oUlKSCzuUbN26pUSv1z+942tAcpFZcciHkZEJZcrYoFI9W9mXn7ub+gA+2TQpcjnnL6BufrYJIYQQQoii486daEqWNMXMzA6FIsdfCwuNkZGSlJRXuxDIL5KLzF71fKSmppKQcI87d6IpV67CM51bYLubCiGEEEKI109KSjJmZpZFskAUojhTKBSYmVk+1yy+FIlCCCGEEOKlkgJRiMLxvD97UiQKIYQQQgghhDCQIlEIIYQQQrw2AgMX8vDhw5c2xv379wkJWfZC4+dVREQ4PXp4Fci1xOtFisRXyIIFc1m/fk1hhyGEEEII8cpasmTRCxeJuY0RH3+flSuX53huSkrKC11biIKQb7ubipcrNHQzY8aMwNTUjGbNWmBtbQ1A9H+XuX3rOim6ZNT1mmBkZFzIkQohhBBCFE3Tp08GoE+fr1EolPj7L8TERMXMmdO5cOE8ycnJ1K/vzoABg1CpVAQF/cLu3TswMSmBQgFz5izkl18CsoxhYWFhuMaMGZOJj4/H27srJUuWZMGCIPr370WNGmpOnTqBpaUl06bN4cCBfSxfHkRSUjLGxsYMGDAYZ+e6RESEM2fODN54ow6nTp0AFPj5TaRqVUcAfvklgN9+24mFhSX167sVeA7F60GKxFfAf/9dZ8iQAVSvXoMLFyJZsGAuI0f6snB2L/yUq0lRpfUbtL0ZP47YVLjBCiGEEELkYs0aI1atejl/1O7S5SGdO+c8UzdkyA9s2LCO+fODMDU1BWDy5PG4uLgyfPho9Ho9fn6jCAvbzLvvNmft2pVs2rSdEiVKotUmYGJSItsxHjd48A/07OnF0qUrMx2/fv0qAQGLMTIy4tq1qyxdGsiMGf6YmZkTFXWB77/34ddfwwC4ePECI0aMYdiwkSxbFsiyZYH4+o5n37497N+/hyVLVlKiRAl+/PH7fMyeEBmkSCziUlNTGTCgD0lJSSxfvprJkyewePFC3GrbMoHVvBljSmuL99io3cFm1V5+0OlQqlSFHbYQQgghxCth377/cfr0SVavDgEgMTGR8uVtMTMzx8GhEuPG+dKwYWOaNHkbU1Oz575Oy5atMTJK+9X74MEDXLt2lX79ehnadTodt2/HAlC5chVq1qwFQJ06ddm/fy8AR4+G07x5S0Nx2rZte5YtC3zumITIiRSJRdw//0SwZ88fjB8/ierVazB48DD2/7mdkZrhlFEqWfDFTqrUrIdi+peML7WR3ZsX8X6H3oUdthBCCCFEtjp3Tsl1tq+gpaamMnHiNBwcKmZpW7hwCSdOHEvfIKYb06f7U716jee6TqlSGbOOqampNGrkwejRP2Xpd+nSRUxMShi+VyqV6HS657qmEM9LNq4p4n79dT0mJiZ07twVgEoO9jh2SuWmmZ7mUW/zQJ+2XKPrFxMwTYYNRwIKM1whhBBCiCLN1NSMhIR4w/dvv/0OwcHLDIVYXFwc169fQ6tNIC4ujvr13ejR41ucnKoRFXUh2zEeZ2ZmRmJiYq4b1DRs2JiDBw8YxgM4c+bUU2N3dW3A77/v5sGDB+h0OrZu3ZynexbiWclMYhGm0+nYtOlXmjdviZVVaVJSHtJvSkMOOzzgq3MNWbMjnFWbPFi8eDlt27ajebQ9u6wvcfd2NFZlbQo7fCGEEEKIIufzz7/Ax6c3JUqUxN9/Id999z3+/rPw9u6CQqHA2NgEH58hGBkZMXLkMJKTk9Dr9dSsWYt33mmW7RiPb1xjaWnF++9/QPfun2NhYcmCBUFZYqhUqTJjxoxj0qRxJCUlkZLykLp136R27Tq5xu7p+TYnTx7H27uLYeOa6Ojo/E2QEIAiNTW1sGMoaFWBi7Gx8ej1RevebWwsiI6+b/h+//69dOjwIXNn+0PSRcLOh7DN4Sb9Yj3wHb2D27dj+eKLTpw+fZKNG7eiidiCT9IM+l9tzpiJG/M1tgO/rWPy/77DOFWFbao1/T+dSy0Xz3y9xuOezMXrTvKRmeQjg+QiM8lHBslFZpKPDAWdixs3LmNnV6XArvesjIyUpKToCzuMIkFykVlxyUd2P4NKpQJra3MAR+DSk+fIctMi7Ndf11OlQjnG/+tDf+0M/ih/k+43nBk9YisAZctas3z5amxsytOlS0d+HLeQynFKQs3+x/mTh/Mtjhv/RjHwYC/OWN0nxljLxvIX+Gn9839w678Xsl9O8XtoECGLhj/3uEIIIYQQQogXJ0ViEZWcnExo6EbqNTcm2jSVqapenPC6wNSf/sq0e6mNjQ0rV67HyMiYt5u+S8sb7xFjqsNr0/tojh0w9LsceZLvx3jw49imBM7rz40rkble/+7taC5q/iEpUcvAhc25aqFjVtUJ/DEmhq6xdfndLoYTh3975vtau3Qsbjs8mDChreHYX7vW0HlUFT6/8h2DHgZw7O+d2Z6bmprKlSuXiYw8/8zXFUIIIYQQQuSNPJNYBMXGxjJq1A9Ym6Wy3ek/2t+oQvfx03LsX7OmmhMnzqFQKAgPP8SJocc50+4mn+xsTev/q01FCyd+UYQRVz5tea1O8Q+LV6xk89dHsamQdfmHXqfjU/86HLNJRKkHvT0MuduCDzoOAKDHJ9MI3tOKJZt/ZEaDQ9y/G8u1i2fztPx0U+RysIfZZfZQeubXxGlvEWC5B7Oy8NXNeiy3OU7INj9afNTRcI5Op+Pnn8exZs1Kbt68gbGxMaGhOw0fIHs58iQOVdUYGb2cz1wSQgghhBDidSIziUXMzp3beOstdzZt+hXHFiqM9DCk8+KnnqdQKACoV8+FE1H36HLlQ5zvlmaVzWl+NgulYkIJfq0dyPkvrzDTuC+XrFIYNK85+my2VN60agrHbBL55HoVvKPf5If7HzB02DpDu/pND1retGNzmbPs2R7Mh3Nq0nzvB8yY0iXXGG9cieR/5WP47LoTHjcs8Cuxntll9vDOTWv++OQQk/320fRGWbaan+RB+o5hiYmJfPONN3PmzMDV1Z2ff55K+fK29Or1Fffu3WVjyGQ8tzZhmN/Lez5SCCGEEEKI14kUiUXIL7/8wpdfdqF+rUp08a7B7iqxdL5dlxp1G+V5DBMTE9zcGnDg+FXWjL/M4Y8OsbjMaEJ/uITHe59hblmaL76ZRL+7b7PTPpqpUz7NMkawJoDyCQqmDdvLJL+9DPlhTaYlrgBeDYZyryR0iuzLTdOHuEWbM8k8jN6j6nL80O5sY1u1xpdkI/i0yUAC+/5N22v2jH7QgWC/SByqpn1gbHvHrtwyS2XR3GHcu3eXzz//hNDQTYwb9zPLlq2ExBPYtoulWoVkRg3qxvc3J6BTwlqbs5yK+N8zZFsIIYQQQgiRHdndtIgYO7w9hxV/c90qiWuWepR6ePeGNbP7/IGtQ9VnGmvy5AnMnDmV8+evYGFhmW0fvU5HV9/q7LGNJcj+Z1p37Aek7WLaXtODPjGN8RuT/bOBkPZ84Mcj7LlXIpmZrVdRp/47jJjYjGW2JwGodseIVjp3ensHYFe5OgAf+lbgjnES+36MNhSdqamp+PvP4u7dOHr16ktpKwsazbanWrwFt/dVRqM5g7//Ajp27ET8vTg851flppkevRKMdVBOq6Di9vIc73CT967ZsnRy8XxeUXbly0zykUFykZnkI4PkIjPJRwbZ3TSz4rKDZX6QXGRWXPIhu5u+wu4k/0eSiZ7692357s677H33N1aPv/jMBSJA48ZN0Ov1HD58MMc+SpWKWd/+hsN9FT9EjuTyueMABP0xllIPodeXc3K9xqFDBzkQlMCpgIccOh6JcYkSTPX7i5311zLkbgtskkoRUO5vPH51pfeousyf9Q2HbRN4X++eqUD08xvN+PG++PvPpEGDugzw6Uvja5XYZ3eXMiZ3CAlZR8eOnQBYtLAv/1nomWU2kB8TPsL1lhmjK/gyac7/0eK8Ldsr3WSLine truncated
"text/plain": [
"<Figure size 1080x360 with 1 Axes>"
]
},
"metadata": {
"needs_background": "light"
},
"output_type": "display_data"
}
],
"source": [
"timestamp = df['timestamp'].tolist()\n",
"pad_test = np.pad(df_test[:,0], (df_train.shape[0], 0), 'constant', constant_values=np.nan)\n",
"\n",
"plt.figure(figsize = (15, 5))\n",
"plt.plot(pad_test, label = 'test trend', c = 'blue')\n",
"plt.plot(df_train[:,0], label = 'train trend', c = 'black')\n",
"plt.plot(deep_future, label = 'forecast without consensus')\n",
"plt.plot(deep_future_negative, label = 'forecast with negative consensus', c = 'red')\n",
"plt.plot(deep_future_positive, label = 'forecast with positive consensus', c = 'green')\n",
"plt.legend()\n",
"plt.xticks(\n",
" np.arange(len(timestamp))[::30], timestamp[::30], rotation = '45', ha = 'right'\n",
")\n",
"plt.show()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## What we can observe\n",
"\n",
"1. The model learn, if positive and negative sentiments increasing, both will increase the price. That is why, using positive consensus or negative consensus caused price going up.\n",
"2. Volatility of price is higher if negative sentiment is higher, still positive volatility.\n",
"3. Momentum of price is higher if negative sentiment is higher, still positive momentum."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.6.8"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -0,0 +1,45 @@
# Copyright 2017 Google Inc.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# ==============================================================================
"""DNC util ops and modules."""
from __future__ import absolute_import
from __future__ import division
from __future__ import print_function
import numpy as np
import tensorflow as tf
def batch_invert_permutation(permutations):
"""Returns batched `tf.invert_permutation` for every row in `permutations`."""
with tf.name_scope('batch_invert_permutation', values=[permutations]):
unpacked = tf.unstack(permutations)
inverses = [tf.invert_permutation(permutation) for permutation in unpacked]
return tf.stack(inverses)
def batch_gather(values, indices):
"""Returns batched `tf.gather` for every row in the input."""
with tf.name_scope('batch_gather', values=[values, indices]):
unpacked = zip(tf.unstack(values), tf.unstack(indices))
result = [tf.gather(value, index) for value, index in unpacked]
return tf.stack(result)
def one_hot(length, index):
"""Return an nd array of given `length` filled with 0s and a 1 at `index`."""
result = np.zeros(length)
result[index] = 1
return result