diff --git a/.drone.yml b/.drone.yml index d48cbda..b5cbefa 100644 --- a/.drone.yml +++ b/.drone.yml @@ -4,6 +4,7 @@ pipeline: environment: - CUB_INC=/extra/cub-1.8.0/ commands: + - git clone --single-branch --depth 1 https://github.com/NVIDIA/cuda-samples.git /extra/samples - export LD_LIBRARY_PATH=/usr/local/cuda/lib64/stubs:$${LD_LIBRARY_PATH} - cd user_ops - cmake . -DPYTHON_EXECUTABLE=python2 diff --git a/.gitignore b/.gitignore index f5c4259..4e25473 100644 --- a/.gitignore +++ b/.gitignore @@ -134,3 +134,5 @@ tensorflow_config.txt .settings .local/ local/ + +train_log/ diff --git a/README.md b/README.md index d2fe833..12a0ebf 100644 --- a/README.md +++ b/README.md @@ -18,30 +18,51 @@ This repository contains the source code of our FlexConv Layer from our 2018 ACC

- -Example - Usage +Provided novel operations ------------------- +In the following we summarize the operations described in our paper along with a highly tuned (online, exhaustive) nearest neighbor search layer for 3d point-clouds. +All layers follow the `tf.layers` interface and can be directly used in your TensorFlow model. We further provide unit-tests to verify the correctness of our implementation. -We provide GPU-tailored CUDA implementations of our novel FlexConv, FlexPool, FlexDeconv operations in TensorFlow. +```python +# some point-cloud data +features = np.random.randn(B, Din, N).astype(np.float32) +positions = np.random.randn(B, Dp, N).astype(np.float32) -```console -user@host $ cd user_ops -user@host $ cmake . -DPYTHON_EXECUTABLE=python2 && make -j -user@host $ cd .. -user@host $ python example.py +# To find neighborhoods of K neighbors (not used in our paper, just for your convenience): +neighborhoods = knn_bruteforce(positions, K=8) + +# To apply our Flex-Convolution operation to a given set of points with some input-features, position and neighborhood information: +new_features = flex_convolution(features, positions, neighborhoods, out_channels=32) +new_features = flex_convolution_transpose(features, positions, neighborhoods, out_channels=32) + +# To apply max-pooling for each neighborhood: +new_features = flex_pooling(features, neighborhoods) ``` - -Experiments +Build Instructions ------------------- -Deep learning on point-clouds is a complex matter and our codebase reflects that complexity. -We are currently working on refactoring our research implementation to ease the usage. Therefore, -`layers.py` contains a Keras/tf.layers compatible implementation. We will add the models later. +We provide GPU-tailored CUDA implementations of our novel FlexConv, FlexPool, FlexDeconv, NearestNeighbor operations in TensorFlow, which require a compilation/linking step. To build our operations just use + +```console +user@host $ pip install tensorflow-gpu --user # optional if not yet installed +user@host $ cd user_ops +user@host $ cmake . -DPYTHON_EXECUTABLE=python2 && make -j # switch the python version when necessary +user@host $ python test_all.py # run all unit-tests to verify the operations +user@host $ cd .. +user@host $ python example.py # fully functional toy-example +``` + +Deep learning on point-clouds is a complex matter and an active research area. Hence, our internal codebase reflects that complexity and we try our best to provide a usable implementation. +To provide a simple training example, we demonstrate training on a very basic 3D-MNIST dataset which deliberately omits fancy parts to give you an idea how to actually train such a model with our operations: + +```console +user@host $ python basic_mnist_3d.py --gpu 0 +``` ### Benchmark -We benchmarked the inference time of *entire* network on the 2D-3D-S dataset and with a recent test on a NVIDIA V100 GPU, we were able to process ~18 Million Points. +We benchmarked the inference time of *entire* network on the 2D-3D-S dataset and with a recent test on a NVIDIA V100 GPU, we were able to process ~18 Million Points (the paper stated 7 Million Points on 1080 GTX).

diff --git a/basic_mnist_3d.py b/basic_mnist_3d.py new file mode 100644 index 0000000..e4eaf4b --- /dev/null +++ b/basic_mnist_3d.py @@ -0,0 +1,268 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- + +# Copyright 2017 ComputerGraphics Tuebingen. All Rights Reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# ============================================================================== +# Authors: Fabian Groh, Patrick Wieschollek, Hendrik P.A. Lensch + + +""" +Run 3D-MNIST classification using Flex-Convolutions, without any fancy parts. + +Example-output: + + python mnist_3d.py --gpu 0 -fusion pooling + + [@base.py:282] Epoch 56 (global_step 210000) finished, time:23 minutes 52 seconds. + [@saver.py:77] Model saved to train_log/conv_position_pooling/model-210000. + 100%|##############################|625/625[00:55<00:00,11.28it/s] + [@monitor.py:459] accuracy: 0.94699 + [@monitor.py:459] learning_rate: 0.0001 + [@monitor.py:459] train_error: 0.053013 + [@monitor.py:459] validation_accuracy: 0.9327 + [@monitor.py:459] validation_cross_entropy_loss: 0.209 + [@group.py:48] Callbacks took 55.513 sec in total. InferenceRunner: 55.4 seconds + [@base.py:272] Start Epoch 57 ... + + + python mnist_3d.py --gpu 0 -fusion conv + + [@base.py:282] Epoch 40 (global_step 150000) finished, time:34 minutes 49 seconds. + [@saver.py:77] Model saved to train_log/conv_position_conv/model-150000. + 100%|##############################|625/625[00:57<00:00,10.89it/s] + [@monitor.py:459] accuracy: 0.89019 + [@monitor.py:459] learning_rate: 0.0001 + [@monitor.py:459] train_error: 0.10981 + [@monitor.py:459] validation_accuracy: 0.8748 + [@monitor.py:459] validation_cross_entropy_loss: 0.38396 + +This implementation is based on Tensorpack +- http://tensorpack.com/ +- https://github.com/tensorpack/tensorpack/ + +""" + +import os +import argparse +import tensorflow as tf +import numpy as np +import cv2 + +from tensorpack import * +from layers import flex_convolution, flex_pooling, knn_bruteforce + + +enable_argscope_for_module(tf.layers) + +TOTAL_BATCH_SIZE = 16 +BATCH_SIZE = 16 +SHAPE = 28 +CHANNELS = 3 + +USE_POOLING = False + +PC = {'num': 1024, 'dp': 3} + + +class Digit2Cloud(RNGDataFlow): + """ A very basic 2D-MNIST to 3D-MNIST sampler on a regular grid. + """ + def __init__(self, incoming_df, num=1024): + super(Digit2Cloud, self).__init__() + self.incoming_df = incoming_df + self.num = num + + def reset_state(self): + super(Digit2Cloud, self).reset_state() + self.incoming_df.reset_state() + + def __len__(self): + return self.incoming_df.__len__() + + def map(self, dp, num=1024): + digit = dp[0] + + # detect edges + def auto_canny(image, sigma=0.33): + v = np.median(image) + lower = int(max(0, (1.0 - sigma) * v)) + upper = int(min(255, (1.0 + sigma) * v)) + edged = cv2.Canny(image, lower, upper) + return edged + + digit = np.tile(np.expand_dims(digit, axis=-1), [1, 1, 3]) + digit = cv2.resize(digit, (32, 32)) + + img = (255 * digit).astype(np.uint8) + blurred = cv2.GaussianBlur(img, (3, 3), 0) + canny = auto_canny(blurred) + edge_x, edge_y = np.nonzero(canny > 0) + + xlen = np.max(edge_x) - np.min(edge_x) + ylen = np.max(edge_y) - np.min(edge_y) + + face_x, face_y = np.nonzero(img[:, :, 0] > 0) + points = [] + + depth = max(ylen, xlen) + padding = (32 - depth) / 2. + + def z_dim(x): + return x * depth + padding + + # start sampling (just extrude digits) + for i in range(1024): + choice = self.rng.randint(2 + 4) + if choice > 1: + idx = self.rng.randint(len(edge_x)) + z = self.rng.rand() + points.append([edge_x[idx], edge_y[idx], z_dim(z)]) + else: + idx = self.rng.randint(len(face_x)) + z = self.rng.randint(2) + points.append([face_x[idx], face_y[idx], z_dim(z)]) + return [np.array(points).T, dp[1]] + + def __iter__(self): + for dp in self.incoming_df: + dp = self.map(dp, self.num) + yield dp + + +class Model(ModelDesc): + def inputs(self): + """Inputs are + - pointcloud [batch, dim_position, num_points] + - label [batch] + """ + return [tf.placeholder(tf.float32, (None, PC['dp'], PC['num']), 'positions'), + tf.placeholder(tf.int32, (None,), 'label')] + + def build_graph(self, positions, label): + + positions = positions / 16. - 1 + # initial features are the position them self + features = positions + neighbors = knn_bruteforce(positions, K=16) + + x = features + + def subsample(x): + # probably too simplistic, just kick out 3 of 4 points randomly + # see our paper IDISS approach in the paper for better sub-sampling + n = x.shape.as_list()[-1] + return x[:, :, :n // 4] + + # similar to traditional networks + for stage in range(4): + if stage > 0: + x = flex_pooling(x, neighbors) + x = subsample(x) + positions = subsample(positions) + neighbors = knn_bruteforce(positions, K=16) + + x = flex_convolution(x, positions, neighbors, 64 * + (stage + 1), activation=tf.nn.relu) + x = flex_convolution(x, positions, neighbors, 64 * + (stage + 1), activation=tf.nn.relu) + + if USE_POOLING: + # either do max-pooling of all remaining points... + x = tf.expand_dims(x, axis=-1) + x = tf.layers.max_pooling2d(x, [1, 16], [1, 16]) + else: + # ... or do a flex-conv in (0, 0, 0) with all points as neighbors + positions = tf.concat([positions, positions[:, :, :1] * 0], axis=-1) + x = tf.concat([x, x[:, :, :1] * 0], axis=-1) + K = positions.shape.as_list()[-1] + neighbors = knn_bruteforce(positions, K=K) + x = flex_convolution(x, positions, neighbors, 1024, activation=tf.nn.relu) + x = x[:, :, -1:] + + # from now on just the code part we copied from the Tensorpack framework + x = tf.layers.flatten(x) + x = tf.layers.dense(x, 512, activation=tf.nn.relu, name='fc0') + logits = tf.layers.dense(x, 10, activation=tf.identity, name='fc1') + + cost = tf.nn.sparse_softmax_cross_entropy_with_logits( + logits=logits, labels=label) + cost = tf.reduce_mean(cost, name='cross_entropy_loss') + + correct = tf.cast(tf.nn.in_top_k(logits, label, 1), + tf.float32, name='correct') + accuracy = tf.reduce_mean(correct, name='accuracy') + + train_error = tf.reduce_mean(1 - correct, name='train_error') + summary.add_moving_summary(train_error, accuracy) + return cost + + def optimizer(self): + # nothing fancy here, just stick with the defaults + return tf.train.AdamOptimizer(1e-4) + + +def get_data(): + df_train = dataset.Mnist('train') + df_train = Digit2Cloud(df_train, num=PC['num']) + df_train = PrefetchDataZMQ(df_train, 2) + df_train = BatchData(df_train, BATCH_SIZE) + + df_val = dataset.Mnist('test') + df_val = Digit2Cloud(df_val, num=PC['num']) + df_val = PrefetchDataZMQ(df_val, 2) + df_val = BatchData(df_val, BATCH_SIZE) + return df_train, df_val + + +if __name__ == '__main__': + parser = argparse.ArgumentParser() + parser.add_argument('--gpu', help='comma separated list of GPU(s) to use.') + parser.add_argument('--load', help='load model') + parser.add_argument('--fusion', help='run sampling', default='', + choices=['pooling', 'conv']) + args = parser.parse_args() + + if args.gpu: + os.environ['CUDA_VISIBLE_DEVICES'] = args.gpu + + logger.set_logger_dir('train_log/fusion_%s' % (args.fusion)) + + dataset_train, dataset_test = get_data() + steps_per_epoch = len(dataset_train) + + USE_POOLING = (args.fusion == 'pooling') + + # get the config which contains everything necessary in a training + config = TrainConfig( + model=Model(), + # The input source for training. FeedInput is slow, this is just for demo purpose. + # In practice it's best to use QueueInput or others. See tutorials for details. + data=FeedInput(dataset_train), + callbacks=[ + ModelSaver(), # save the model after every epoch + InferenceRunner( # run inference(for validation) after every epoch + dataset_test, # the DataFlow instance used for validation + ScalarStats(['cross_entropy_loss', 'accuracy'])), + ], + extra_callbacks=[ + MovingAverageSummary(), + ProgressBar(['accuracy', 'cross_entropy_loss']), + MergeAllSummaries(), + RunUpdateOps() + ], + steps_per_epoch=steps_per_epoch, + max_epoch=100, + ) + launch_train_with_config(config, SimpleTrainer()) diff --git a/user_ops/CMakeLists.txt b/user_ops/CMakeLists.txt index 25da6b5..6a800cb 100644 --- a/user_ops/CMakeLists.txt +++ b/user_ops/CMakeLists.txt @@ -33,6 +33,7 @@ set(CUDA_NVCC_FLAGS "${CUDA_NVCC_FLAGS} -std=c++11 -O3 -Xptxas=-v --expt-relaxed # quick fix for drone-ci include_directories(SYSTEM "/usr/local/") +include_directories(SYSTEM "/extra/samples/Common/") # fix cgtuebingen include_directories(SYSTEM "/graphics/opt/opt_Ubuntu18.04/cuda/toolkit_9.2") diff --git a/user_ops/profile_flexconv.py b/user_ops/profile_flexconv.py index 3cc38e7..c84998a 100644 --- a/user_ops/profile_flexconv.py +++ b/user_ops/profile_flexconv.py @@ -21,8 +21,9 @@ import numpy as np import tensorflow as tf -from PointTestCase import FakePointCloud, random_values -from __init__ import flex_conv +from misc import FakePointCloud +from __init__ import flex_convolution + """ export LD_LIBRARY_PATH=/graphics/opt/opt_Ubuntu16.04/cuda/toolkit_9.0/cuda/extras/CUPTI/lib64:$LD_LIBRARY_PATH @@ -32,85 +33,42 @@ export LD_LIBRARY_PATH=/graphics/opt/opt_Ubuntu16.04/cuda/toolkit_9.0/cuda/extra np.random.seed(42) tf.set_random_seed(42) -N = 8 -TPC = FakePointCloud(8, 4096, N, 64, 64, 2, 1, N) +case = FakePointCloud(B=8, N=4096, K=8, Din=64, Dout=64, Dp=3) +case.init_ops(dtype=np.float32) -class PointTestCase(object): - - def __init__(self, data): - self.position = random_values([data.B, data.Dp, data.N]) - self.features = random_values([data.B, data.Din, data.N]) - - # make sure, each neighbor hood has no duplicates and first entry is point n - # THIS IS IMPORTANT!! - self.neighborhood = np.zeros((data.B, data.K, data.N), dtype=np.int32) - for b in range(data.B): - for n in range(data.N): - x = np.arange(data.N) - # does not support axis, hence the loop - np.random.shuffle(x) - offset = np.argwhere(x == n)[0][0] - # roll array such that n is first entry - x = np.roll(x, -offset) - self.neighborhood[b, :, n] = x[:data.K].astype(np.int32) - - self.neighborhood_ds = np.zeros((data.B, data.K, data.N2), dtype=np.int32) - for b in range(data.B): - for n in range(data.N2): - x = np.arange(data.N2) - # does not support axis, hence the loop - np.random.shuffle(x) - offset = np.argwhere(x == n)[0][0] - # roll array such that n is first entry - x = np.roll(x, -offset) - self.neighborhood[b, :, n] = x[:data.K].astype(np.int32) - - self.theta = random_values([data.Degree, data.Dp, data.Din, data.Dout]) - self.bias = random_values([data.Din, data.Dout]) - - def init_ops(self): - self.features_op = tf.Variable(self.features, name='f') - self.position_op = tf.Variable(self.position, name='p') - self.neighborhood_op = tf.Variable(self.neighborhood, name='n') - self.neighborhood_ds_op = tf.Variable(self.neighborhood_ds, name='m') - self.theta_op = tf.Variable(self.theta, name='t') - self.bias_op = tf.Variable(self.bias, name='b') - - -TestCase = PointTestCase(data=TPC) -TestCase.init_ops() - - -forward_op = flex_conv(TestCase.features_op, - TestCase.theta_op, TestCase.bias_op, TestCase.neighborhood_op, - TestCase.position_op, degree=1) +forward_op = flex_convolution(case.features_op, + case.position_op, + case.neighborhood_op, + case.theta_op, + case.bias_op) builder = tf.profiler.ProfileOptionBuilder opts = builder(builder.time_and_memory()).order_by('micros').build() -with tf.contrib.tfprof.ProfileContext('./.profiling_outputs/flex_conv') as pctx: +with tf.contrib.tfprof.ProfileContext('./.profiling_outputs/flex_convolution') as pctx: - with tf.Session(config=tf.ConfigProto(log_device_placement=True)) as sess: - sess.run(tf.global_variables_initializer()) + with tf.Session(config=tf.ConfigProto(log_device_placement=True)) as sess: + sess.run(tf.global_variables_initializer()) - back_prop = tf.gradients(forward_op, [TestCase.theta_op, TestCase.bias_op, TestCase.features_op]) - back_prop2 = tf.gradients(forward_op, [TestCase.theta_op, TestCase.bias_op, - TestCase.features_op, TestCase.position_op]) + back_prop = tf.gradients( + forward_op, [case.theta_op, case.bias_op, case.features_op]) + back_prop2 = tf.gradients(forward_op, [case.theta_op, case.bias_op, + case.features_op, case.position_op]) - # warmup - for i in range(2): - actual = sess.run([forward_op, back_prop]) + # warmup + for i in range(2): + actual = sess.run([forward_op, back_prop]) - # benchmark - for i in range(10): - pctx.trace_next_step() - pctx.dump_next_step() - _ = sess.run([forward_op]) - pctx.profiler.profile_operations(options=opts) + # benchmark + for i in range(10): + pctx.trace_next_step() + pctx.dump_next_step() + _ = sess.run([forward_op]) + pctx.profiler.profile_operations(options=opts) - for i in range(10): - pctx.trace_next_step() - pctx.dump_next_step() - _ = sess.run([back_prop]) - pctx.profiler.profile_operations(options=opts) + for i in range(10): + pctx.trace_next_step() + pctx.dump_next_step() + _ = sess.run([back_prop]) + pctx.profiler.profile_operations(options=opts) diff --git a/user_ops/test_flex_convolution.py b/user_ops/test_flex_convolution.py index 12d2262..8a0f7e4 100644 --- a/user_ops/test_flex_convolution.py +++ b/user_ops/test_flex_convolution.py @@ -114,23 +114,23 @@ class FlexConvTest(VerboseTestCase): actual, expected = self._backward_theta(use_gpu=True, dtype=np.float64) self.assertAllClose(actual, expected) + # float32 has some numerical instabilities due to summation + # central difference as derivatives are totally instable + # hence we just compare cpu and gpu outputs (float64 num-diff tests pass) def test_backward_features_gpu_float32(self, dtype=np.float32): cpu = self._backward_features(use_gpu=False, dtype=dtype, numdiff=False) gpu = self._backward_features(use_gpu=True, dtype=dtype, numdiff=False) - self.assertAllClose(cpu, gpu) + self.assertAllClose(cpu, gpu, 1e-3) def test_backward_bias_gpu_float32(self, dtype=np.float32): cpu = self._backward_bias(use_gpu=False, dtype=dtype, numdiff=False) gpu = self._backward_bias(use_gpu=True, dtype=dtype, numdiff=False) - self.assertAllClose(cpu, gpu, 1e-5) + self.assertAllClose(cpu, gpu, 1e-4) def test_backward_theta_gpu_float32(self, dtype=np.float32): cpu = self._backward_theta(use_gpu=False, dtype=dtype, numdiff=False) gpu = self._backward_theta(use_gpu=True, dtype=dtype, numdiff=False) - self.assertAllClose(cpu, gpu) - - - + self.assertAllClose(cpu, gpu, 1e-4) if __name__ == '__main__':