[dnn/dot] reverted back to peak tensorcores performance

2019-07-16 16:14:58 -07:00
parent 164d85077f
commit 5f6dd23fc2
5 changed files with 17 additions and 31 deletions
--- a/lib/dnn/base.cpp
+++ b/lib/dnn/base.cpp
@@ -51,7 +51,7 @@ void base::enqueue(driver::stream *stream, std::vector<driver::buffer *> args, b
      jit->add_module(name_.c_str(), src.c_str(), best.params);
    }
    else {
-      jit->add_module(name_.c_str(), src.c_str(), jit->get_valid(name_.c_str(), src.c_str()));
+      jit->add_module(name_.c_str(), src.c_str(), {16, 4, 128, 16, 4, 128, 2, 2, 2, 2, 8, 16, 8, 1});
    }
    triton::driver::kernel* kernel = jit->get_function(name_.c_str());
    clone->init_impl(stream, (triton::driver::cu_module*)kernel->module());
--- a/lib/dnn/gemm.cpp
+++ b/lib/dnn/gemm.cpp
@@ -109,8 +109,8 @@ const tunable int32 TN = {16, 32, 64, 128};
 const tunable int32 TK = {16};
 const tunable int32 GZ = {1};

-void matmul(restrict read_only )" + a_ty_ + R"( *A,
-            restrict read_only )" + b_ty_ + R"( *B,
+void matmul(restrict read_only align(16) )" + a_ty_ + R"( *A,
+            restrict read_only align(16) )" + b_ty_ + R"( *B,
            fp32 *C,
            int32 M, int32 N, int32 K,
            )" + align_lda_str + R"( int32 lda, )" + align_ldb_str + R"(" int32 ldb, int32 ldc,
@@ -158,20 +158,7 @@ void matmul(restrict read_only )" + a_ty_ + R"( *A,
  int1 checkc1[TN] = ryc < N;
  int1 checkc[TM, TN] = checkc0[:, newaxis] && checkc1[newaxis, :];
  fp32* pc[TM, TN] = C + ryc[newaxis, :]*ldc + rxc[:, newaxis];
-  int32 *plock = locks + ridx + ridy*grid0;
-  while(__atomic_cas(plock, 0, 1));
-  int32 *pcount = plock + grid0*grid1;
-  int32 count = *pcount;
-  int32 countp1 = select(count == GZ - 1, 0, count + 1);
-  if(count == 0) {
-    @checkc *pc = c;
-    *pcount = countp1;
-  }
-  else {
-    @checkc *pc = c + *pc;
-    *pcount = countp1;
-  }
-  __atomic_cas(plock, 1, 0);
+  @checkc *pc = c;
 }
 )";
  os << res;