<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:content="http://purl.org/rss/1.0/modules/content/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" xmlns:taxo="http://purl.org/rss/1.0/modules/taxonomy/" version="2.0">
  <channel>
    <title>topic clEnqueueNDRangeKernel may fail when using 2D local arrays in OpenCL* for CPU</title>
    <link>https://community.intel.com/t5/OpenCL-for-CPU/clEnqueueNDRangeKernel-may-fail-when-using-2D-local-arrays/m-p/1015488#M3127</link>
    <description>&lt;P&gt;clEnqueueNDRangeKernel may fail on kernel with 2D local arrays but succeed with 1D local arrays and manual index computing.&lt;/P&gt;

&lt;P&gt;For example, the following matrix multiplication kernel fails with&amp;nbsp;CL_OUT_OF_RESOURCES if USE_2D is defined and succeedes otherwise.&lt;/P&gt;

&lt;P&gt;Matricies are [24, 72] * [24, 72]&lt;SUP&gt;T&lt;/SUP&gt; = [24, 24] and BLOCK_SIZE = 24.&lt;/P&gt;

&lt;PRE class="brush:cpp;"&gt;#define BLOCK_SIZE 24
#define C_WIDTH 24
#define AB_COMMON 72

__kernel __attribute__((reqd_work_group_size(BLOCK_SIZE, BLOCK_SIZE, 1)))
void mx_mul(__global const float *A,
            __global const float *B,
            __global float *C) {

#ifdef USE_2D
  __local float AS[BLOCK_SIZE][BLOCK_SIZE];
  __local float BS[BLOCK_SIZE][BLOCK_SIZE];
#else
  __local float AS[BLOCK_SIZE * BLOCK_SIZE];
  __local float BS[BLOCK_SIZE * BLOCK_SIZE];
#endif

  int bx = get_group_id(0);
  int by = get_group_id(1);

  int tx = get_local_id(0);
  int ty = get_local_id(1);

  int a_offs = (by * BLOCK_SIZE + ty) * AB_COMMON + tx;
  int b_offs = (bx * BLOCK_SIZE + ty) * AB_COMMON + tx;

  float sum = 0;
  for (int i = 0; i &amp;lt; AB_COMMON / BLOCK_SIZE; i++, a_offs += BLOCK_SIZE, b_offs += BLOCK_SIZE) {
#ifdef USE_2D
    AS[ty][tx] = A[a_offs];
    BS[ty][tx] = B[b_offs];
#else
    AS[ty * BLOCK_SIZE + tx] = A[a_offs];
    BS[ty * BLOCK_SIZE + tx] = B[b_offs];
#endif

    barrier(CLK_LOCAL_MEM_FENCE);

    #pragma unroll
    for (int k = 0; k &amp;lt; BLOCK_SIZE; k++) {
#ifdef USE_2D
      sum += AS[ty]&lt;K&gt; * BS[tx]&lt;K&gt;;
#else
      sum += AS[ty * BLOCK_SIZE + k] * BS[tx * BLOCK_SIZE + k];
#endif
    }

    barrier(CLK_LOCAL_MEM_FENCE);
  }

  C[get_global_id(1) * C_WIDTH + get_global_id(0)] = sum;
}&lt;/K&gt;&lt;/K&gt;&lt;/PRE&gt;

&lt;P&gt;&amp;nbsp;&lt;/P&gt;

&lt;P&gt;Tested on Ubuntu 14.10 and Core i7-3770 with intel_sdk_for_ocl_applications_xe_2013_r3_sdk_3.2.1.16712_x64.&lt;/P&gt;</description>
    <pubDate>Thu, 15 May 2014 11:37:24 GMT</pubDate>
    <dc:creator>Alexey_K_3</dc:creator>
    <dc:date>2014-05-15T11:37:24Z</dc:date>
    <item>
      <title>clEnqueueNDRangeKernel may fail when using 2D local arrays</title>
      <link>https://community.intel.com/t5/OpenCL-for-CPU/clEnqueueNDRangeKernel-may-fail-when-using-2D-local-arrays/m-p/1015488#M3127</link>
      <description>&lt;P&gt;clEnqueueNDRangeKernel may fail on kernel with 2D local arrays but succeed with 1D local arrays and manual index computing.&lt;/P&gt;

&lt;P&gt;For example, the following matrix multiplication kernel fails with&amp;nbsp;CL_OUT_OF_RESOURCES if USE_2D is defined and succeedes otherwise.&lt;/P&gt;

&lt;P&gt;Matricies are [24, 72] * [24, 72]&lt;SUP&gt;T&lt;/SUP&gt; = [24, 24] and BLOCK_SIZE = 24.&lt;/P&gt;

&lt;PRE class="brush:cpp;"&gt;#define BLOCK_SIZE 24
#define C_WIDTH 24
#define AB_COMMON 72

__kernel __attribute__((reqd_work_group_size(BLOCK_SIZE, BLOCK_SIZE, 1)))
void mx_mul(__global const float *A,
            __global const float *B,
            __global float *C) {

#ifdef USE_2D
  __local float AS[BLOCK_SIZE][BLOCK_SIZE];
  __local float BS[BLOCK_SIZE][BLOCK_SIZE];
#else
  __local float AS[BLOCK_SIZE * BLOCK_SIZE];
  __local float BS[BLOCK_SIZE * BLOCK_SIZE];
#endif

  int bx = get_group_id(0);
  int by = get_group_id(1);

  int tx = get_local_id(0);
  int ty = get_local_id(1);

  int a_offs = (by * BLOCK_SIZE + ty) * AB_COMMON + tx;
  int b_offs = (bx * BLOCK_SIZE + ty) * AB_COMMON + tx;

  float sum = 0;
  for (int i = 0; i &amp;lt; AB_COMMON / BLOCK_SIZE; i++, a_offs += BLOCK_SIZE, b_offs += BLOCK_SIZE) {
#ifdef USE_2D
    AS[ty][tx] = A[a_offs];
    BS[ty][tx] = B[b_offs];
#else
    AS[ty * BLOCK_SIZE + tx] = A[a_offs];
    BS[ty * BLOCK_SIZE + tx] = B[b_offs];
#endif

    barrier(CLK_LOCAL_MEM_FENCE);

    #pragma unroll
    for (int k = 0; k &amp;lt; BLOCK_SIZE; k++) {
#ifdef USE_2D
      sum += AS[ty]&lt;K&gt; * BS[tx]&lt;K&gt;;
#else
      sum += AS[ty * BLOCK_SIZE + k] * BS[tx * BLOCK_SIZE + k];
#endif
    }

    barrier(CLK_LOCAL_MEM_FENCE);
  }

  C[get_global_id(1) * C_WIDTH + get_global_id(0)] = sum;
}&lt;/K&gt;&lt;/K&gt;&lt;/PRE&gt;

&lt;P&gt;&amp;nbsp;&lt;/P&gt;

&lt;P&gt;Tested on Ubuntu 14.10 and Core i7-3770 with intel_sdk_for_ocl_applications_xe_2013_r3_sdk_3.2.1.16712_x64.&lt;/P&gt;</description>
      <pubDate>Thu, 15 May 2014 11:37:24 GMT</pubDate>
      <guid>https://community.intel.com/t5/OpenCL-for-CPU/clEnqueueNDRangeKernel-may-fail-when-using-2D-local-arrays/m-p/1015488#M3127</guid>
      <dc:creator>Alexey_K_3</dc:creator>
      <dc:date>2014-05-15T11:37:24Z</dc:date>
    </item>
    <item>
      <title>This issue seems to be</title>
      <link>https://community.intel.com/t5/OpenCL-for-CPU/clEnqueueNDRangeKernel-may-fail-when-using-2D-local-arrays/m-p/1015489#M3128</link>
      <description>&lt;P&gt;This issue seems to be resolved in intel_sdk_for_ocl_applications_2014_beta_sdk_4.0.1.17537_x64.&lt;/P&gt;</description>
      <pubDate>Thu, 15 May 2014 12:17:37 GMT</pubDate>
      <guid>https://community.intel.com/t5/OpenCL-for-CPU/clEnqueueNDRangeKernel-may-fail-when-using-2D-local-arrays/m-p/1015489#M3128</guid>
      <dc:creator>Alexey_K_3</dc:creator>
      <dc:date>2014-05-15T12:17:37Z</dc:date>
    </item>
  </channel>
</rss>

