<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:content="http://purl.org/rss/1.0/modules/content/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" xmlns:taxo="http://purl.org/rss/1.0/modules/taxonomy/" version="2.0">
  <channel>
    <title>topic Missed optimization opportunities using dvec.h in Intel® ISA Extensions</title>
    <link>https://community.intel.com/t5/Intel-ISA-Extensions/Missed-optimization-opportunities-using-dvec-h/m-p/771007#M116</link>
    <description>Compile this for use on AVX system (Intel C++) and compare runtimes of two loops.&lt;BR /&gt;&lt;BR /&gt;The first loop (not using dvec.h) generates nice vector code but incorporates vinsertf128's and vextractf128 in the preponderant computational section of this loop. This shows as 7 memory references.&lt;BR /&gt;&lt;BR /&gt;The second loop (using dvec.h) generates nice vector code as well, but does not use the the vinsert/vextract in the preponderant computational section of this loop. This shows as4 memory references.&lt;BR /&gt;&lt;BR /&gt;*** the first loop runs faster??? By about 2x!!!!&lt;BR /&gt;&lt;BR /&gt;In looking at the disassembly it is interleaving reads (not unrolled) in the first loop but not in the second loop.&lt;BR /&gt;&lt;BR /&gt;This may be a good example for your compiler optimization team to examine for optimization opportunities.&lt;BR /&gt;[cpp]// Felix.cpp : Defines the entry point for the console application.
//

#include "stdafx.h"
#include "dvec.h"
#include "omp.h"
#include &lt;IOSTREAM&gt;

#define USE_AVX
#ifdef USE_AVX
struct aosoa
{
F32vec8 a1;
F32vec8 a2;
F32vec8 a3;
F32vec8 a4; 
};
const int VecWidth = 8;
#else
struct aosoa
{
F32vec4 a1;
F32vec4 a2;
F32vec4 a3;
F32vec4 a4; 
};
const int VecWidth = 4;
#endif
const int N = 64*1024*1024;
const int Nvecs = N / VecWidth;

float* a1;	// &lt;N&gt;;
float* a2;	// &lt;N&gt;;
float* a3;	// &lt;N&gt;;
float* a4;	// &lt;N&gt;;

aosoa*	s;	// [Nvecs];

float coeff1 = 1.2345f;
float coeff2 = .987654321f;

int _tmain(int argc, _TCHAR* argv[])
{
	a1 = new float&lt;N&gt;;
	a2 = new float&lt;N&gt;;
	a3 = new float&lt;N&gt;;
	a4 = new float&lt;N&gt;;
	s = new aosoa[Nvecs];
	std::cout &amp;lt;&amp;lt; "a1 " &amp;lt;&amp;lt; &amp;amp;a1[0] &amp;lt;&amp;lt; std::endl;
	std::cout &amp;lt;&amp;lt; "a2 " &amp;lt;&amp;lt; &amp;amp;a2[0] &amp;lt;&amp;lt; std::endl;
	std::cout &amp;lt;&amp;lt; "a3 " &amp;lt;&amp;lt; &amp;amp;a3[0] &amp;lt;&amp;lt; std::endl;
	std::cout &amp;lt;&amp;lt; "a4 " &amp;lt;&amp;lt; &amp;amp;a4[0] &amp;lt;&amp;lt; std::endl;
	std::cout &amp;lt;&amp;lt; "s " &amp;lt;&amp;lt; &amp;amp;s[0] &amp;lt;&amp;lt; std::endl;
	for(int i=0; i&lt;N&gt; = 1.0f / float(i);
		a2&lt;I&gt; = 2.0f / float(i);
		a3&lt;I&gt; = 3.0f / float(i);
		a4&lt;I&gt; = 4.0f / float(i);
	}
	for(int j=0; j&lt;NVECS&gt;.a1&lt;I&gt; = (float(i) + 1.0f) / float(j*VecWidth + i);
	}
	double totAOS = 0.0;
	double totPAOS = 0.0;
	for(int iRep=0; iRep &amp;lt; 50; ++iRep)
	{
		// test 1
		double t0 = omp_get_wtime();
		for (int i=0; i&lt;N&gt; = a2&lt;I&gt;*a2&lt;I&gt;*coeff1*a3&lt;I&gt; - a2&lt;I&gt; + coeff2*a1&lt;I&gt;;
		}
		double t1 = omp_get_wtime();
		totAOS += t1 - t0;
		// test 2
		for (int i=0; i&lt;NVECS&gt;.a3 = s&lt;I&gt;.a2*s&lt;I&gt;.a2*coeff1*s&lt;I&gt;.a3 - s&lt;I&gt;.a2 + s&lt;I&gt;.a1*coeff2;
		}
		double t2 = omp_get_wtime();
		totPAOS += t2 - t1;
	}

	std::cout &amp;lt;&amp;lt; totAOS &amp;lt;&amp;lt; "  " &amp;lt;&amp;lt; totPAOS &amp;lt;&amp;lt; std::endl;
	return 0;
}

[/cpp]&lt;BR /&gt;Jim Dempsey&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/NVECS&gt;&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/N&gt;&lt;/I&gt;&lt;/NVECS&gt;&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/IOSTREAM&gt;</description>
    <pubDate>Wed, 08 Aug 2012 15:21:01 GMT</pubDate>
    <dc:creator>jimdempseyatthecove</dc:creator>
    <dc:date>2012-08-08T15:21:01Z</dc:date>
    <item>
      <title>Missed optimization opportunities using dvec.h</title>
      <link>https://community.intel.com/t5/Intel-ISA-Extensions/Missed-optimization-opportunities-using-dvec-h/m-p/771007#M116</link>
      <description>Compile this for use on AVX system (Intel C++) and compare runtimes of two loops.&lt;BR /&gt;&lt;BR /&gt;The first loop (not using dvec.h) generates nice vector code but incorporates vinsertf128's and vextractf128 in the preponderant computational section of this loop. This shows as 7 memory references.&lt;BR /&gt;&lt;BR /&gt;The second loop (using dvec.h) generates nice vector code as well, but does not use the the vinsert/vextract in the preponderant computational section of this loop. This shows as4 memory references.&lt;BR /&gt;&lt;BR /&gt;*** the first loop runs faster??? By about 2x!!!!&lt;BR /&gt;&lt;BR /&gt;In looking at the disassembly it is interleaving reads (not unrolled) in the first loop but not in the second loop.&lt;BR /&gt;&lt;BR /&gt;This may be a good example for your compiler optimization team to examine for optimization opportunities.&lt;BR /&gt;[cpp]// Felix.cpp : Defines the entry point for the console application.
//

#include "stdafx.h"
#include "dvec.h"
#include "omp.h"
#include &lt;IOSTREAM&gt;

#define USE_AVX
#ifdef USE_AVX
struct aosoa
{
F32vec8 a1;
F32vec8 a2;
F32vec8 a3;
F32vec8 a4; 
};
const int VecWidth = 8;
#else
struct aosoa
{
F32vec4 a1;
F32vec4 a2;
F32vec4 a3;
F32vec4 a4; 
};
const int VecWidth = 4;
#endif
const int N = 64*1024*1024;
const int Nvecs = N / VecWidth;

float* a1;	// &lt;N&gt;;
float* a2;	// &lt;N&gt;;
float* a3;	// &lt;N&gt;;
float* a4;	// &lt;N&gt;;

aosoa*	s;	// [Nvecs];

float coeff1 = 1.2345f;
float coeff2 = .987654321f;

int _tmain(int argc, _TCHAR* argv[])
{
	a1 = new float&lt;N&gt;;
	a2 = new float&lt;N&gt;;
	a3 = new float&lt;N&gt;;
	a4 = new float&lt;N&gt;;
	s = new aosoa[Nvecs];
	std::cout &amp;lt;&amp;lt; "a1 " &amp;lt;&amp;lt; &amp;amp;a1[0] &amp;lt;&amp;lt; std::endl;
	std::cout &amp;lt;&amp;lt; "a2 " &amp;lt;&amp;lt; &amp;amp;a2[0] &amp;lt;&amp;lt; std::endl;
	std::cout &amp;lt;&amp;lt; "a3 " &amp;lt;&amp;lt; &amp;amp;a3[0] &amp;lt;&amp;lt; std::endl;
	std::cout &amp;lt;&amp;lt; "a4 " &amp;lt;&amp;lt; &amp;amp;a4[0] &amp;lt;&amp;lt; std::endl;
	std::cout &amp;lt;&amp;lt; "s " &amp;lt;&amp;lt; &amp;amp;s[0] &amp;lt;&amp;lt; std::endl;
	for(int i=0; i&lt;N&gt; = 1.0f / float(i);
		a2&lt;I&gt; = 2.0f / float(i);
		a3&lt;I&gt; = 3.0f / float(i);
		a4&lt;I&gt; = 4.0f / float(i);
	}
	for(int j=0; j&lt;NVECS&gt;.a1&lt;I&gt; = (float(i) + 1.0f) / float(j*VecWidth + i);
	}
	double totAOS = 0.0;
	double totPAOS = 0.0;
	for(int iRep=0; iRep &amp;lt; 50; ++iRep)
	{
		// test 1
		double t0 = omp_get_wtime();
		for (int i=0; i&lt;N&gt; = a2&lt;I&gt;*a2&lt;I&gt;*coeff1*a3&lt;I&gt; - a2&lt;I&gt; + coeff2*a1&lt;I&gt;;
		}
		double t1 = omp_get_wtime();
		totAOS += t1 - t0;
		// test 2
		for (int i=0; i&lt;NVECS&gt;.a3 = s&lt;I&gt;.a2*s&lt;I&gt;.a2*coeff1*s&lt;I&gt;.a3 - s&lt;I&gt;.a2 + s&lt;I&gt;.a1*coeff2;
		}
		double t2 = omp_get_wtime();
		totPAOS += t2 - t1;
	}

	std::cout &amp;lt;&amp;lt; totAOS &amp;lt;&amp;lt; "  " &amp;lt;&amp;lt; totPAOS &amp;lt;&amp;lt; std::endl;
	return 0;
}

[/cpp]&lt;BR /&gt;Jim Dempsey&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/NVECS&gt;&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/N&gt;&lt;/I&gt;&lt;/NVECS&gt;&lt;/I&gt;&lt;/I&gt;&lt;/I&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/N&gt;&lt;/IOSTREAM&gt;</description>
      <pubDate>Wed, 08 Aug 2012 15:21:01 GMT</pubDate>
      <guid>https://community.intel.com/t5/Intel-ISA-Extensions/Missed-optimization-opportunities-using-dvec-h/m-p/771007#M116</guid>
      <dc:creator>jimdempseyatthecove</dc:creator>
      <dc:date>2012-08-08T15:21:01Z</dc:date>
    </item>
    <item>
      <title>Missed optimization opportunities using dvec.h</title>
      <link>https://community.intel.com/t5/Intel-ISA-Extensions/Missed-optimization-opportunities-using-dvec-h/m-p/771008#M117</link>
      <description>Hello Jim,&lt;BR /&gt;&lt;BR /&gt;I've seen you already posted a link in the compiler forum to here:&lt;BR /&gt;&lt;DIV style="text-align: left;"&gt;&lt;A href="http://software.intel.com/en-us/forums/showthread.php?t=107141"&gt;&lt;/A&gt;&lt;A href="http://software.intel.com/en-us/forums/showthread.php?t=107141" target="_blank"&gt;http://software.intel.com/en-us/forums/showthread.php?t=107141&lt;/A&gt;&lt;BR /&gt;&lt;/DIV&gt;As it's about the C++ class libraries it fits better to the compiler forum. Let's continue the discussion there.&lt;BR /&gt;&lt;BR /&gt;I'll close the thread at hand.&lt;BR /&gt;&lt;BR /&gt;Best regards,&lt;BR /&gt;&lt;BR /&gt;Georg Zitzlsberger</description>
      <pubDate>Thu, 09 Aug 2012 14:36:30 GMT</pubDate>
      <guid>https://community.intel.com/t5/Intel-ISA-Extensions/Missed-optimization-opportunities-using-dvec-h/m-p/771008#M117</guid>
      <dc:creator>Georg_Z_Intel</dc:creator>
      <dc:date>2012-08-09T14:36:30Z</dc:date>
    </item>
  </channel>
</rss>

