对应你的内核的反汇编代码(编译为compute_20,sm_20)如下
/*0000*/        MOV R1, c[0x1][0x100];                     
/*0008*/        S2R R0, SR_CTAID.X;                        
/*0010*/        S2R R2, SR_TID.X;                            
/*0018*/        IMAD R0, R0, c[0x0][0x8], R2;              
/*0020*/        ISETP.GE.AND P0, PT, R0, c[0x0][0x2c], PT; 
/*0028*/    @P0 EXIT ;                                     
/*0030*/        SHL R0, R0, 0x2;                           
/*0038*/        IADD R2, R0, c[0x0][0x20];                 
/*0040*/        IADD R3, R0, c[0x0][0x24];                 
/*0048*/        IADD R0, R0, c[0x0][0x28];                 
/*0050*/        LD R2, [R2];                               R2 = x = g_A[idx]
/*0058*/        LD R3, [R3];                               R3 = y = g_B[idx]
/*0060*/        I2I.S32.S32 R5, |R2|;                      
/*0068*/        I2F.F32.U32.RP R4, R5;                     R4 = (float)x 
/*0070*/        MUFU.RCP R4, R4;                           R4 = 1/R4
/*0078*/        IADD32I R4, R4, 0xffffffe;                 
/*0080*/        F2I.FTZ.U32.F32.TRUNC R4, R4;              
/*0088*/        IMUL.U32.U32 R6, R5, R4;                   R6 = x * (1/y)
/*0090*/        I2I.S32.S32 R7, -R6;                       
/*0098*/        I2I.S32.S32 R6, |R3|;                      
/*00a0*/        IMAD.U32.U32.HI R7, R4, R7, R4;            
/*00a8*/        IMUL.U32.U32.HI R4, R7, R6;               
/*00b0*/        LOP.XOR R7, R3, R2;                        
/*00b8*/        IMAD.U32.U32 R6, -R5, R4, R6;              
/*00c0*/        ISETP.GE.AND P1, PT, R7, RZ, PT;           
/*00c8*/        ISETP.LE.U32.AND P0, PT, R5, R6, PT;       
/*00d0*/    @P0 ISUB R6, R6, R5;                           
/*00d8*/    @P0 IADD R4, R4, 0x1;                         
/*00e0*/        ISETP.GE.U32.AND P0, PT, R6, R5, PT;      
/*00e8*/        LOP.PASS_B R6, RZ, ~R2;                    
/*00f0*/        ISUB R5, R3, R2;                           
/*00f8*/    @P0 IADD R4, R4, 0x1;                          
/*0100*/   @!P1 I2I.S32.S32 R4, -R4;                       
/*0108*/        ICMP.EQ R4, R6, R4, R2;                    
/*0110*/        IADD R4, R5, R4;                          
/*0118*/        IMAD R2, R3, R2, R4;                      
/*0120*/        ST [R0], R2;                             
/*0128*/        EXIT ;                                    
从上面的代码中,有以下浮点运算
I2F.F32.U32.RP R4, R5;                     Integer to Float conversion
MUFU.RCP R4, R4;                           Multifunction Floating Point Operation (Reciprocal)
F2I.FTZ.U32.F32.TRUNC R4, R4;              Float to Integer conversion
这些操作似乎与(x/y)哪个是两个整数之间的除法有关,但需要转换为浮点数。我真的不知道转换是否算作浮点运算。我在代码中看不到任何其他浮点运算。
全局内存操作如下3
LD R2, [R2];                               
LD R3, [R3];                              
ST [R0], R2;                             
CGMA = 3/3 = 1对于您的情况,我会说(将int2floatandfloat2int转换计为浮点运算)。