Annotation of tme/libtme/host/x86/rc-x86-subs-asm.S, revision 1.1.1.1

1.1       root        1: /* $Id: rc-x86-subs-asm.S,v 1.2 2009/09/07 15:25:23 fredette Exp $ */
                      2: 
                      3: /* libtme/host/x86/rc-x86-subs-asm.S - hand-coded x86 host recode subs: */
                      4: 
                      5: /*
                      6:  * Copyright (c) 2007 Matt Fredette
                      7:  * All rights reserved.
                      8:  *
                      9:  * Redistribution and use in source and binary forms, with or without
                     10:  * modification, are permitted provided that the following conditions
                     11:  * are met:
                     12:  * 1. Redistributions of source code must retain the above copyright
                     13:  *    notice, this list of conditions and the following disclaimer.
                     14:  * 2. Redistributions in binary form must reproduce the above copyright
                     15:  *    notice, this list of conditions and the following disclaimer in the
                     16:  *    documentation and/or other materials provided with the distribution.
                     17:  * 3. All advertising materials mentioning features or use of this software
                     18:  *    must display the following acknowledgement:
                     19:  *      This product includes software developed by Matt Fredette.
                     20:  * 4. The name of the author may not be used to endorse or promote products
                     21:  *    derived from this software without specific prior written permission.
                     22:  *
                     23:  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR ``AS IS'' AND ANY EXPRESS OR
                     24:  * IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
                     25:  * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
                     26:  * DISCLAIMED.  IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY DIRECT,
                     27:  * INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
                     28:  * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
                     29:  * SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
                     30:  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
                     31:  * STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN
                     32:  * ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
                     33:  * POSSIBILITY OF SUCH DAMAGE.
                     34:  */
                     35: 
                     36:        .text
                     37: 
                     38:        .ascii "$Id: rc-x86-subs-asm.S,v 1.2 2009/09/07 15:25:23 fredette Exp $"
                     39: 
                     40: # concatenation:
                     41: #
                     42: #if ((defined(__STDC__) || defined(__cplusplus) || defined(c_plusplus)) && !defined(UNIXCPP)) || defined(ANSICPP)
                     43: #define __TME_CONCAT(a,b) a ## b
                     44: #define _TME_CONCAT(a,b) __TME_CONCAT(a,b)
                     45: #else
                     46: #define _TME_CONCAT(a,b) a/**/b
                     47: #endif
                     48: 
                     49: #ifdef __x86_64__
                     50: #define TME_RECODE_SIZE_HOST                   6
                     51: #define TME_RECODE_BITS_HOST(x)                        _TME_CONCAT(x, 64)
                     52: #define TME_RECODE_BITS_DOUBLE_HOST(x)         _TME_CONCAT(x, 128)
                     53: #define TME_RECODE_X86_OPN(x)                  _TME_CONCAT(x, q)
                     54: #define TME_RECODE_X86_REGN(x)                 _TME_CONCAT(%r, x)
                     55: #define TME_RECODE_X86_REG_HOST_SUBS_DST_N     %r12
                     56: #define TME_RECODE_X86_REG_HOST_SUBS_DST_L     %r12d
                     57: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_N    %rbp
                     58: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_L    %ebp
                     59: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_W    %bp
                     60: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_B    %bpl
                     61: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_P1_N %rax
                     62: #else  /* !__x86_64__ */
                     63: #define TME_RECODE_SIZE_HOST                   5
                     64: #define TME_RECODE_BITS_HOST(x)                        _TME_CONCAT(x, 32)
                     65: #define TME_RECODE_BITS_DOUBLE_HOST(x)         _TME_CONCAT(x, 64)
                     66: #define TME_RECODE_X86_OPN(x)                  _TME_CONCAT(x, l)
                     67: #define TME_RECODE_X86_REGN(x)                 _TME_CONCAT(%e, x)
                     68: #define TME_RECODE_X86_REG_HOST_SUBS_DST_N     %edi
                     69: #define TME_RECODE_X86_REG_HOST_SUBS_DST_L     %edi
                     70: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_N    %ebp
                     71: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_L    %ebp
                     72: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_W    %bp
                     73: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_P1_N %eax
                     74: #endif /* !__x86_64__ */
                     75: 
                     76:        # this macro does a double-host-size shift:
                     77:        #
                     78: .macro tme_recode_x86_double_shift name shift shiftd reg_first reg_second arith=0
                     79:        .align  16
                     80:        .globl  \name
                     81: \name:
                     82: 
                     83:        # branch if the most-significant half of the shift count in
                     84:        # src1 is nonzero, swapping the least-significant half of the
                     85:        # shift count in src1 with the scratch c register in the
                     86:        # meantime.  we will swap them back after the shift.
                     87:        #
                     88:        # NB: we do the swaps with the c register instead of just
                     89:        # overwriting it, to cooperate with subs that keep the
                     90:        # most-significant half of src0 in the c register:      
                     91:        #
                     92:        TME_RECODE_X86_OPN(test)        TME_RECODE_X86_REG_HOST_SUBS_SRC1_P1_N, TME_RECODE_X86_REG_HOST_SUBS_SRC1_P1_N
                     93:        TME_RECODE_X86_OPN(xchg)        TME_RECODE_X86_REG_HOST_SUBS_SRC1_N, TME_RECODE_X86_REGN(cx)
                     94:        jnz     .L_double_shift_2_\@
                     95: 
                     96:        # branch if the shift count is greater than the host size:
                     97:        #
                     98:        TME_RECODE_X86_OPN(cmp)         $(1 << TME_RECODE_SIZE_HOST), TME_RECODE_X86_REGN(cx)
                     99:        jae     .L_double_shift_1_\@
                    100: 
                    101:        # do the double-precision shift:
                    102:        # 
                    103:        \shiftd                         %cl, \reg_second, \reg_first
                    104:        \shift                          %cl, \reg_second
                    105:        TME_RECODE_X86_OPN(xchg)        TME_RECODE_X86_REG_HOST_SUBS_SRC1_N, TME_RECODE_X86_REGN(cx)
                    106:        ret
                    107: 
                    108: .L_double_shift_1_\@:
                    109: 
                    110:        # branch if the shift count is greater than or equal to the
                    111:        # double-host size:
                    112:        #
                    113:        TME_RECODE_X86_OPN(cmp)         $2*(1 << TME_RECODE_SIZE_HOST), TME_RECODE_X86_REGN(cx)
                    114:        jae     .L_double_shift_2_\@
                    115:        
                    116:        # first do a host-size shift by doing a register move and
                    117:        # then a clear for a logical shift, or a copy of the most
                    118:        # significant bit of the second register down into all other
                    119:        # bits:
                    120:        #
                    121:        TME_RECODE_X86_OPN(mov)         \reg_second, \reg_first
                    122: .if \arith
                    123:        TME_RECODE_X86_OPN(shl)         $1, \reg_second
                    124:        TME_RECODE_X86_OPN(sbb)         \reg_second, \reg_second
                    125: .else
                    126:        TME_RECODE_X86_OPN(xor)         \reg_second, \reg_second
                    127: .endif
                    128: 
                    129:        # do the remainder of the shift as a host-size shift (which
                    130:        # masks the count in %cl to the host size):
                    131:        #
                    132:        \shift                          %cl, \reg_first
                    133:        TME_RECODE_X86_OPN(xchg)        TME_RECODE_X86_REG_HOST_SUBS_SRC1_N, TME_RECODE_X86_REGN(cx)
                    134:        ret
                    135: 
                    136:        # the shift count is greater than the double-host-size.  for a
                    137:        # logical shift, clear both registers.  for an arithmetic shift,
                    138:        # copy the most significant bit of the second register down into
                    139:        # all bits in both registers:
                    140:        #
                    141: .L_double_shift_2_\@:
                    142: .if \arith
                    143:        TME_RECODE_X86_OPN(shl)         $1, \reg_second
                    144:        TME_RECODE_X86_OPN(sbb)         \reg_second, \reg_second
                    145:        TME_RECODE_X86_OPN(mov)         \reg_second, \reg_first
                    146: .else
                    147:        TME_RECODE_X86_OPN(xor)         \reg_first, \reg_first
                    148:        TME_RECODE_X86_OPN(xor)         \reg_second, \reg_second
                    149: .endif
                    150:        TME_RECODE_X86_OPN(xchg)        TME_RECODE_X86_REG_HOST_SUBS_SRC1_N, TME_RECODE_X86_REGN(cx)
                    151:        ret
                    152: .endm
                    153: 
                    154:        # this macro does a host-size or smaller shift:
                    155:        #
                    156: .macro tme_recode_x86_shift insn size shift
                    157:        .align  16
                    158:        .globl  tme_recode_x86_\insn\size
                    159: tme_recode_x86_\insn\size:
                    160: 
                    161:        # _if this is a right shift smaller than host size, first
                    162:        # zero-extend or sign-extend the destination to host size,
                    163:        # so we can do a 32-bit shift, or branch to
                    164:        # _tme_recode_x86_shift_arithmetic_all (which assumes that
                    165:        # the destination is host size):
                    166:        #
                    167: .ifnc \insn,shll
                    168: .if \size < TME_RECODE_BITS_HOST(/**/)
                    169: 
                    170:        # _if this is an 8-bit shift on an ia32 host, we can't encode
                    171:        # %dil for a movzbl or movsbl, so we do an and for a movzbl and
                    172:        # a movsbl through the c register:
                    173:        #
                    174: .ifeq (TME_RECODE_BITS_HOST(/**/) - 32) | (\size - 8)
                    175: .ifc \insn,shra
                    176:        movl    %edi, %ecx
                    177:        movsbl  %cl, %edi
                    178: .else
                    179:        andl    $0xff, %edi
                    180: .endif
                    181: .else
                    182: 
                    183:        # otherwise, emit a movz or movs instruction to extend
                    184:        # the destination in TME_RECODE_X86_REG_HOST_SUBS_DST:
                    185:        #
                    186: .ifeq \size - 32
                    187:        # the x86_64 32-bit extensions are different:
                    188:        #
                    189: .ifc \insn,shra
                    190:        movslq  TME_RECODE_X86_REG_HOST_SUBS_DST_L, TME_RECODE_X86_REG_HOST_SUBS_DST_N
                    191: .else
                    192:        movl    TME_RECODE_X86_REG_HOST_SUBS_DST_L, TME_RECODE_X86_REG_HOST_SUBS_DST_L
                    193: .endif
                    194: .else
                    195: #ifdef __x86_64__
                    196:        .byte   0x48 + (1 << 0) + (1 << 2) # TME_RECODE_X86_REX_R(TME_RECODE_SIZE_64, %r12) + TME_RECODE_X86_REX_B(0, %r12)
                    197: #endif /* __x86_64__ */
                    198:        .byte   0x0f                    # TME_RECODE_X86_OPCODE_ESC_0F
                    199: .ifc \insn,shra
                    200:        .byte   0xbf - ((\size / 8) & 1)        # TME_RECODE_X86_OPCODE0F_MOVS_Ew_Gv or TME_RECODE_X86_OPCODE0F_MOVS_Eb_Gv
                    201: .else
                    202:        .byte   0xb7 - ((\size / 8) & 1)        # TME_RECODE_X86_OPCODE0F_MOVZ_Ew_Gv or TME_RECODE_X86_OPCODE0F_MOVZ_Eb_Gv
                    203: .endif
                    204: #ifdef __x86_64__
                    205:        .byte   (0xc0 + (12 % 8)) + ((12 % 8) << 3) # %r12, %r12
                    206: #else  /* !__x86_64__ */
                    207:        .byte   (0xc0 + 7) + (7 << 3) # %edi, %edi
                    208: #endif /* __x86_64__ */
                    209: .endif
                    210: .endif
                    211: .endif
                    212: .endif
                    213: 
                    214:        # compare the shift count in TME_RECODE_X86_REG_HOST_SUBS_SRC1
                    215:        # to the size:
                    216:        #
                    217: .ifeq \size - 64
                    218:        cmpq    $\size, TME_RECODE_X86_REG_HOST_SUBS_SRC1_N
                    219: .else
                    220: .ifeq \size - 32
                    221:        cmpl    $\size, TME_RECODE_X86_REG_HOST_SUBS_SRC1_L
                    222: .else
                    223: .ifeq \size - 16
                    224:        cmpw    $\size, TME_RECODE_X86_REG_HOST_SUBS_SRC1_W
                    225: .else
                    226: .ifeq \size - 8
                    227: #ifdef TME_RECODE_X86_REG_HOST_SUBS_SRC1_B
                    228:        cmpb    $\size, TME_RECODE_X86_REG_HOST_SUBS_SRC1_B
                    229: #else
                    230:        movl    TME_RECODE_X86_REG_HOST_SUBS_SRC1_L, %ecx
                    231:        cmpb    $\size, %cl
                    232: #endif
                    233: .endif
                    234: .endif
                    235: .endif
                    236: .endif
                    237: 
                    238:        # put the shift count into the c register.  this has already
                    239:        # been done if this is an 8-bit shift on an ia32 host:
                    240:        #
                    241: .ifne (TME_RECODE_BITS_HOST(/**/) - 32) | (\size - 8)
                    242:        movl    TME_RECODE_X86_REG_HOST_SUBS_SRC1_L, %ecx
                    243: .endif
                    244: 
                    245:        # _if the shift count is greater than or equal to the size,
                    246:        # for an arithmetic shift copy the most-significant bit down
                    247:        # into all other bits, otherwise do a clear:
                    248:        #
                    249: .ifc \insn,shra
                    250:        jae     _tme_recode_x86_shift_arithmetic_all
                    251: .else
                    252:        jae     _tme_recode_x86_shift_logical_all
                    253: .endif
                    254: 
                    255:        # otherwise, do the shift:
                    256:        #
                    257: .ifeq (\size - 64)
                    258:        \shift  %cl, TME_RECODE_X86_REG_HOST_SUBS_DST_N
                    259: .else
                    260:        \shift  %cl, TME_RECODE_X86_REG_HOST_SUBS_DST_L
                    261: .endif
                    262:        ret
                    263: .endm
                    264: 
                    265:        # the shifts:
                    266:        #
                    267: #ifdef __x86_64__
                    268: tme_recode_x86_double_shift tme_recode_x86_shll128 shlq shldq %r13 %r12
                    269: tme_recode_x86_double_shift tme_recode_x86_shrl128 shrq shrdq %r12 %r13
                    270: tme_recode_x86_double_shift tme_recode_x86_shra128 sarq shrdq %r12 %r13 1
                    271: tme_recode_x86_shift shll 64 shlq
                    272: tme_recode_x86_shift shrl 64 shrq
                    273: tme_recode_x86_shift shra 64 sarq
                    274: #else  /* !__x86_64__ */
                    275: tme_recode_x86_double_shift tme_recode_x86_shll64 shll shldl %esi %edi
                    276: tme_recode_x86_double_shift tme_recode_x86_shrl64 shrl shrdl %edi %esi
                    277: tme_recode_x86_double_shift tme_recode_x86_shra64 sarl shrdl %edi %esi 1
                    278: #endif /* !__x86_64__ */
                    279: tme_recode_x86_shift shll 32 shll
                    280: tme_recode_x86_shift shrl 32 shrl
                    281: tme_recode_x86_shift shra 32 sarl
                    282: tme_recode_x86_shift shll 16 shll
                    283: tme_recode_x86_shift shrl 16 shrl
                    284: tme_recode_x86_shift shra 16 sarl
                    285: tme_recode_x86_shift shll 8 shll
                    286: tme_recode_x86_shift shrl 8 shrl
                    287: tme_recode_x86_shift shra 8 sarl
                    288: 
                    289: _tme_recode_x86_shift_arithmetic_all:
                    290:        TME_RECODE_X86_OPN(add)         TME_RECODE_X86_REG_HOST_SUBS_DST_N, TME_RECODE_X86_REG_HOST_SUBS_DST_N
                    291:        TME_RECODE_X86_OPN(sbb)         TME_RECODE_X86_REG_HOST_SUBS_DST_N, TME_RECODE_X86_REG_HOST_SUBS_DST_N
                    292:        ret
                    293: 
                    294: _tme_recode_x86_shift_logical_all:
                    295:        TME_RECODE_X86_OPN(xor)         TME_RECODE_X86_REG_HOST_SUBS_DST_N, TME_RECODE_X86_REG_HOST_SUBS_DST_N
                    296:        ret

unix.superglobalmegacorp.com

This archive runs on limited infrastructure. Preserving old code on modern bandwidth. Automated agents are requested to crawl responsibly.