Annotation of tme/libtme/host/x86/rc-x86-subs-asm.S, revision 1.1

1.1     ! root        1: /* $Id: rc-x86-subs-asm.S,v 1.2 2009/09/07 15:25:23 fredette Exp $ */
        !             2: 
        !             3: /* libtme/host/x86/rc-x86-subs-asm.S - hand-coded x86 host recode subs: */
        !             4: 
        !             5: /*
        !             6:  * Copyright (c) 2007 Matt Fredette
        !             7:  * All rights reserved.
        !             8:  *
        !             9:  * Redistribution and use in source and binary forms, with or without
        !            10:  * modification, are permitted provided that the following conditions
        !            11:  * are met:
        !            12:  * 1. Redistributions of source code must retain the above copyright
        !            13:  *    notice, this list of conditions and the following disclaimer.
        !            14:  * 2. Redistributions in binary form must reproduce the above copyright
        !            15:  *    notice, this list of conditions and the following disclaimer in the
        !            16:  *    documentation and/or other materials provided with the distribution.
        !            17:  * 3. All advertising materials mentioning features or use of this software
        !            18:  *    must display the following acknowledgement:
        !            19:  *      This product includes software developed by Matt Fredette.
        !            20:  * 4. The name of the author may not be used to endorse or promote products
        !            21:  *    derived from this software without specific prior written permission.
        !            22:  *
        !            23:  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR ``AS IS'' AND ANY EXPRESS OR
        !            24:  * IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
        !            25:  * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
        !            26:  * DISCLAIMED.  IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY DIRECT,
        !            27:  * INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
        !            28:  * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
        !            29:  * SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
        !            30:  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
        !            31:  * STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN
        !            32:  * ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
        !            33:  * POSSIBILITY OF SUCH DAMAGE.
        !            34:  */
        !            35: 
        !            36:        .text
        !            37: 
        !            38:        .ascii "$Id: rc-x86-subs-asm.S,v 1.2 2009/09/07 15:25:23 fredette Exp $"
        !            39: 
        !            40: # concatenation:
        !            41: #
        !            42: #if ((defined(__STDC__) || defined(__cplusplus) || defined(c_plusplus)) && !defined(UNIXCPP)) || defined(ANSICPP)
        !            43: #define __TME_CONCAT(a,b) a ## b
        !            44: #define _TME_CONCAT(a,b) __TME_CONCAT(a,b)
        !            45: #else
        !            46: #define _TME_CONCAT(a,b) a/**/b
        !            47: #endif
        !            48: 
        !            49: #ifdef __x86_64__
        !            50: #define TME_RECODE_SIZE_HOST                   6
        !            51: #define TME_RECODE_BITS_HOST(x)                        _TME_CONCAT(x, 64)
        !            52: #define TME_RECODE_BITS_DOUBLE_HOST(x)         _TME_CONCAT(x, 128)
        !            53: #define TME_RECODE_X86_OPN(x)                  _TME_CONCAT(x, q)
        !            54: #define TME_RECODE_X86_REGN(x)                 _TME_CONCAT(%r, x)
        !            55: #define TME_RECODE_X86_REG_HOST_SUBS_DST_N     %r12
        !            56: #define TME_RECODE_X86_REG_HOST_SUBS_DST_L     %r12d
        !            57: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_N    %rbp
        !            58: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_L    %ebp
        !            59: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_W    %bp
        !            60: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_B    %bpl
        !            61: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_P1_N %rax
        !            62: #else  /* !__x86_64__ */
        !            63: #define TME_RECODE_SIZE_HOST                   5
        !            64: #define TME_RECODE_BITS_HOST(x)                        _TME_CONCAT(x, 32)
        !            65: #define TME_RECODE_BITS_DOUBLE_HOST(x)         _TME_CONCAT(x, 64)
        !            66: #define TME_RECODE_X86_OPN(x)                  _TME_CONCAT(x, l)
        !            67: #define TME_RECODE_X86_REGN(x)                 _TME_CONCAT(%e, x)
        !            68: #define TME_RECODE_X86_REG_HOST_SUBS_DST_N     %edi
        !            69: #define TME_RECODE_X86_REG_HOST_SUBS_DST_L     %edi
        !            70: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_N    %ebp
        !            71: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_L    %ebp
        !            72: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_W    %bp
        !            73: #define TME_RECODE_X86_REG_HOST_SUBS_SRC1_P1_N %eax
        !            74: #endif /* !__x86_64__ */
        !            75: 
        !            76:        # this macro does a double-host-size shift:
        !            77:        #
        !            78: .macro tme_recode_x86_double_shift name shift shiftd reg_first reg_second arith=0
        !            79:        .align  16
        !            80:        .globl  \name
        !            81: \name:
        !            82: 
        !            83:        # branch if the most-significant half of the shift count in
        !            84:        # src1 is nonzero, swapping the least-significant half of the
        !            85:        # shift count in src1 with the scratch c register in the
        !            86:        # meantime.  we will swap them back after the shift.
        !            87:        #
        !            88:        # NB: we do the swaps with the c register instead of just
        !            89:        # overwriting it, to cooperate with subs that keep the
        !            90:        # most-significant half of src0 in the c register:      
        !            91:        #
        !            92:        TME_RECODE_X86_OPN(test)        TME_RECODE_X86_REG_HOST_SUBS_SRC1_P1_N, TME_RECODE_X86_REG_HOST_SUBS_SRC1_P1_N
        !            93:        TME_RECODE_X86_OPN(xchg)        TME_RECODE_X86_REG_HOST_SUBS_SRC1_N, TME_RECODE_X86_REGN(cx)
        !            94:        jnz     .L_double_shift_2_\@
        !            95: 
        !            96:        # branch if the shift count is greater than the host size:
        !            97:        #
        !            98:        TME_RECODE_X86_OPN(cmp)         $(1 << TME_RECODE_SIZE_HOST), TME_RECODE_X86_REGN(cx)
        !            99:        jae     .L_double_shift_1_\@
        !           100: 
        !           101:        # do the double-precision shift:
        !           102:        # 
        !           103:        \shiftd                         %cl, \reg_second, \reg_first
        !           104:        \shift                          %cl, \reg_second
        !           105:        TME_RECODE_X86_OPN(xchg)        TME_RECODE_X86_REG_HOST_SUBS_SRC1_N, TME_RECODE_X86_REGN(cx)
        !           106:        ret
        !           107: 
        !           108: .L_double_shift_1_\@:
        !           109: 
        !           110:        # branch if the shift count is greater than or equal to the
        !           111:        # double-host size:
        !           112:        #
        !           113:        TME_RECODE_X86_OPN(cmp)         $2*(1 << TME_RECODE_SIZE_HOST), TME_RECODE_X86_REGN(cx)
        !           114:        jae     .L_double_shift_2_\@
        !           115:        
        !           116:        # first do a host-size shift by doing a register move and
        !           117:        # then a clear for a logical shift, or a copy of the most
        !           118:        # significant bit of the second register down into all other
        !           119:        # bits:
        !           120:        #
        !           121:        TME_RECODE_X86_OPN(mov)         \reg_second, \reg_first
        !           122: .if \arith
        !           123:        TME_RECODE_X86_OPN(shl)         $1, \reg_second
        !           124:        TME_RECODE_X86_OPN(sbb)         \reg_second, \reg_second
        !           125: .else
        !           126:        TME_RECODE_X86_OPN(xor)         \reg_second, \reg_second
        !           127: .endif
        !           128: 
        !           129:        # do the remainder of the shift as a host-size shift (which
        !           130:        # masks the count in %cl to the host size):
        !           131:        #
        !           132:        \shift                          %cl, \reg_first
        !           133:        TME_RECODE_X86_OPN(xchg)        TME_RECODE_X86_REG_HOST_SUBS_SRC1_N, TME_RECODE_X86_REGN(cx)
        !           134:        ret
        !           135: 
        !           136:        # the shift count is greater than the double-host-size.  for a
        !           137:        # logical shift, clear both registers.  for an arithmetic shift,
        !           138:        # copy the most significant bit of the second register down into
        !           139:        # all bits in both registers:
        !           140:        #
        !           141: .L_double_shift_2_\@:
        !           142: .if \arith
        !           143:        TME_RECODE_X86_OPN(shl)         $1, \reg_second
        !           144:        TME_RECODE_X86_OPN(sbb)         \reg_second, \reg_second
        !           145:        TME_RECODE_X86_OPN(mov)         \reg_second, \reg_first
        !           146: .else
        !           147:        TME_RECODE_X86_OPN(xor)         \reg_first, \reg_first
        !           148:        TME_RECODE_X86_OPN(xor)         \reg_second, \reg_second
        !           149: .endif
        !           150:        TME_RECODE_X86_OPN(xchg)        TME_RECODE_X86_REG_HOST_SUBS_SRC1_N, TME_RECODE_X86_REGN(cx)
        !           151:        ret
        !           152: .endm
        !           153: 
        !           154:        # this macro does a host-size or smaller shift:
        !           155:        #
        !           156: .macro tme_recode_x86_shift insn size shift
        !           157:        .align  16
        !           158:        .globl  tme_recode_x86_\insn\size
        !           159: tme_recode_x86_\insn\size:
        !           160: 
        !           161:        # _if this is a right shift smaller than host size, first
        !           162:        # zero-extend or sign-extend the destination to host size,
        !           163:        # so we can do a 32-bit shift, or branch to
        !           164:        # _tme_recode_x86_shift_arithmetic_all (which assumes that
        !           165:        # the destination is host size):
        !           166:        #
        !           167: .ifnc \insn,shll
        !           168: .if \size < TME_RECODE_BITS_HOST(/**/)
        !           169: 
        !           170:        # _if this is an 8-bit shift on an ia32 host, we can't encode
        !           171:        # %dil for a movzbl or movsbl, so we do an and for a movzbl and
        !           172:        # a movsbl through the c register:
        !           173:        #
        !           174: .ifeq (TME_RECODE_BITS_HOST(/**/) - 32) | (\size - 8)
        !           175: .ifc \insn,shra
        !           176:        movl    %edi, %ecx
        !           177:        movsbl  %cl, %edi
        !           178: .else
        !           179:        andl    $0xff, %edi
        !           180: .endif
        !           181: .else
        !           182: 
        !           183:        # otherwise, emit a movz or movs instruction to extend
        !           184:        # the destination in TME_RECODE_X86_REG_HOST_SUBS_DST:
        !           185:        #
        !           186: .ifeq \size - 32
        !           187:        # the x86_64 32-bit extensions are different:
        !           188:        #
        !           189: .ifc \insn,shra
        !           190:        movslq  TME_RECODE_X86_REG_HOST_SUBS_DST_L, TME_RECODE_X86_REG_HOST_SUBS_DST_N
        !           191: .else
        !           192:        movl    TME_RECODE_X86_REG_HOST_SUBS_DST_L, TME_RECODE_X86_REG_HOST_SUBS_DST_L
        !           193: .endif
        !           194: .else
        !           195: #ifdef __x86_64__
        !           196:        .byte   0x48 + (1 << 0) + (1 << 2) # TME_RECODE_X86_REX_R(TME_RECODE_SIZE_64, %r12) + TME_RECODE_X86_REX_B(0, %r12)
        !           197: #endif /* __x86_64__ */
        !           198:        .byte   0x0f                    # TME_RECODE_X86_OPCODE_ESC_0F
        !           199: .ifc \insn,shra
        !           200:        .byte   0xbf - ((\size / 8) & 1)        # TME_RECODE_X86_OPCODE0F_MOVS_Ew_Gv or TME_RECODE_X86_OPCODE0F_MOVS_Eb_Gv
        !           201: .else
        !           202:        .byte   0xb7 - ((\size / 8) & 1)        # TME_RECODE_X86_OPCODE0F_MOVZ_Ew_Gv or TME_RECODE_X86_OPCODE0F_MOVZ_Eb_Gv
        !           203: .endif
        !           204: #ifdef __x86_64__
        !           205:        .byte   (0xc0 + (12 % 8)) + ((12 % 8) << 3) # %r12, %r12
        !           206: #else  /* !__x86_64__ */
        !           207:        .byte   (0xc0 + 7) + (7 << 3) # %edi, %edi
        !           208: #endif /* __x86_64__ */
        !           209: .endif
        !           210: .endif
        !           211: .endif
        !           212: .endif
        !           213: 
        !           214:        # compare the shift count in TME_RECODE_X86_REG_HOST_SUBS_SRC1
        !           215:        # to the size:
        !           216:        #
        !           217: .ifeq \size - 64
        !           218:        cmpq    $\size, TME_RECODE_X86_REG_HOST_SUBS_SRC1_N
        !           219: .else
        !           220: .ifeq \size - 32
        !           221:        cmpl    $\size, TME_RECODE_X86_REG_HOST_SUBS_SRC1_L
        !           222: .else
        !           223: .ifeq \size - 16
        !           224:        cmpw    $\size, TME_RECODE_X86_REG_HOST_SUBS_SRC1_W
        !           225: .else
        !           226: .ifeq \size - 8
        !           227: #ifdef TME_RECODE_X86_REG_HOST_SUBS_SRC1_B
        !           228:        cmpb    $\size, TME_RECODE_X86_REG_HOST_SUBS_SRC1_B
        !           229: #else
        !           230:        movl    TME_RECODE_X86_REG_HOST_SUBS_SRC1_L, %ecx
        !           231:        cmpb    $\size, %cl
        !           232: #endif
        !           233: .endif
        !           234: .endif
        !           235: .endif
        !           236: .endif
        !           237: 
        !           238:        # put the shift count into the c register.  this has already
        !           239:        # been done if this is an 8-bit shift on an ia32 host:
        !           240:        #
        !           241: .ifne (TME_RECODE_BITS_HOST(/**/) - 32) | (\size - 8)
        !           242:        movl    TME_RECODE_X86_REG_HOST_SUBS_SRC1_L, %ecx
        !           243: .endif
        !           244: 
        !           245:        # _if the shift count is greater than or equal to the size,
        !           246:        # for an arithmetic shift copy the most-significant bit down
        !           247:        # into all other bits, otherwise do a clear:
        !           248:        #
        !           249: .ifc \insn,shra
        !           250:        jae     _tme_recode_x86_shift_arithmetic_all
        !           251: .else
        !           252:        jae     _tme_recode_x86_shift_logical_all
        !           253: .endif
        !           254: 
        !           255:        # otherwise, do the shift:
        !           256:        #
        !           257: .ifeq (\size - 64)
        !           258:        \shift  %cl, TME_RECODE_X86_REG_HOST_SUBS_DST_N
        !           259: .else
        !           260:        \shift  %cl, TME_RECODE_X86_REG_HOST_SUBS_DST_L
        !           261: .endif
        !           262:        ret
        !           263: .endm
        !           264: 
        !           265:        # the shifts:
        !           266:        #
        !           267: #ifdef __x86_64__
        !           268: tme_recode_x86_double_shift tme_recode_x86_shll128 shlq shldq %r13 %r12
        !           269: tme_recode_x86_double_shift tme_recode_x86_shrl128 shrq shrdq %r12 %r13
        !           270: tme_recode_x86_double_shift tme_recode_x86_shra128 sarq shrdq %r12 %r13 1
        !           271: tme_recode_x86_shift shll 64 shlq
        !           272: tme_recode_x86_shift shrl 64 shrq
        !           273: tme_recode_x86_shift shra 64 sarq
        !           274: #else  /* !__x86_64__ */
        !           275: tme_recode_x86_double_shift tme_recode_x86_shll64 shll shldl %esi %edi
        !           276: tme_recode_x86_double_shift tme_recode_x86_shrl64 shrl shrdl %edi %esi
        !           277: tme_recode_x86_double_shift tme_recode_x86_shra64 sarl shrdl %edi %esi 1
        !           278: #endif /* !__x86_64__ */
        !           279: tme_recode_x86_shift shll 32 shll
        !           280: tme_recode_x86_shift shrl 32 shrl
        !           281: tme_recode_x86_shift shra 32 sarl
        !           282: tme_recode_x86_shift shll 16 shll
        !           283: tme_recode_x86_shift shrl 16 shrl
        !           284: tme_recode_x86_shift shra 16 sarl
        !           285: tme_recode_x86_shift shll 8 shll
        !           286: tme_recode_x86_shift shrl 8 shrl
        !           287: tme_recode_x86_shift shra 8 sarl
        !           288: 
        !           289: _tme_recode_x86_shift_arithmetic_all:
        !           290:        TME_RECODE_X86_OPN(add)         TME_RECODE_X86_REG_HOST_SUBS_DST_N, TME_RECODE_X86_REG_HOST_SUBS_DST_N
        !           291:        TME_RECODE_X86_OPN(sbb)         TME_RECODE_X86_REG_HOST_SUBS_DST_N, TME_RECODE_X86_REG_HOST_SUBS_DST_N
        !           292:        ret
        !           293: 
        !           294: _tme_recode_x86_shift_logical_all:
        !           295:        TME_RECODE_X86_OPN(xor)         TME_RECODE_X86_REG_HOST_SUBS_DST_N, TME_RECODE_X86_REG_HOST_SUBS_DST_N
        !           296:        ret

unix.superglobalmegacorp.com

This archive runs on limited infrastructure. Preserving old code on modern bandwidth. Automated agents are requested to crawl responsibly.