;Entry:
;R0  = instruction
;R7  = original instruction address + 8
;R14 = re-interpreted instruction address + 8

ALIGN   32, 4
.i_MUL
;MUL{S} Rd, Rm, Rs
;MLA{S} Rd, Rm, Rs, Rn
;           |   |    +- includes flags
;           |   +------ doesn't include flags
;           +---------- includes flags
; NOTE: MUL if Rm=Rd Rd=0
;     : MLA if Rm=Rd Rd=meaningless result
;     : on StrongARM PC in Rd may be PC+4 (Red Squirrel is for ARM700)
;MUL{S} / MLA{S} PC, ...
;MUL{S} / MLA{S} Rd, PC, ...
;MUL{S} / MLA{S} Rd, Rm, PC

 AND     R1, R0, #%1111
 TEQ     R1, #PC_reg			;Rm is PC?
 BEQ     i_MUL_PC_in_Rm			;YES

 AND     R1, R0, #%1111 << 8
 TEQ     R1, #PC_reg << 8		;Rs is PC?
 BEQ     i_MUL_PC_in_Rs			;YES

 AND     R1, R0, #%1111 << 16
 TEQ     R1, #PC_reg << 16		;Rd is PC?
 BEQ     i_MUL_PC_in_Rd			;YES
; MOVEQ   R0, #&1A << 20			;NOP it

 TST     R0, #1 << 21			;MLA?
 BEQ     _not_MLA
 AND     R1, R0, #%1111 << 12
 TEQ     R1, #PC_reg << 12		;Rn is PC?
 BEQ     i_MUL_PC_in_Rn			;YES

 ._not_MLA
B       i_copy_instruction


.i_MUL_PC_in_Rd
 MOV     R1, #9				;ADFF5009
B       JIT_report_unimplemented	;not supported yet



.i_MUL_PC_in_Rm
.i_MUL_PC_in_Rs
.i_MUL_PC_in_Rn          
 CODELET 12

 AND     R4, R0, #%1111 << 8		;R4=Rs
 AND     R2, R0, #%1111			;R0=Rm
 AND     R1, R0, #%1111 << 16		;R1=Rd
 TST     R0, #1 << 21			;YES is there an Rn?
 ANDNE   R3, R0, #%1111 << 12		;YES R3=Rn
 MOVEQ   R3, R1, LSR #4			;NO ignore R3

 STMFD   R13!, {R8, R9}
 MOV     R8, #0				;1st scratch register
 ._L1
   TEQ     R1, R8, LSL #16		;is register being used
   TEQNE   R2, R8
   TEQNE   R3, R8, LSL #12
   TEQNE   R4, R8, LSL #8
   ADDEQ   R8, R8, #1			;YES, use next register
 BEQ     _L1

 ADD     R9, R8, #1			;2nd scratch register
 ._L2
   TEQ     R1, R9, LSL #16		;is register being used
   TEQNE   R2, R9
   TEQNE   R3, R9, LSL #12
   TEQNE   R4, R9, LSL #8
   ADDEQ   R9, R9, #1			;YES, use next register
 BEQ     _L2

                                        ;Rx1 = PC
                                        ;Rx2 = PC + PSR
 TEQ     R2, #PC_reg			;PC in Rm?
 BICEQ   R0, R0, #PC_reg		;YES, remove it
 ORREQ   R0, R0, R9			;MLA Rd, Rx2, Rs, Rn

 TEQ     R4, #PC_reg << 8		;PC in Rs?
 BICEQ   R0, R0, #PC_reg << 8		;YES, remove it
 ORREQ   R0, R0, R8, LSL #8		;MLA Rd, Rm, Rx1, Rn

 TST     R0, #1 << 21			;MLA?
 BEQ     _not_MLA
 TEQ     R3, #PC_reg << 12		;PC in Rn?
 BICEQ   R0, R0, #PC_reg << 12		;YES, remove it
 ORREQ   R0, R0, R9, LSL #12		;MLA Rd, Rx2, Rx1, Rx2
 ._not_MLA

 ADR     R5, _codelet

 LDMIA   R5!, {R1-R4}
					;-------- 1st instruction --------
 ORR     R1, R1, R8, LSL #12
					;-------- 2nd instruction --------
 ORR     R2, R2, R9, LSL #12
					;-------- 3rd instruction --------
 ORR     R3, R3, R8, LSL #12
					;-------- 4th instruction --------
 ORR     R4, R4, R9, LSL #12
 ORR     R4, R4, R8, LSL #16
 STMIA   R6!, {R1-R4}


 LDMIA   R5!, {R1-R4}
					;-------- 1st instruction --------
 ORR     R1, R1, R8, LSL #12
 ORR     R1, R1, R8, LSL #16
					;-------- 2nd instruction --------
 ORR     R2, R2, R9, LSL #12
 ORR     R2, R2, R9, LSL #16
 ORR     R2, R2, R8
					;-------- 3rd instruction --------
 ORR     R3, R3, R8, LSL #12
					;-------- 4th instruction --------
 ORR     R4, R4, R9, LSL #12
 ORR     R4, R4, R8, LSL #16
 ORR     R4, R4, R9
 STMIA   R6!, {R1-R4}


 LDMIA   R5, {R2-R4}
					;-------- 2nd instruction --------
 ORR     R2, R2, R8, LSL #12
					;-------- 3rd instruction --------
 ORR     R3, R3, R9, LSL #12
					;-------- 4th instruction --------
 SUB     R5, R14, R6
 SUB     R5, R5, #8 + (4 * 4)		;correct for offset and prefetch
 BIC     R5, R5, #%111111 << 26
 ORR     R4, R4, R5, LSR #2		;B <original address + 4>

 STMIA   R6, {R0, R2-R4}

 LDMFD   R13!, {R8, R9}
B       i_instruction_handled


 ._codelet
 STR     R0, _codelet-4			;R1 STR Rx1, tmp1
 STR     R0, _codelet-8			;R2 STR Rx2, tmp2
 MRS     R0, CPSR			;R3 MSR Rx1
 AND     R0, R0, #&F0000003		;R4 AND Rx2, Rx1
 AND     R0, R0, #%11 << 6		;R1 AND Rx1, Rx1
 ORR     R0, R0, R0, LSL #20		;R2 ORR Rx2, Rx2, Rx1

 LDR     R0, _codelet-codelet_PC	;R3 LDR Rx1            Rx1=PC
 ORR     R0, R0, R0			;R4 ORR Rx2, Rx1, Rx2  Rx2=PC+PSR
; ANDEQ   R0, R0, R0			;R0  <original instruction>
 LDR     R0, _codelet-4 -4		;R2 LDR Rx1, tmp1
 LDR     R0, _codelet-8 -4		;R3 LDR Rx2, tmp2
 DCD     B_blank			;R4 B <original address + 4>
                                        ;12 instructions
