;Entry:
;R0  = instruction
;R1  = Rd << 12
;R2  = Rn << 16
;R3  = Rm if appropriate eg. LDR R0, [R1, R2] or LDR R0, [R1], R2
;R7  = original instruction address + 8
;R14 = re-interpreted instruction address + 8

ALIGN   32, 4

; LDR     Rd, [Rn], Rm
; LDR     Rd, [PC,
; LDR     PC, [Rn,
; LDR     PC, [PC,
; LDR     R0, [R0, #<immed>]!
; LDR     R0, [R0], #24

.i_LDR
 TEQ     R1, #PC_reg << 12		;is PC in Rd?
 BEQ     i_LDR_PC_in_Rd			;YES

 TEQ     R1, R2, LSR #4			;does Rd=Rn?
 BNE     _rd_rn_different		;NO

 TST     R0, #1 << 24			;post-indexed eg: LDR R0,[R0],<op2>
 BEQ     i_LDR_Rd_Rn_postindexed	;YES
					;pre-indexed eg: LDR R0,[R0,<op2>]{!}
 BIC     R0, R0, #1 << 21		;ensure it's not doing writeback

 ._rd_rn_different
 TEQ     R2, #PC_reg << 16		;Rn = PC?
 BNE     i_copy_instruction		;NO

 TST     R0, #1 << 25			;is Op2 a register?
 BEQ     i_LDR_immediate		;NO

 TEQ     R3, R1, LSR #12		;Rm=Rd?
 BNE     i_LDR_immediate		;NO

; LDR     Rd, [Rn], Rm
; LDR/STR Rd, [R0, R0]{!}    deprecated but appears to work on Pi3
; LDR/STR Rd, [R0], R0       deprecated but appears to work on Pi3

.i_LDR_STR_Rn
 TST     R0, #1 << 24			;is it pre-indexed
 MOVEQ   R1, #9				;YES, ADFF5009
 BEQ     JIT_report_unimplemented	;YES, not supported
 TST     R0, #1 << 21			;is it doing writeback
 MOVNE   R1, #9				;YES, ADFF5009
 BNE     JIT_report_unimplemented	;YES, not supported

 MOV     R5, R1				;preserve R1
 CODELET 5
 MOV     R1, R5				;preserve R1

 MOV     R4, #0
 TEQ     R1, R4, LSL #12		;is register being used
 TEQNE   R3, R4
 ADDEQ   R4, R4, #1			;YES, use next register
 TEQ     R1, R4, LSL #12		;is register being used
 TEQNE   R3, R4
 ADDEQ   R4, R4, #1			;YES, use next register

 ADR     R5, _codelet
 LDMIA   R5!, {R1-R2}			;get replacement instructions
					;-------- 1st instruction --------
 ORR     R1, R1, R4, LSL #12		;add Rd into instruction
					;-------- 2nd instruction --------
 ORR     R2, R2, R4, LSL #12		;add Rd into instruction
					;-------- 3rd instruction --------
 BIC     R3, R0, #%1111 << 16		;LDR/STR Rd, [__, ...
 ORR     R3, R3, R4, LSL #16		;LDR/STR Rd, [Rx, ...
 STMIA   R6!, {R1-R3}


 LDMIA   R5, {R1-R2}			;get replacement instructions
					;-------- 1st instruction --------
 ORR     R1, R1, R4, LSL #12		;add Rd into instruction
					;-------- 2nd instruction --------
 SUB     R5, R14, R6
 SUB     R5, R5, #8 + (2 * 4)		;correct for offset and prefetch
 BIC     R5, R5, #%111111 << 26
 ORR     R2, R2, R5, LSR #2		;B <original address + 4>

 STMIA   R6, {R1-R2}			;store codelet

B       i_instruction_handled


 ._codelet
 STR     R0, _codelet-4			;R1 STR Rx, tmp1
 LDR     R0, _codelet-codelet_PC	;R2 LDR Rx, ...
 ;DCD     %111001000001 << 20		;R3 LDR Rd, [Rx, ...
 LDR     R0, _codelet-4 -4		;R1 LDR Rx, tmp1
 DCD     B_blank			;R2 B <original address + 4>
					;5 instructions




; LDR R0, [R0], <op2>
.i_LDR_Rd_Rn_postindexed
 TST     R0, #1 << 25			;YES, is it an immediate?
 MOVEQS  R4, R0, LSL #32 - 12		; |+ YES, does the immediate = 0
 ORREQ   R0, R0, #1 << 24		;     |+ YES, change to pre-indexing

 ORR     R0, R0, #1 << 24		;change to pre-indexing
 BIC     R0, R0, #1 << 25		;change to immediate
 MOV     R0, R0, LSR #12
 MOV     R0, R0, LSL #12		;change to LDR R0,[R0,#0]
B       i_copy_instruction




; LDR/STR     Rd, [PC, #<immed>]{!}
; LDR/STR     Rd, [PC]{, #<immed>}
;

ALIGN   32, 4
.i_LDR_immediate
 BIC     R0, R0, #%1111 << 16		;clear Rn
					;------ 1st/2nd instruction -------
 TST     R0, #1 << 24			;is it pre-indexed
 MOVEQ   R1, #9				;YES, ADFF5009
 BEQ     JIT_report_unimplemented	;YES, not supported
 TST     R0, #1 << 21			;is it doing writeback (KerBang! @ 966C)
 MOVNE   R1, #9				;YES, ADFF5009
 BNE     JIT_report_unimplemented	;YES, not supported

 LDR     R2, i_LDR_immediate_codelet	;get 1st instruction
 ORR     R3, R0, R1, LSL #4		;add Rn into 2nd instruction
 ORR     R2, R2, R1			;add Rd into 1st instruction

 CODELET 3
					;-------- 3rd instruction --------
 LDR     R4, i_LDR_immediate_codelet + 4	;get 3rd instruction
 SUB     R5, R14, R6
 SUB     R5, R5, #8 + (3 * 4)		;correct for offset and prefetch
 BIC     R5, R5, #%111111 << 26
 ORR     R4, R4, R5, LSR #2		;B <original address + 4>

 STMIA   R6, {R2-R4}			;store codelet

B       i_instruction_handled


 .i_LDR_immediate_codelet
 LDR     R0, i_LDR_immediate_codelet-codelet_PC	;R2 LDR Rd, ...
; DCD     %111001000001 << 20		;R3 LDR Rd, [Rd, ...
 DCD     B_blank			;R4 B <original address + 4>
					;3 instructions



;i_LDR_PC_in_Rd
;--------------
;LDR PC, [Rx
;LDR PC, [PC

;Entry:
;R0  = instruction
;R1  = Rd << 12
;R2  = Rn << 16

ALIGN   32, 4
.i_LDR_PC_in_Rd
 TEQ     R2, #PC_reg << 16		;LDR PC, [PC ...?
 BEQ     i_LDR_PC_in_Rd_and_Rn		;YES, can't reuse codelet

;LDR PC, [Rx
 CODELET_REUSE 7
 B       i_LDR_PC_in_Rd_P1

;LDR PC, [PC
.i_LDR_PC_in_Rd_and_Rn
 MOV     R5, R1				;preserve R1
 CODELET 8
 MOV     R1, R5				;preserve R1

 .i_LDR_PC_in_Rd_P1
 MOV     R4, #0
 TEQ     R3, R4				;is register being used?
 TEQNE   R2, R4, LSL #16
 ADDEQ   R4, R4, #1			;YES, use next register
 TEQ     R3, R4				;is register being used?
 TEQNE   R2, R4, LSL #16
 ADDEQ   R4, R4, #1			;YES, use next register

 CMP     R4, #1
 ADRLO   R5, _codelet_R0
 ADREQ   R5, _codelet_R1
 ADRHI   R5, _codelet_R2

 TEQ     R2, #PC_reg << 16		;LDR PC, [PC ...?
 BICEQ   R0, R0, #%1111 << 16		;YES, LDR PC, [__, ...
 ORREQ   R0, R0, R4, LSL #16		; |+ LDR PC, [Rx
 BIC     R0, R0, #%1111 << 12		;LDR __, ...
 ORR     R0, R0, R4, LSL #12
 LDMIA   R5!, {R1, R4}
 STMNEIA R6!, {R1}			;NO, drop: LDR R0, _codelet_R0-codelet_PC
 STMEQIA R6!, {R1, R4}


 LDMIA   R5!, {R1-R3}
					;-------- 1st instruction --------
 MOV     R1, R0				;LDR Rx, ...
 STMIA   R6!, {R1-R3}			;store codelet


 LDMIA   R5!, {R0, R3-R4}
 SUBNE   R0, R0, #4			;correct
 SUBNE   R3, R3, #4			; |
 SUBNE   R4, R4, #4			; |+  optional instruction
 STMIA   R6!, {R0, R3-R4}

B       i_und_exit


 ._codelet_R0
 STR     R0, _codelet_R0-4		;R1 tmp1
 LDR     R0, _codelet_R0-codelet_PC	;R4 optional: LDR PC, [PC ... only
 DCD     0 << 12			;R1 LDR R0, ...
 ADD     R0, R0, #jit_code_start	;R2 correct
 BIC     R0, R0, #&FC000003		;R3
 STR     R0, _codelet_R0-8		;R0 tmp2
 LDR     R0, _codelet_R0-4		;R3 tmp1
 LDR     PC, _codelet_R0-8		;R4 tmp2
					;8 instructions

 ._codelet_R1
 STR     R1, _codelet_R1-4		;R1 tmp1
 LDR     R1, _codelet_R1-codelet_PC	;R4
 DCD     1 << 12			;R1 LDR R1, ...
 ADD     R1, R1, #jit_code_start	;R2
 BIC     R1, R1, #&FC000003		;R3
 STR     R1, _codelet_R1-8		;R0 tmp2
 LDR     R1, _codelet_R1-4		;R3 tmp1
 LDR     PC, _codelet_R1-8		;R4 tmp2


 ._codelet_R2
 STR     R2, _codelet_R2-4		;R1 tmp1
 LDR     R2, _codelet_R2-codelet_PC	;R4
 DCD     2 << 12			;R1 LDR R2, ...
 ADD     R2, R2, #jit_code_start	;R2
 BIC     R2, R2, #&FC000003		;R3
 STR     R2, _codelet_R2-8		;R0 tmp2
 LDR     R2, _codelet_R2-4		;R3 tmp1
 LDR     PC, _codelet_R2-8		;R4 tmp2



;LDR_check
;---------
;checks an LDR instruction for page zero addresses
;
;If it's condition fails, it checks the next instruction to see if its a
;JIT entry instruction, if not it exits without flushing the cache
;If the next instruction is a JIT entry instruction, the LDR is skipped and
;instruction processing continues

;Entry:
; The JIT has already executed up to the LDR, its now the 1st instruction to
; be seen by the JIT
;
; R0 - instruction
; R1 - Rd << 12
; R2 - Rn << 16
; R7 - PC + 8 (non-JIT space)
; stack: R0-R7 as they were on entry to the JIT
;
;Exit:
; R8-R12, R14 - preserved

#if High_Vectors == 0
ALIGN   32, 4
.LDR_check
 #if High_Vectors == 0 AND JIT_debug == 1
   ADR     R3, jit_LDR_interpreted
   LDR     R4, [R3]
   ADD     R4, R4, #1
   STR     R4, [R3]
 #endif

 AND     R3, R0, #%1111 << 28		;check instruction condition is met
 TEQ     R3, #%1110 << 28		;AL?
 BEQ     _P1				;YES

 LDR     R3, [PC, R3, LSR #28 - 2]	;load conditions
 B       _check_condition
;         NZCVnzcvNZCVnzcvNZCVnzcvNZCVnzcv
 DCD     %01000100000000000000000000000000	;EQ Z=1
 DCD     %01000000000000000000000000000000	;NE Z=0
 DCD     %00100010000000000000000000000000	;CS C=1
 DCD     %00100000000000000000000000000000	;CC C=0
 DCD     %10001000000000000000000000000000	;MI N=1
 DCD     %10000000000000000000000000000000	;PL N=0
 DCD     %00010001000000000000000000000000	;VS V=1
 DCD     %00010000000000000000000000000000	;VC V=0
 DCD     %01100010000000000000000000000000	;HI C=1 Z=0
 DCD     %00100000010001000000000000000000	;LS C=0 or Z=1
 DCD     %10011001100100000000000000000000	;GE N=1 V=1 or N=0 V=0
 DCD     %10011000100100010000000000000000	;LT N=1 V=0 or N=0 V=1
 DCD     %11011001110100000000000000000000	;GT N=1 Z=0 V=1 or N=0 Z=0 V=0
 DCD     %01000100100110001001000100000000	;LE Z=1 or N=1 V=0 or N=0 V=1
;         NZCVnzcvNZCVnzcvNZCVnzcvNZCVnzcv
;         MMMM^^^^MMMM^^^^MMMM^^^^--------	MASK followed by value

 ._check_condition
 MRS     R5, SPSR			;get flags
 AND     R5, R5, #%1111 << 28
 ._L1
   AND     R4, R5, R3			;flags to check
   AND     R6, R3, #%1111 << 24		;get flags values
   TEQ     R4, R6, LSL #4		;do flags match
   BEQ     _P1				;YES

   MOVS    R3, R3, LSL #8
 BNE     _L1				;loop if more condition checks

 AND     R3, R0, #%1111 << 28		;instruction condition
 LDR     R0, branch_to_jit_no_condition	;JIT entry instruction
 LDR     R1, [R14, #-8]			;get the current JIT instruction
 ORR     R0, R3, R0, LSR #4		;add conditional
 TEQ     R0, R1				;has it changed?
 BEQ     i_instruction_handled_no_codelet	;NO
B       i_copy_instruction_handled	;continue processing

 ._P1
 STMFD   R13!, {R0-R2, R14}		;preserve R0-R2, R14
 ADD     R0, R13, #4 * 4		;R0 points to stack on entry
 ADR     R14, DAV_r0
 STR     R7, [R14, #15 * 4]		;store relative PC in JIT register dump
 LDMIA   R0, {R0-R7}			;read stacked registers
 STMIA   R14!, {R0-R7}			;store in JIT register dump
 MOV     R0, R14

 MRS     R3, SPSR			;get original CPU state
 TST     R3, #%1111			;is abort in User mode?
 #if High_Vectors == 0 AND compile_IOMD == 1
   STREQ   R0, [R0, #0]			;fix for errata on SA-110 rev.2
 #endif
 STMEQIA R0, {R8-R14}^			;YES, get User mode registers
 BEQ     _P2				; |+ and skip IRQ/FIQ/SVC code

 MRS     R4, CPSR
 ORR     R3, R4, #%11 << 6		;disable IRQ/FIQ
 MSR     CPSR_c, R3			;switch to faulting mode
 #if High_Vectors == 0 AND compile_IOMD == 1
   NOP
 #endif
 STMIA   R0, {R8-R14}			;get aborting mode registers

 MSR     CPSR_c, R4			;switch back to Und32
 #if High_Vectors == 0 AND compile_IOMD == 1
   NOP
 #endif

 ._P2					;registers now dumped at DAV_r0
 SUB     R3, R0, #8 * 4			;R3 = register table
 LDMFD   R13!, {R0-R2, R14}		;preserve R0-R2, R14

 LDR     R7, [R3, R2, LSR #16 - 2]	;R7=Rn (address)
					;-------- DECODE INSTRUCTION --------
 TST     R0, #1 << 24			;LDR Rd, [Rn, xx]?
 BEQ     _LDR_decoded			;NO, address already decoded

					;Reference: ARM710VE-36
					;decode Operand 2

 TST     R0, #1 << 25			;LDR Rd, [Rn, #{+/-}<immed_12>] ?
 BNE     _xTR_shift			;NO, LDR Rd, [Rn, {+/-}Rm{,<shift>}]

				;:LDR Rd, [Rn], #{+/-}<immed_12>
				;:LDR Rd, [Rn, #{+/-}<immed_12>]!
 MOV     R4, R0, LSL #31 - 11		;R4 = offset in bits 31..20
 TST     R0, #1 << 23			;U (Up / Down)
 SUBEQ   R7, R7, R4, LSR #31 - 11	;LDR Rd, [Rn, #-<immed_12>]
 ADDNE   R7, R7, R4, LSR #31 - 11	;LDR Rd, [Rn, #+<immed_12>]
 B       _LDR_decoded
				;Reference: ARM710VE-25
				;NOTE: STR Rd, [Rn, {+/-}Rm, <shift_type> Rx]!
				;      is not a valid instruction
				;:LDR Rd, [Rn], {+/-}Rm {, <shift> #<immed_5>}
 ._xTR_shift			;:LDR Rd, [Rn, {+/-}Rm {, <shift> #<immed_5>}]!
 AND     R5, R0, #%1111			;R5 = Rm
 LDR     R2, [R3, R5, LSL #2]		;R2 = value of Rm
 AND     R1, R0, #%11111 << 7
 MOV     R1, R1, LSR #7			;R1 = #<immed_5>

 ANDS    R5, R0, #%11 << 5		;R5 = shift type
 MOVEQ   R4, R2, LSL R1			;R4=Rm, LSL #<immed_5>
 BEQ     _xTR_shift_decoded
 CMP     R5, #%10 << 5
 MOVLT   R4, R2, LSR R1			;R4=Rm, LSR #<immed_5>
 MOVEQ   R4, R2, ASR R1			;R4=Rm, ASR #<immed_5>
 BLS     _xTR_shift_decoded
				;Reference: ARM710VE-25
				;:LDR Rd, [Rn, {+/-}Rm, ROR #<immed_5>/RRX]!
 TST     R0, #%11111 << 7		;is it RRX?
 MOVNE   R4, R2, ROR R1			;NO, Rm, ROR #n
 BNE     _xTR_shift_decoded
				;:LDR Rd, [Rn, {+/-}Rm, RRX]!
 MRS     R4, SPSR			;get CPSR
 MOVS    R4, R4, LSL #3			; |  shift carry into C
 MOV     R4, R2, RRX			; |+ Rm, RRX

 ._xTR_shift_decoded
 TST     R0, #1 << 23			;U (Up / Down)
 ADDNE   R7, R7, R4			;Rm, <shift> #<immed_5>
 SUBEQ   R7, R7, R4			;-Rm, <shift> #<immed_5>

 ._LDR_decoded				;R7=address
 AND     R1, R0, #%1111 << 12		;R1=Rd

 CMP     R7, #page_zero_Scratch_Space	;in page zero?
 TEMP    R4
 MRSLO   R2, SPSR
 STRLO   R2, DAV_CPSR
 LOCK    R4
 BLO     translate_page_zero_read	;YES, perform the translation

 AND     R2, R0, #%1111 << 16		;R2=Rn
 AND     R3, R0, #%1111			;R3=Rm
 SUB     R7, R14, #jit_code_start	;original instruction address + 8
 B       i_LDR
#endif
