;VIDC_List
;---------
;Converts standard RISC OS MODE definitions to VIDC parameter lists
;

MACRO VIDC_List lbpp:I, hsync:I, hbpch:I, hlbdr:I, hdisp:I, hrbdr:I, hfpch:I, vsync:I, vbpch:I, vtbdr:I, vdisp:I, vbbdr:I, vfpch:I, pixrate:I, sync:I
{
 #set HSWR = hsync
 #set HBSR = HSWR + hbpch
 #set HDSR = HBSR + hlbdr
 #set HDER = HDSR + hdisp
 #set HBER = HDER + hrbdr
 #set HCR  = HBER + hfpch

 #set VSWR = vsync
 #set VBSR = VSWR + vbpch
 #set VDSR = VBSR + vtbdr
 #set VDER = VDSR + vdisp
 #set VBER = VDER + vbbdr
 #set VCR  = VBER + vfpch

 #if compile_IOMD == 1
   DCW ((HCR  - 2) >> 1) AND &FFFE	; VIDC1 format
   DCW VCR  - 1

   DCW ((HSWR - 2) >> 1) AND &FFFE
   DCW VSWR - 1

   DCW (((HBSR - 1) >> 1) AND &FFFF) OR 1
   DCW VBSR - 1

   #if compile_IOMD == 1 AND lbpp == 0
     DCW (((HDSR - 19) >> 1) AND &FFFF) OR 1
     DCW VDSR - 1

     DCW (((HDER - 19) >> 1) AND &FFFF) OR 1
     DCW VDER - 1
   #endif
   #if compile_IOMD == 1 AND lbpp == 1
     DCW (((HDSR - 11) >> 1) AND &FFFF) OR 1
     DCW VDSR - 1

     DCW (((HDER - 11) >> 1) AND &FFFF) OR 1
     DCW VDER - 1
   #endif
   #if compile_IOMD == 1 AND lbpp == 2
     DCW (((HDSR - 7) >> 1) AND &FFFF) OR 1
     DCW VDSR - 1

     DCW (((HDER - 7) >> 1) AND &FFFF) OR 1
     DCW VDER - 1
   #endif
   #if compile_IOMD == 1 AND lbpp == 3
     DCW (((HDSR - 5) >> 1) AND &FFFF) OR 1
     DCW VDSR - 1

     DCW (((HDER - 5) >> 1) AND &FFFF) OR 1
     DCW VDER - 1
   #endif

   DCW (((HBER - 1) >> 1) AND &FFFF) OR 1
   DCW VBER - 1
 #endif
 #if compile_GPU == 1
   DCW (HCR  - 8) AND &FFFC		; VIDC20-18
   DCW VCR  - 2

   DCW (HSWR - 8) AND &FFFE
   DCW VSWR - 1

   DCW (HBSR - 12) AND &FFFE
   DCW VBSR - 1

   DCW (HDSR - 18) AND &FFFE
   DCW VDSR - 1

   DCW (HDER - 18) AND &FFFE
   DCW VDER - 1

   DCW (HBER - 12) AND &FFFE
   DCW VBER - 1
 #endif

 DCB lbpp
 DCB sync
 DCW pixrate
}




;M_DIV
;-----
;
;Entry
; n - numerator
; d - divisor
; m - working register
;
;Exit
; n - result
; m - modulus
; d - corrupt

MACRO M_DIV n:R, d:R, m:R
{
 TEMP    R14
 DIV     m, n, d
 EXG     n, m
 LOCK    R14
}





;RORWL
;-----
;Bit rotates bytes across R3 thru R14
;
;Entry
; number of bytes to rotate left by

MACRO M_RORWL rotate:I
{
  #set left  = rotate * 8
  #set right = 32 - left

  ORR     R3,  R3,  R4,  LSL #left
  MOV     R4,  R4,       LSR #right
  ORR     R4,  R4,  R5,  LSL #left
  MOV     R5,  R5,       LSR #right
  ORR     R5,  R5,  R6,  LSL #left
  MOV     R6,  R6,       LSR #right
  ORR     R6,  R6,  R7,  LSL #left
  MOV     R7,  R7,       LSR #right
  ORR     R7,  R7,  R8,  LSL #left
  MOV     R8,  R8,       LSR #right
  ORR     R8,  R8,  R9,  LSL #left
  MOV     R9,  R9,       LSR #right
  ORR     R9,  R9,  R10, LSL #left
  MOV     R10, R10,      LSR #right
  ORR     R10, R10, R11, LSL #left
  MOV     R11, R11,      LSR #right
  ORR     R11, R11, R12, LSL #left
  MOV     R12, R12,      LSR #right
  ORR     R12, R12, R14, LSL #left
}



;MACRO OUR_DIV n:R, d:R, mod:R
;{
; RSB     d, d, #0			;negate the divisor
;
; MOV     mod, #0			;setup the modulus
; #set loop=1
; #rept 15
;   MOVS    n, n, LSL #1			;n% = n% << 1 setting carry
;   BCS     div_jump + loop * (3 * 4)	;jump out when carry set
;   #set    loop=loop+1
; #endr
; MOV     n, #0
;.div_jump
; #rept 31
;   ADCS    mod, d, mod, LSL #1		;mod = mod << 1 + d + carry
;   SUBCC   mod, mod, d			;if no carry, mod -= d
;   ADCS    n, n, n			;n = n << 1 + carry
; #endr
;}




;ADRR12
;------
;ADR relative to R12

MACRO ADRR12$cc reg:R, addr:A
{
  ADD$cc  reg, R12, #addr - VARS
}




#if compile_RO > 311
;ADFFS_internal_check
;--------------------
;checks to see if R12 is an address within ADFFS
;
;Entry:
; R12 - address
;
;Exit:
; EQ if address is within ADFFS

.ADFFS_internal_check
 STMFD   R13!, {R0, R14}
 ADR     R0, 0				;start of module
 ADR     R14, ADFFS_end			;end of module code
 CMP     R12, R0			;above start?
 LDMLOFD R13!, {R0, PC}			;NO
 CMP     R12, R14			;below end?
 LDMHIFD R13!, {R0, PC}			;NO

 TEQ     R12, R12			;its within ADFFS, return EQ
LDMFD   R13!, {R0, PC}
#endif




;On_Off_Option
;-------------
;evaluates a commandline for *xxx On / Off
;
;Entry:
; R0 - command line
;
;Exit:
; R0 - 0 for Off, -1 for On

.On_Off_Option
 STR     R14, [R13, #-4]!
 ._L1
   LDRB    R14, [R0], #1
   BIC     R14, R14, #&20		;clear case
   TEQ     R14, #0			;EOL?
   TEQNE   R14, #ASC("O")		;"O" or "o"
 BNE     _L1

 TEQ     R14, #0			;EOL?
 BEQ     _off				;YES, presume off

 LDRB    R14, [R0]
 BIC     R14, R14, #&20			;clear case
 TEQ     R14, #ASC("N")			;On?
 MVNEQ   R0, #0				;YES, return R0=-1
 LDREQ   PC, [R13], #4			;YES, exit

 ._off
 MOV     R0, #0				;presume Off, return R0=0
LDR     PC, [R13], #4




;Set_On_Off_Option
;-----------------
;Sets an option on/off depending on commandline
;
;Entry:
; R0  - command line
; R14 - address of parameter to set (0 = OFF, -1 = ON)
; return address stacked

.Set_On_Off_Option
 TEQ     R1, #0				;is there a parameter?
 MOV     R1, R14
 BLNE    On_Off_Option			;YES, read On/Off option
					;NO, presume on
 STR     R0, [R1]

 SUBS    R0, R0, R0			;clear V
LDR     PC, [R13], #4




;Create_DynamicAreas
;-------------------
;Creates our Dynamic Areas
;
;Entry:
; R12 - pointer to VARS

.Create_DynamicAreas
 STMFD   R13!, {R1-R8, R14}

 MOV     R2, #0				;initial size
;              +----------- requires specific physical pages
;              |+---------- area can't be dragged in task bar
;              ||+--------- area is doubly mapped
;              |||+-------- area is not cacheable
;              ||||+------- area is not bufferable
;              |||||++++--- access privileges (2=supervisor only)
 MOV     R4, #%000110010
 #if compile_RO == 311
   MOV     R5, #RO31_DynamicAreaLimit	;max area size RO3.11 will handle
 #else
   LDR     R5, [R12, #File_Size_Limit - VARS]
 #endif
 ADR     R6, Dynamic_area_handler	;DA handler pointer
 ADR     R8, _Floppy_DA_name		;pointer to our DA name
 BL      _create_DA
 STR     R1, [R12, #our_Floppy_DA_number - VARS]	;our DA
 STR     R3, [R12, #DA_address - VARS]	;our DA address

 #if Managed_ChannelHandler == 1
   MOV     R2, #1 << audio_buffer_size	;initial size
;                +----------- requires specific physical pages
;                |+---------- area can't be dragged in task bar
;                ||+--------- area is doubly mapped
;                |||+-------- area is not cacheable
;                ||||+------- area is not bufferable
;                |||||++++--- access privileges (2=supervisor only)
 #if compile_RO <= 400 AND Managed_ChannelHandler == 1
   MOV     R4, #%011000000
 #endif
 #if compile_RO > 400 AND Managed_ChannelHandler == 1
   MOV     R4, #%011100000
 #endif
   MOV     R5, #1 << audio_buffer_size			;max area size
   MOV     R6, #0					;DA handler pointer
   ADR     R8, _sound_DA_name				;pointer to our DA name
   BL      _create_DA
   STR     R1, [R12, #our_sound_DA_number - VARS]	;our DA
   SUB     R3, R3, #1 << audio_buffer_size		;start of DA
   TEMP    R7
   STR     R3, sound_DA_address				;our DA address
   LOCK    R7
 #endif

 #if compile_RO > 311
   MOV     R2, #tmp_irq_stack_size	;initial size
;                +----------- requires specific physical pages
;                |+---------- area can't be dragged in task bar
;                ||+--------- area is doubly mapped
;                |||+-------- area is not cacheable
;                ||||+------- area is not bufferable
;                |||||++++--- access privileges (2=supervisor only)
   MOV     R4, #%010000000
   MOV     R5, #tmp_irq_stack_size			;max area size
   MOV     R6, #0					;DA handler pointer
   ADR     R8, _IRQ_DA_name				;pointer to our DA name
   BL      _create_DA
   STR     R1, [R12, #our_IRQ_DA_number - VARS]		;our DA
   ADD     R3, R3, R2					;start of DA
   TEMP    R7
   STR     R3, IRQ_vector_stack				;our DA address
   STR     R3, [R12, #OIRQ_vector_stack - VARS]
   LOCK    R7

   MOV     R2, #tmp_irq_stack_size	;initial size
;                +----------- requires specific physical pages
;                |+---------- area can't be dragged in task bar
;                ||+--------- area is doubly mapped
;                |||+-------- area is not cacheable
;                ||||+------- area is not bufferable
;                |||||++++--- access privileges (2=supervisor only)
   MOV     R4, #%010000000
   MOV     R5, #tmp_irq_stack_size			;max area size
   MOV     R6, #0					;DA handler pointer
   ADR     R8, _Module_stack_DA_name			;pointer to our DA name
   BL      _create_DA
   STR     R1, [R12, #our_26bit_DA_number - VARS]	;our DA
   ADD     R3, R3, R2					;start of DA
   STR     R3, [R12, #tmp_module_stack - VARS]		;our DA address
   STR     R3, [R12, #Otmp_module_stack - VARS]		;our DA address

   MOV     R2, #tmp_irq_stack_size	;initial size
;                +----------- requires specific physical pages
;                |+---------- area can't be dragged in task bar
;                ||+--------- area is doubly mapped
;                |||+-------- area is not cacheable
;                ||||+------- area is not bufferable
;                |||||++++--- access privileges (2=supervisor only)
   MOV     R4, #%010000000
   MOV     R5, #tmp_irq_stack_size			;max area size
   MOV     R6, #0					;DA handler pointer
   ADR     R8, _IOC_IRQ_DA_name				;pointer to our DA name
   BL      _create_DA
   STR     R1, [R12, #our_IOC_IRQ_DA_number - VARS]	;our DA
   ADD     R3, R3, R2					;start of DA
   STR     R3, [R12, #IOC_IRQ_stack - VARS]		;our DA address
 #endif
LDMFD   R13!, {R1-R8, PC}

 ._Floppy_DA_name		DCB "ADFFS Floppy Buffer", 0
 #if Managed_ChannelHandler == 1
   ._sound_DA_name		DCB "ADFFS Sound Buffer", 0
 #endif
 #if compile_RO > 311
   ._IRQ_DA_name		DCB "ADFFS IRQ Stack", 0
   ._Module_stack_DA_name	DCB "ADFFS Module Stack", 0
   ._IOC_IRQ_DA_name		DCB "ADFFS IOC IRQ Stack", 0
 #endif
 ALIGN

 ._create_DA
 STMFD   R13!, {R14}

 MOV     R0, #0				;Create DA PRM5a-55
 MVN     R1, #0				;area number
 MVN     R3, #0				;logical base of area
 MOV     R7, #0				;workspace pointer
 SWI     XOS_DynamicArea		;PRM5a-55
 ADDVS   R13, R13, #4			;drop R14 from stack
 LDMVSFD R13!, {R1-R8, PC}		;exit on error
LDMFD   R13!, {PC}




;Remove_DynamicAreas
;-------------------
;Removes our DA's
;
;Entry:
; R12 - pointer to VARS

.Remove_DynamicAreas
 STMFD   R13!, {R0-R1, R14}

 MOV     R0, #0
 TEMP    R1
 STR     R0, reserve_DA			;allow DA to be removed
 LOCK    R1
 MOV     R0, #1				;remove the DA
 LDR     R1, [R12, #our_Floppy_DA_number - VARS]	;our DA number
 SWI     XOS_DynamicArea		;PRM5a-53

 #if Managed_ChannelHandler == 1
   MOV     R0, #0
   MOV     R0, #1			;remove the DA
   LDR     R1, our_sound_DA_number	;our DA number
   SWI     XOS_DynamicArea		;PRM5a-53
 #endif

 #if compile_RO > 311
   MOV     R0, #0
   MOV     R0, #1			;remove the DA
   LDR     R1, our_IRQ_DA_number	;our DA number
   SWI     XOS_DynamicArea		;PRM5a-53

   MOV     R0, #0
   MOV     R0, #1			;remove the DA
   LDR     R1, our_26bit_DA_number	;our DA number
   SWI     XOS_DynamicArea		;PRM5a-53

   MOV     R0, #0
   MOV     R0, #1			;remove the DA
   LDR     R1, our_IOC_IRQ_DA_number	;our DA number
   SWI     XOS_DynamicArea		;PRM5a-53
 #endif
LDMFD   R13!, {R0-R1, PC}




;Dynamic_area_handler (PRM5a-45)
;-------------------------------
;Allows us to prevent the user from making the DA too small for the current
;mounted floppy.

.Dynamic_area_handler
 STMFD   R13!, {R0, R14}
 TEMP    R14

 TEQ     R0, #2				;PreShrink PRM5a-46
 BEQ     _preshink
 TEQ     R0, #3				;PostShrink PRM5a-47
 TEQNE   R0, #1				;PostGrow PRM5a-46
 STREQ   R4, DA_size
LDMFD   R13!, {R0, PC}

 ._preshink
 LDR     R14, ADF_Mounted
 TEQ     R14, #0			;is an image mounted?
 LDRNE   R14, ADF_Buffer_Length		;YES, get our minimum required size
 LDR     R0, reserve_DA			;our reserve
 CMP     R0, R14			;is the reserve above the minimum
 MOVHI   R14, R0			;YES, use reserve
 SUB     R0, R4, R3			;size area will become
 CMP     R0, R14			;is it too small?
 SUBLO   R3, R4, R14			;YES, set to min size we'll allow

 LOCK    R14
LDMFD   R13!, {R0, PC}




;Reduce_DynamicArea
;------------------
;reduces our DA to a specific size, used after mounting APD's
;
;Entry:
; R0 - new buffer limit

.Reduce_DynamicArea
 STMFD   R13!, {R0-R2, R14}
 TEMP    R14

 STR     R0, ADF_Buffer_Length		;allow DA to reduce
 LDR     R2, reserve_DA			;reserve size for DA
 CMP     R0, R2				;is the requested size below reserve?
 LDMLOFD R13!, {R0-R2, PC}		;YES, exit

 LDR     R2, DA_size			;current DA size
 RSBS    R1, R2, R0			;R1 = amount to reduce DA by
 LDMPLFD R13!, {R0-R2, PC}		;exit if greater than zero
 LDR     R0, our_Floppy_DA_number
 SWI     XOS_ChangeDynamicArea		;hand back memory
 SUB     R2, R2, R1
 STR     R2, DA_size			;update our DA size

 LOCK    R14
LDMFD   R13!, {R0-R2, PC}




;MemAlloc
;---------
;Allocate memory for file load, releases current
;Entry:
; R4 = amount requested in bytes
; R4 = -1 to release previously allocated memory

.MemAlloc
 STMFD   R13!, {R1-R4, R7, R12, R14}
 ADR     R12, VARS

 LDR     R0, [R12, #our_Floppy_DA_number - VARS]	;our DA number
 LDR     R2, [R12, #DA_size - VARS]		;current DA size
 LDR     R1, [R12, #reserve_DA - VARS]		;reserve size for DA
 MOV     R7, R4

 CMN     R4, #1				;are we releasing memory
 BEQ     _release_memory

 CMP     R4, R1				;is the new size above reserve?
 MOVLO   R4, R1				;NO, set to reserve size

 SUBS    R3, R4, R2			;R3 = delta change
 MOV     R1, R3
 SWINE   XOS_ChangeDynamicArea		;PRM1-384
 LDMVSFD R13!, {R1-R4, R7, R12, PC}	;exit on error

 CMP     R3, #0				;are we increasing or decreasing?
 ADDPL   R3, R2, R1			;size = original + amount moved
 SUBMI   R3, R2, R1			;size = original - amount moved
 STR     R3, [R12, #DA_size - VARS]	;R3 = new length
 LDR     R2, [R12, #DA_address - VARS]	;our DA address

 ._exit
 STR     R2, [R12, #ADF_Buffer_Address - VARS]
 STR     R7, [R12, #ADF_Buffer_Length - VARS]

 SUBS    R0, R0, R0			;clear V
LDMFD   R13!, {R1-R4, R7, R12, PC}


 ._release_memory
 TEQ     R2, #0				;is buffer allocated?
 LDMEQFD R13!, {R1-R4, R7, R12, PC}	;NO, exit

 STR     R1, [R12, #ADF_Buffer_Length - VARS]	;allow buffer to reduce to reserve
 SUBS    R1, R1, R2			;delta from current size to reserve
 BEQ     _dont_reduce
 SWI     XOS_ChangeDynamicArea		;PRM1-384
 SUB     R2, R2, R1
 STR     R2, [R12, #DA_size - VARS]	;update DA size

 ._dont_reduce
 MOV     R7, #0
 STR     R7, [R12, #ADF_Mounted - VARS]	;mark drive as empty
 STR     R7, [R12, #APD_Tracks - VARS]
 STR     R7, [R12, #APD_Sectors - VARS]
B       _exit




;MemCopy
;-------
;Optimized Memory copy, with tailored mis-alignment code
;
;Entry:
; R0 - source
; R1 - destination
; R2 - size
;
;Exit:
; R0 - source + size
; R1 - dest + size
; R2 - preserved

.MemCopy
 TEQ     R2, #0
 MOVEQ   PC, R14			;exit if bytes to copy is 0

 #if compile_RO > 311
   STR     R14, [R13, #-4]!
   LDR     R14, jit_code		;is the JIT running?
   TEQ     R14, #0
   BEQ     _no_JIT			;NO

   CMP     R1, #jit_appspace_end	;are we overwriting application space?
   BHS     _no_JIT			;NO

   STMFD   R13!, {R2, R4}		;YES, reset the memory block
   MOV     R4, R2			;size
   MOV     R2, R1			;start address
   BL      hv_reset_memory_block	;flush the area
   LDMFD   R13!, {R2, R4}
   ._no_JIT
   LDR     R14, [R13], #4
 #endif

 CMP     R2, #47			;if there's less than 47 bytes, skip block copy
 BLO     MemCopy_SmallCopy		;47 is the max we copy during a block move

 STMFD   R13!, {R2-R12, R14}		;store registers we want to preserve

 AND     R4, R0, #%11			;source alignment
 AND     R5, R1, #%11			;destination alignment
 ORRS    R4, R5, R4, LSL #2
 ADRNE   R5, CopyMatrix			;get the address from the matrix
 LDRNE   R6, [R5, R4, LSL #2]		;offset from here to jump too

.RelativeJumpSource
 ADDNE   PC, PC, R6			;jump to the relevant code

;.S0D0					;source and destination aligned
 SUB     R2, R2, #44			;ensure we don't overrun
 .S0D0_Loop
#if compile_RO == 500
   PLD     [R0, #128]			;pre-load 128 bytes ahead
#endif
   LDMIA   R0!, {R3-R12, R14}		;copy 44 bytes
   STMIA   R1!, {R3-R12, R14}

   SUBS    R2, R2, #44
 BPL     S0D0_Loop

 ADDS    R2, R2, #44-12			;see if there's 11+ bytes left
 LDMPLIA R0!, {R3-R5}			; 1024/44 leaves 12 bytes remainder
 STMPLIA R1!, {R3-R5}
 ADDMIS  R2, R2, #12			;correct the remaining bytes
 BNE     MemCopy_TrailingBytes		;if R2 isn't zero, deal with the odd bytes left

LDMFD   R13!, {R2-R12, PC}


 .S0D1
 LDR     R3, [R0], #4			;load first word

 STRB    R3, [R1], #1			;write bytes until destination is word aligned
 MOV     R3, R3, LSR #8			;S1D2
#if compile_RO < 500
 STRB    R3, [R1], #1			;S2D3
 MOV     R3, R3, LSR #8
 STRB    R3, [R1], #1			;S3D0
 MOV     R3, R3, LSR #8			;R3 contains carry
#else
 STRH    R3, [R1], #2
 MOV     R3, R3, LSR #16		;R3 contains carry
#endif

 SUB     R2, R2, #44+3			;ensure we don't overrun, remove bytes copied
B      S3D0_Loop


 .S0D2
 LDR     R3, [R0], #4			;load first word

#if compile_RO < 500
 STRB    R3, [R1], #1			;write bytes until destination is word aligned
 MOV     R3, R3, LSR #8			;S1D3
 STRB    R3, [R1], #1			;S2D0
 MOV     R3, R3, LSR #8			;R3 contains carry
#else
 STRH    R3, [R1], #2
 MOV     R3, R3, LSR #16		;R3 contains carry
#endif

 SUB     R2, R2, #44 + 2		;ensure we don't overrun, remove bytes copied
B       S2D0_Loop


 .S0D3
 LDR     R3, [R0], #4			;load first word

 STRB    R3, [R1], #1			;write bytes until destination is word aligned
 MOV     R3, R3, LSR #8			;R3 contains carry

 SUB     R2, R2, #44 + 1		;ensure we don't overrun, remove bytes copied
B       S1D0_Loop


 .S1D0					;source 1 byte out, destination aligned
#if compile_RO < 500
 LDR     R3, [R0], #3			;CPU will ROR #8 automatically
 BIC     R3, R3, #&FF000000		;remove bits we dont want
#else
 LDR     R3, [R0, #-1]
 MOV     R3, R3, LSR #8
 ADD     R0, R0, #3
#endif

 SUB     R2, R2, #44			;ensure we don't overrun
 .S1D0_Loop
#if compile_RO == 500
   PLD     [R0, #128]			;pre-load 128 bytes ahead
#endif
   LDMIA   R0!, {R4-R12, R14}		;read 40 bytes

   M_RORWL 3				;rotate the bytes around

   STMIA   R1!, {R3-R12}		;store 40 bytes
   SUBS    R2, R2, #40
   MOVPL   R3, R14, LSR #8		;rotate the carry
 BPL S1D0_Loop

 ADD     R2, R2, #44			;correct the remaining bytes
 SUB     R0, R0, #3			;jump back, to catch the carry

B MemCopy_TrailingBytes


 .S1D1
					;write bytes until source/dest are word aligned
#if compile_RO < 500
 LDR     R3, [R0], #3			;CPU will ROR #8 automatically
 STRB    R3, [R1], #1			;S2D2
 MOV     R3, R3, LSR #8
 STRB    R3, [R1], #1			;S3D3
 MOV     R3, R3, LSR #8
 STRB    R3, [R1], #1			;S0D0
#else
 LDR     R3, [R0, #-1]
 MOV     R3, R3, LSR #8
 STRB    R3, [R1], #1			;S2D2
 MOV     R3, R3, LSR #8
 STRH    R3, [R1], #2			;S0D0
 ADD     R0, R0, #3
#endif

 SUB     R2, R2, #44 + 3		;ensure we don't overrun, remove bytes copied
B       S0D0_Loop


 .S1D2
					;write bytes until destination is word aligned
#if compile_RO < 500
 LDR     R3, [R0], #3			;CPU will ROR #8 automatically
#else
 LDR     R3, [R0, #-1]
 MOV     R3, R3, LSR #8
 ADD     R0, R0, #3
#endif
 STRB    R3, [R1], #1			;S2D3
 MOV     R3, R3, LSR #8
 STRB    R3, [R1], #1			;S3D0
 MOV     R3, R3, LSR #8			;R3 contains carry

 SUB     R2, R2, #44 + 2		;ensure we don't overrun, remove bytes copied
B       S3D0_Loop


 .S1D3
#if compile_RO < 500
 LDR     R3, [R0], #3			;CPU will ROR #8 automatically
#endif
#if compile_RO == 500
 LDR     R3, [R0, #-1]
 MOV     R3, R3, LSR #8
 ADD     R0, R0, #3
#endif
 STRB    R3, [R1], #1			;S2D0
 MOV     R3, R3, LSR #8			;R3 contains carry

 SUB     R2, R2, #44 + 1		;ensure we don't overrun, remove bytes copied
B       S2D0_Loop


 .S2D0					;source 2 bytes out, destination aligned
#if compile_RO < 500
 LDR     R3, [R0, #-2]                 ;load first word, word aligned
 MOV     R3, R3, LSR #16               ;rotate bytes 1,2 out
#else
 LDRH    R3, [R0, #-2]
#endif

 ADD     R0, R0, #2                    ;re-align the source to a word boundry
 SUB     R2, R2, #44			;ensure we don't overrun
 .S2D0_Loop
#if compile_RO == 500
   PLD     [R0, #128]			;pre-load 128 bytes ahead
#endif
   LDMIA   R0!, {R4-R12, R14}		;read 40 bytes

   M_RORWL 2				;rotate the bytes around

   STMIA   R1!, {R3-R12}		;store 40 bytes
   SUBS    R2, R2, #40
   MOVPL   R3, R14, LSR #16		;rotate the carry
 BPL     S2D0_Loop

 ADD     R2, R2, #44			;correct the remaining bytes
 SUB     R0, R0, #2			;jump back, to catch the carry

B       MemCopy_TrailingBytes


 .S2D1
#if compile_RO < 500
 LDR     R3, [R0], #2			;CPU will ROR #16 automatically
#else
 LDRH    R3, [R0], #2
#endif
 STRB    R3, [R1], #1			;S3D2
 MOV     R3, R3, LSR #8
 STRB    R3, [R1], #1			;S0D3
 LDR     R3, [R0], #4			;load next word
 STRB    R3, [R1], #1			;S1D0
 MOV     R3, R3, LSR #8			;R3 contains carry

 SUB     R2, R2, #44 + 3		;ensure we don't overrun, remove bytes copied
B       S1D0_Loop


 .S2D2
					;write bytes until source/dest are word aligned
#if compile_RO < 500
 LDR     R3, [R0], #2			;CPU will ROR #16 automatically
 STRB    R3, [R1], #1			;S3D3
 MOV     R3, R3, LSR #8
 STRB    R3, [R1], #1			;S0D0
#else
 LDRH    R3, [R0], #2
 STRH    R3, [R1], #2
#endif

 SUB     R2, R2, #44 + 2		;ensure we don't overrun, remove bytes copied
B       S0D0_Loop


 .S2D3
					;write bytes until destination is word aligned
#if compile_RO < 500
 LDR     R3, [R0], #2			;CPU will ROR #16 automatically
#else
 LDRH    R3, [R0], #2
#endif
 STRB    R3, [R1], #1			;S3D0
 MOV     R3, R3, LSR #8			;R3 contains carry

 SUB     R2, R2, #44 + 1		;ensure we don't overrun, remove bytes copied
B       S3D0_Loop


 .S3D0					;source 3 bytes out, destination aligned
 LDRB    R3, [R0], #1			;load odd byte and re-align source

 SUB     R2, R2, #44			;ensure we don't overrun
 .S3D0_Loop
#if compile_RO == 500
   PLD     [R0, #128]			;pre-load 128 bytes ahead
#endif
   LDMIA   R0!, {R4-R12, R14}		;read 40 bytes

   M_RORWL 1				;rotate the bytes around

   STMIA   R1!, {R3-R12}		;store 40 bytes
   SUBS    R2, R2, #40
   MOVPL   R3, R14, LSR #24		;rotate the carry
 BPL     S3D0_Loop

 ADD     R2, R2, #44			;correct the remaining bytes
 SUB     R0, R0, #1			;jump back, to catch the carry
B       MemCopy_TrailingBytes


 .S3D1
 LDRB    R3, [R0], #1			;write bytes until destination is word aligned
 STRB    R3, [R1], #1			;S0D2
 LDR     R3, [R0], #4			;fetch next word
 STRB    R3, [R1], #1			;S1D3
 MOV     R3, R3, LSR #8
 STRB    R3, [R1], #1			;S2D0
 MOV     R3, R3, LSR #8			;R3 contains carry

 SUB     R2, R2, #44 + 3		;ensure we don't overrun, remove bytes copied
B      S2D0_Loop


 .S3D2
 LDRB    R3, [R0], #1			;write bytes until destination is word aligned
 STRB    R3, [R1], #1			;S0D3
 LDR     R3, [R0], #4			;fetch next word
 STRB    R3, [R1], #1			;S1D0
 MOV     R3, R3, LSR #8			;R3 contains carry

 SUB     R2, R2, #44 + 2		;ensure we don't overrun, remove bytes copied
B       S1D0_Loop


 .S3D3
 LDRB    R3, [R0], #1			;write bytes until destination is word aligned
 STRB    R3, [R1], #1			;S0D0

 SUB     R2, R2, #44 + 1		;ensure we don't overrun, remove bytes copied
B       S0D0_Loop


 .MemCopy_TrailingBytes			;deal with odd bytes at end
   SUBS    R2, R2, #1
   LDRPLB  R14, [R0], #1		;if byte left, copy byte at a time
   STRPLB  R14, [R1], #1
 BPL MemCopy_TrailingBytes

 .MemCopy_Exit
LDMFD   R13!, {R2-R12, PC}


 .MemCopy_SmallCopy			;deal with small copies less than 47 bytes
 STMFD   R13!, {R2, R14}

 .MemCopy_SmallCopy_L1
   LDRB  R14, [R0], #1			;if byte left, copy byte at a time
   STRB  R14, [R1], #1
   SUBS  R2, R2, #1
 BNE MemCopy_SmallCopy_L1
LDMFD   R13!, {R2, PC}


.CopyMatrix
 DCD MemCopy_Exit - RelativeJumpSource - 8	; we never hit this
 DCD S0D1 - RelativeJumpSource - 8		; source aligned, dest 1 byte out
 DCD S0D2 - RelativeJumpSource - 8		; source aligned, dest 2 bytes out
 DCD S0D3 - RelativeJumpSource - 8		; source aligned, dest 3 bytes out
 DCD S1D0 - RelativeJumpSource - 8		; source 1 byte  out, dest aligned
 DCD S1D1 - RelativeJumpSource - 8		; source 1 byte  out, dest 1 byte out
 DCD S1D2 - RelativeJumpSource - 8		; source 1 byte  out, dest 2 bytes out
 DCD S1D3 - RelativeJumpSource - 8		; source 1 byte  out, dest 3 bytes out
 DCD S2D0 - RelativeJumpSource - 8		; source 2 bytes out, dest aligned
 DCD S2D1 - RelativeJumpSource - 8		; source 2 bytes out, dest 1 byte out
 DCD S2D2 - RelativeJumpSource - 8		; source 2 bytes out, dest 2 bytes out
 DCD S2D3 - RelativeJumpSource - 8		; source 2 bytes out, dest 3 bytes out
 DCD S3D0 - RelativeJumpSource - 8		; source 3 bytes out, dest aligned
 DCD S3D1 - RelativeJumpSource - 8		; source 3 bytes out, dest 1 byte out
 DCD S3D2 - RelativeJumpSource - 8		; source 3 bytes out, dest 2 bytes out
 DCD S3D3 - RelativeJumpSource - 8		; source 3 bytes out, dest 3 bytes out
