aboutsummaryrefslogtreecommitdiff
path: root/libpsn00b/psxpress/vlc.s
diff options
context:
space:
mode:
Diffstat (limited to 'libpsn00b/psxpress/vlc.s')
-rw-r--r--libpsn00b/psxpress/vlc.s404
1 files changed, 404 insertions, 0 deletions
diff --git a/libpsn00b/psxpress/vlc.s b/libpsn00b/psxpress/vlc.s
new file mode 100644
index 0000000..fe51642
--- /dev/null
+++ b/libpsn00b/psxpress/vlc.s
@@ -0,0 +1,404 @@
+# PSn00bSDK MDEC library (GTE-accelerated VLC decompressor)
+# (C) 2022 spicyjpeg - MPL licensed
+#
+# Register map:
+# - $a0 = ctx
+# - $a1 = output
+# - $a2 = max_size
+# - $a3 = input
+# - $t0 = window
+# - $t1 = next_window
+# - $t2 = remaining
+# - $t3 = quant_scale
+# - $t4 = is_v3
+# - $t5 = bit_offset
+# - $t6 = block_index
+# - $t7 = coeff_index
+# - $t8 = _vlc_huffman_table
+# - $t9 = &ac_jump_area
+
+.set noreorder
+
+.set VLC_Context_input, 0
+.set VLC_Context_window, 4
+.set VLC_Context_next_window, 8
+.set VLC_Context_remaining, 12
+.set VLC_Context_quant_scale, 16
+.set VLC_Context_is_v3, 18
+.set VLC_Context_bit_offset, 19
+.set VLC_Context_block_index, 20
+.set VLC_Context_coeff_index, 21
+
+.set DECDCTSMALLTAB_lut0, 0
+.set DECDCTSMALLTAB_lut2, 4
+.set DECDCTSMALLTAB_lut3, 36
+.set DECDCTSMALLTAB_lut4, 292
+.set DECDCTSMALLTAB_lut5, 308
+.set DECDCTSMALLTAB_lut7, 324
+.set DECDCTSMALLTAB_lut8, 356
+.set DECDCTSMALLTAB_lut9, 420
+.set DECDCTSMALLTAB_lut10, 484
+.set DECDCTSMALLTAB_lut11, 548
+.set DECDCTSMALLTAB_lut12, 612
+
+.section .text.DecDCTvlcStart
+.global DecDCTvlcStart
+.type DecDCTvlcStart, @function
+DecDCTvlcStart:
+ # Create a new context on-the-fly without writing it to memory then jump
+ # into DecDCTvlcContinue(), skipping context loading.
+ lw $t0, 8($a3) # window = (bs->data[0] << 16) | (bs->data[0] >> 16)
+ nop
+ srl $v0, $t0, 16
+ sll $t0, 16
+
+ lw $t1, 12($a3) # next_window = (bs->data[1] << 16) | (bs->data[1] >> 16)
+ or $t0, $v0
+ srl $v0, $t1, 16
+ sll $t1, 16
+
+ lhu $t2, 0($a3) # remaining = bs->uncomp_length * 2
+ or $t1, $v0
+
+ lhu $t3, 4($a3) # quant_scale = (bs->quant_scale & 63) << 10
+ sll $t2, 1
+ andi $t3, 63
+
+ lhu $t4, 6($a3) # is_v3 = !(bs->version < 3)
+ sll $t3, 10
+ sltiu $t4, $t4, 3
+ xori $t4, 1
+
+ li $t5, 32 # bit_offset = 32
+ li $t6, 5 # block_index = 5
+ li $t7, 0 # coeff_index = 0
+ j _vlc_skip_context_load
+ addiu $a3, 16 # input = &(bs->data[2])
+
+.section .text.DecDCTvlcContinue
+.global DecDCTvlcContinue
+.type DecDCTvlcContinue, @function
+DecDCTvlcContinue:
+ lw $a3, VLC_Context_input($a0)
+ lw $t0, VLC_Context_window($a0)
+ lw $t1, VLC_Context_next_window($a0)
+ lw $t2, VLC_Context_remaining($a0)
+ lhu $t3, VLC_Context_quant_scale($a0)
+ lb $t4, VLC_Context_is_v3($a0)
+ lb $t5, VLC_Context_bit_offset($a0)
+ lb $t6, VLC_Context_block_index($a0)
+ lb $t7, VLC_Context_coeff_index($a0)
+
+_vlc_skip_context_load:
+ # Determine how many bytes to output. This whole block of code basically
+ # does this:
+ # max_size = min((max_size - 1) * 2, remaining)
+ # remaining -= max_size
+ bgtz $a2, .Lmax_size_valid # if (max_size <= 0) max_size = 0x7ffe0000
+ addiu $a2, -1 # else max_size = (max_size - 1) * 2
+ lui $a2, 0x3fff
+.Lmax_size_valid:
+ sll $a2, 1
+
+ blt $a2, $t2, .Lmax_size_ok # if (max_size > remaining) max_size = remaining
+ lui $v1, 0x3800
+ move $a2, $t2
+.Lmax_size_ok:
+ subu $t2, $a2 # remaining -= max_size
+
+ # Write the length of the data that will be decoded to first 4 bytes of the
+ # output buffer, which will be then parsed by DecDCTin().
+ srl $v0, $a2, 1 # output[0] = 0x38000000 | (max_size / 2)
+ or $v0, $v1
+ sw $v0, 0($a1)
+
+ # Obtain the addresses of the lookup table and jump area in advance so that
+ # they don't have to be retrieved for each coefficient decoded.
+ lw $t8, _vlc_huffman_table
+ la $t9, .Lac_jump_area
+
+ beqz $a2, .Lstop_processing
+ addiu $a1, 4 # output = (uint16_t *) &output[1]
+
+.Lprocess_next_code_loop: # while (max_size)
+ # This is the "hot" part of the decoder, executed for each code in the
+ # bitstream. The first step is to determine if the next code is a DC or AC
+ # coefficient. The GTE is also given the task of counting the number of
+ # leading zeroes/ones, which takes 2 more cycles.
+ bnez $t7, .Lprocess_ac_coefficient
+ mtc2 $t0, $30
+ bnez $t4, .Lprocess_dc_v3_coefficient
+ #nop
+
+.Lprocess_dc_v2_coefficient: # if (!coeff_index && !is_v3)
+ # The DC coefficient in version 2 frames is not compressed.
+ srl $v0, $t0, 22 # *output = (window >> (32 - 10)) | quant_scale
+ or $v0, $t3
+ addiu $t7, 1 # coeff_index++
+ sll $t0, 10 # window <<= 10
+ addiu $t5, -10 # bit_offset -= 10
+ b .Lwrite_value
+ addiu $t7, 1 # coeff_index++
+
+.Lprocess_dc_v3_coefficient: # if (!coeff_index && is_v3)
+ # TODO: version 3 is currently not supported.
+ jr $ra
+ li $v0, -1
+ #b .Lwrite_value
+
+.Lprocess_ac_coefficient: # if (coeff_index)
+ # Check whether the prefix code is one of the shorter, more common ones.
+ srl $v0, $t0, 30
+ li $v1, 3
+ beq $v0, $v1, .Lac_prefix_11
+ li $v1, 2
+ beq $v0, $v1, .Lac_prefix_10
+ li $v1, 1
+ beq $v0, $v1, .Lac_prefix_01
+ #srl $v0, $t0, 29
+ #beq $v0, $v1, .Lac_prefix_001
+ #nop
+
+ # If the code is longer, retrieve the number of leading zeroes from the GTE
+ # and use it as an index into the jump area. Each block in the area is 8
+ # instructions long and handles decoding a specific prefix.
+ mfc2 $v0, $31
+ nop
+ andi $v0, 15 # jump_addr = &ac_jump_area[(prefix % 16) * 8 * sizeof(u32)]
+ sll $v0, 5
+ addu $v0, $t9
+ jr $v0
+ nop
+
+.Lac_prefix_11:
+ # Prefix 11 is followed by a single bit.
+ srl $v0, $t0, 28 # index = ((window >> (32 - 2 - 1)) & 1) * sizeof(u16)
+ andi $v0, 2
+ addu $v0, $t8 # value = table->lut0[index]
+ lhu $v0, DECDCTSMALLTAB_lut0($v0)
+ sll $t0, 3 # window <<= 3
+ addiu $t5, -3 # bit_offset -= 3
+ b .Lwrite_value
+ addiu $t7, 1 # coeff_index++
+
+.Lac_jump_area:
+.Lac_prefix_10:
+ # Prefix 10 marks the end of a block.
+ li $v0, 0xfe00 # value = 0xfe00
+ sll $t0, 2 # window <<= 2
+ addiu $t5, -2 # bit_offset -= 2
+ addiu $t6, -1 # block_index--
+ bgez $t6, .Lwrite_value
+ li $t7, 0 # coeff_index = 0
+ b .Lwrite_value
+ li $t6, 5 # if (block_index < 0) block_index = 5
+
+.Lac_prefix_01:
+ # Prefix 01 can be followed by a 2-bit lookup index starting with 1, or a
+ # 3-bit lookup index starting with 0. A 32-bit lookup table is used,
+ # containing both MDEC codes and lengths.
+ srl $v0, $t0, 25 # index = ((window >> (32 - 2 - 3)) & 7) * sizeof(u32)
+ andi $v0, 28
+ addu $v0, $t8 # value = table->lut2[index]
+ lw $v0, DECDCTSMALLTAB_lut2($v0)
+ addiu $t7, 1 # coeff_index++
+ b .Lupdate_window_and_write
+ srl $v1, $v0, 16 # length = value >> 16
+ .word 0
+
+.Lac_prefix_001:
+ # Prefix 001 can be followed by a 6-bit lookup index starting with 00, or a
+ # 3-bit lookup index starting with 01/10/11.
+ srl $v0, $t0, 21 # index = ((window >> (32 - 3 - 6)) & 63) * sizeof(u32)
+ andi $v0, 252
+ addu $v0, $t8 # value = table->lut3[index]
+ lw $v0, DECDCTSMALLTAB_lut3($v0)
+ addiu $t7, 1 # coeff_index++
+ b .Lupdate_window_and_write
+ srl $v1, $v0, 16 # length = value >> 16
+ .word 0
+
+.Lac_prefix_0001:
+ # Prefix 0001 is followed by a 3-bit lookup index.
+ srl $v0, $t0, 24 # index = ((window >> (32 - 4 - 3)) & 7) * sizeof(u16)
+ andi $v0, 14
+ addu $v0, $t8 # value = table->lut4[index]
+ lhu $v0, DECDCTSMALLTAB_lut4($v0)
+ sll $t0, 7 # window <<= 4 + 3
+ addiu $t5, -7 # bit_offset -= 4 + 3
+ b .Lwrite_value
+ addiu $t7, 1 # coeff_index++
+
+.Lac_prefix_00001:
+ # Prefix 00001 is followed by a 3-bit lookup index.
+ srl $v0, $t0, 23 # index = ((window >> (32 - 5 - 3)) & 7) * sizeof(u16)
+ andi $v0, 14
+ addu $v0, $t8 # value = table->lut5[index]
+ lhu $v0, DECDCTSMALLTAB_lut5($v0)
+ sll $t0, 8 # window <<= 5 + 3
+ addiu $t5, -8 # bit_offset -= 5 + 3
+ b .Lwrite_value
+ addiu $t7, 1 # coeff_index++
+
+.Lac_prefix_000001:
+ # Prefix 000001 is an escape code followed by a full 16-bit MDEC value.
+ srl $v0, $t0, 10 # value = window >> (32 - 6 - 16)
+ sll $t0, 22 # window <<= 6 + 16
+ addiu $t5, -22 # bit_offset -= 6 + 16
+ b .Lwrite_value
+ addiu $t7, 1 # coeff_index++
+ .word 0, 0, 0
+
+.Lac_prefix_0000001:
+ # Prefix 0000001 is followed by a 4-bit lookup index.
+ srl $v0, $t0, 20 # index = ((window >> (32 - 7 - 4)) & 15) * sizeof(u16)
+ andi $v0, 30
+ addu $v0, $t8 # value = table->lut7[index]
+ lhu $v0, DECDCTSMALLTAB_lut7($v0)
+ sll $t0, 11 # window <<= 7 + 4
+ addiu $t5, -11 # bit_offset -= 7 + 4
+ b .Lwrite_value
+ addiu $t7, 1 # coeff_index++
+
+.Lac_prefix_00000001:
+ # Prefix 00000001 is followed by a 5-bit lookup index.
+ srl $v0, $t0, 18 # index = ((window >> (32 - 8 - 5)) & 31) * sizeof(u16)
+ andi $v0, 62
+ addu $v0, $t8 # value = table->lut8[index]
+ lhu $v0, DECDCTSMALLTAB_lut8($v0)
+ sll $t0, 13 # window <<= 8 + 5
+ addiu $t5, -13 # bit_offset -= 8 + 5
+ b .Lwrite_value
+ addiu $t7, 1 # coeff_index++
+
+.Lac_prefix_000000001:
+ # Prefix 000000001 is followed by a 5-bit lookup index.
+ srl $v0, $t0, 17 # index = ((window >> (32 - 9 - 5)) & 31) * sizeof(u16)
+ andi $v0, 62
+ addu $v0, $t8 # value = table->lut9[index]
+ lhu $v0, DECDCTSMALLTAB_lut9($v0)
+ sll $t0, 14 # window <<= 9 + 5
+ addiu $t5, -14 # bit_offset -= 9 + 5
+ b .Lwrite_value
+ addiu $t7, 1 # coeff_index++
+
+.Lac_prefix_0000000001:
+ # Prefix 0000000001 is followed by a 5-bit lookup index.
+ srl $v0, $t0, 16 # index = ((window >> (32 - 10 - 5)) & 31) * sizeof(u16)
+ andi $v0, 62
+ addu $v0, $t8 # value = table->lut10[index]
+ lhu $v0, DECDCTSMALLTAB_lut10($v0)
+ sll $t0, 15 # window <<= 10 + 5
+ addiu $t5, -15 # bit_offset -= 10 + 5
+ b .Lwrite_value
+ addiu $t7, 1 # coeff_index++
+
+.Lac_prefix_00000000001:
+ # Prefix 00000000001 is followed by a 5-bit lookup index.
+ srl $v0, $t0, 15 # index = ((window >> (32 - 11 - 5)) & 31) * sizeof(u16)
+ andi $v0, 62
+ addu $v0, $t8 # value = table->lut11[index]
+ lhu $v0, DECDCTSMALLTAB_lut11($v0)
+ sll $t0, 16 # window <<= 11 + 5
+ addiu $t5, -16 # bit_offset -= 11 + 5
+ b .Lwrite_value
+ addiu $t7, 1 # coeff_index++
+
+.Lac_prefix_000000000001:
+ # Prefix 000000000001 is followed by a 5-bit lookup index.
+ srl $v0, $t0, 14 # index = ((window >> (32 - 12 - 5)) & 31) * sizeof(u16)
+ andi $v0, 62
+ addu $v0, $t8 # value = table->lut12[index]
+ lhu $v0, DECDCTSMALLTAB_lut12($v0)
+ sll $t0, 17 # window <<= 12 + 5
+ addiu $t5, -17 # bit_offset -= 12 + 5
+ b .Lwrite_value
+ addiu $t7, 1 # coeff_index++
+
+ # Prefix 0000000000001 is not valid.
+ beqz $t0, .Lstop_processing
+ nop
+ jr $ra
+ li $v0, -1
+ .word 0, 0, 0, 0
+
+ # Prefix 00000000000001 is not valid.
+ beqz $t0, .Lstop_processing
+ nop
+ jr $ra
+ li $v0, -1
+ .word 0, 0, 0, 0
+
+ # Prefix 000000000000001 is not valid.
+ beqz $t0, .Lstop_processing
+ nop
+ jr $ra
+ li $v0, -1
+ .word 0, 0, 0, 0
+
+ # Prefix 0000000000000001 is not valid.
+ beqz $t0, .Lstop_processing
+ nop
+ jr $ra
+ li $v0, -1
+ #.word 0, 0, 0, 0
+
+.Lupdate_window_and_write:
+ sllv $t0, $t0, $v1 # window <<= length
+ subu $t5, $v1 # bit_offset -= length
+.Lwrite_value:
+ sh $v0, 0($a1)
+.Lfeed_bitstream:
+ # Update the window. This makes sure the next iteration of the loop will be
+ # able to read up to 32 bits from the bitstream.
+ bgez $t5, .Lskip_feeding # if (bit_offset < 0)
+ addiu $a2, -1 # max_size--
+
+ subu $v0, $0, $t5 # window = next_window << (-bit_offset)
+ sllv $t0, $t1, $v0
+ lw $t1, 0($a3) # next_window = (*input << 16) | (*input >> 16)
+ addiu $t5, 32 # bit_offset += 32
+ srl $v0, $t1, 16
+ sll $t1, 16
+ or $t1, $v0
+ addiu $a3, 4 # input++
+
+.Lskip_feeding:
+ srlv $v0, $t1, $t5 # window |= next_window >> bit_offset
+ or $t0, $v0
+
+ bnez $a2, .Lprocess_next_code_loop
+ addiu $a1, 2 # output++
+
+.Lstop_processing:
+ # If remaining = 0, skip flushing the context, pad the output buffer with
+ # end-of-block codes if necessary and return 0. Otherwise flush the context
+ # and return 1.
+ beqz $t2, .Lpad_output_buffer
+ nop
+
+ sw $a3, VLC_Context_input($a0)
+ sw $t0, VLC_Context_window($a0)
+ sw $t1, VLC_Context_next_window($a0)
+ sw $t2, VLC_Context_remaining($a0)
+ sh $t3, VLC_Context_quant_scale($a0)
+ sb $t4, VLC_Context_is_v3($a0)
+ sb $t5, VLC_Context_bit_offset($a0)
+ sb $t6, VLC_Context_block_index($a0)
+ sb $t7, VLC_Context_coeff_index($a0)
+
+ jr $ra
+ li $v0, 1
+
+.Lpad_output_buffer:
+ beqz $a2, .Lreturn_zero
+ li $v0, 0xfe00
+.Lpad_output_buffer_loop: # while (max_size)
+ sh $v0, 0($a1) # *output = 0xfe00
+ addiu $a2, -1 # max_size--
+ bnez $a2, .Lpad_output_buffer_loop
+ addiu $a1, 2 # output++
+
+.Lreturn_zero:
+ jr $ra
+ li $v0, 0