From 2bcea72f4982efbcacf5af162b1a8af4f6c27ca3 Mon Sep 17 00:00:00 2001 From: octorock <79596758+octorock@users.noreply.github.com> Date: Sat, 6 Mar 2021 18:32:11 +0100 Subject: Clean up script files --- tools/script_disassembler/definitions.py | 103 ++++++++++---------- tools/script_disassembler/incbin_parser.py | 115 ---------------------- tools/script_disassembler/script_disassembler.py | 118 ++++++++++------------- tools/script_disassembler/split_script_data.py | 118 +++++++++++++++++++++++ 4 files changed, 222 insertions(+), 232 deletions(-) delete mode 100644 tools/script_disassembler/incbin_parser.py create mode 100644 tools/script_disassembler/split_script_data.py (limited to 'tools/script_disassembler') diff --git a/tools/script_disassembler/definitions.py b/tools/script_disassembler/definitions.py index f12ce493..5c6ce7d5 100644 --- a/tools/script_disassembler/definitions.py +++ b/tools/script_disassembler/definitions.py @@ -1,19 +1,20 @@ from utils import barray_to_u16_hex, barray_to_u32_hex, barray_to_s16 import struct +# A list of all the commands, their correspondingScriptCommand_ functions and what kind of parameters they take commands = [ {'fun': 'ScriptCommandNop', 'params': ''}, - {'fun': 'ScriptCommand_StartScript', 'params': '', 'name': 'start executing scripts'}, - {'fun': 'ScriptCommand_StopScript', 'params': '', 'name': 'stop executing scripts'}, - {'fun': 'ScriptCommand_Jump', 'params': 'j', 'name': 'jump by offset'}, - {'fun': 'ScriptCommand_JumpIf', 'params': 'j', 'name': 'jump if'}, - {'fun': 'ScriptCommand_JumpIfNot', 'params': 'j', 'name': 'jump if not'}, + {'fun': 'ScriptCommand_StartScript', 'params': ''}, + {'fun': 'ScriptCommand_StopScript', 'params': ''}, + {'fun': 'ScriptCommand_Jump', 'params': 'j'}, + {'fun': 'ScriptCommand_JumpIf', 'params': 'j'}, + {'fun': 'ScriptCommand_JumpIfNot', 'params': 'j'}, {'fun': 'ScriptCommand_JumpSwitch', 'params': ['jj', 'jjj', 'jjjj', 'jjjjjjj', 'jjjjjjjjj']}, - {'fun': 'ScriptCommand_JumpAbsolute', 'params': 'x','name': 'abs jump' }, - {'fun': 'ScriptCommand_JumpAbsoluteIf', 'params': 'x', 'name': 'abs jump if'}, - {'fun': 'ScriptCommand_JumpAbsoluteIfNot', 'params': 'x', 'name': 'abs jump if not'}, + {'fun': 'ScriptCommand_JumpAbsolute', 'params': 'x'}, + {'fun': 'ScriptCommand_JumpAbsoluteIf', 'params': 'x'}, + {'fun': 'ScriptCommand_JumpAbsoluteIfNot', 'params': 'x'}, {'fun': 'ScriptCommand_JumpAbsoluteSwitch', 'params': 'xx'}, - {'fun': 'ScriptCommand_Call', 'params':'p', 'name': 'Execute function via pointer'},# 'exec': ScriptCommand_Call}, + {'fun': 'ScriptCommand_Call', 'params': 'p'}, {'fun': 'ScriptCommand_CallWithArg', 'params': ['pw', 'p']}, {'fun': 'ScriptCommand_LoadRoomEntityList', 'params': 'd'}, {'fun': 'ScriptCommand_TestBit', 'params': 'w'}, @@ -37,12 +38,12 @@ commands = [ {'fun': 'ScriptCommand_0807E4CC', 'params': 'w'}, {'fun': 'ScriptCommand_0807E4EC', 'params': 'w'}, {'fun': 'ScriptCommand_0807E514', 'params': 'w'}, - {'fun': 'ScriptCommand_CheckPlayerFlags', 'params':'w'}, + {'fun': 'ScriptCommand_CheckPlayerFlags', 'params': 'w'}, {'fun': 'ScriptCommand_0807E564', 'params': ''}, {'fun': 'ScriptCommand_EntityHasHeight', 'params': ''}, {'fun': 'ScriptCommand_ComparePlayerAction', 'params': 's'}, {'fun': 'ScriptCommand_ComparePlayerAnimationState', 'params': 's'}, - {'fun': 'ScriptCommand_0807E5F8', 'params': 'w'},# 'exec': ScriptCommand_0807E5F8}, + {'fun': 'ScriptCommand_0807E5F8', 'params': 'w'}, {'fun': 'ScriptCommand_0807E610', 'params': 'w'}, {'fun': 'ScriptCommand_SetLocalFlag', 'params': 's'}, {'fun': 'ScriptCommand_SetLocalFlagByOffset', 'params': 'ss'}, @@ -75,10 +76,10 @@ commands = [ {'fun': 'ScriptCommand_SetPlayerAction', 'params': 'w'}, {'fun': 'ScriptCommand_StartPlayerScript', 'params': 'x'}, {'fun': 'ScriptCommand_0807E8D4', 'params': 's'}, - {'fun': 'ScriptCommand_0807E8E4_0', 'params': ''}, # duplicate - {'fun': 'ScriptCommand_0807E8E4_1', 'params': ''}, # duplicate - {'fun': 'ScriptCommand_0807E8E4_2', 'params': ''}, # duplicate - {'fun': 'ScriptCommand_0807E8E4_3', 'params': ''}, # duplicate + {'fun': 'ScriptCommand_0807E8E4_0', 'params': ''}, # duplicate + {'fun': 'ScriptCommand_0807E8E4_1', 'params': ''}, # duplicate + {'fun': 'ScriptCommand_0807E8E4_2', 'params': ''}, # duplicate + {'fun': 'ScriptCommand_0807E8E4_3', 'params': ''}, # duplicate {'fun': 'ScriptCommand_0807E908', 'params': 's'}, {'fun': 'ScriptCommand_0807E914', 'params': 'w'}, {'fun': 'ScriptCommand_0807E924', 'params': ''}, @@ -94,7 +95,7 @@ commands = [ {'fun': 'ScriptCommand_0807EA94', 'params': ''}, {'fun': 'ScriptCommand_TextboxNoOverlapFollow', 'params': 's'}, {'fun': 'ScriptCommand_TextboxNoOverlap', 'params': 's'}, - {'fun': 'ScriptCommand_TextboxNoOverlapFollowPos', 'params': ['w', 's']}, # TODO w or ss? + {'fun': 'ScriptCommand_TextboxNoOverlapFollowPos', 'params': ['ss', 's']}, {'fun': 'ScriptCommand_0807EAF0', 'params': ['ss', 'sss', 'ssss']}, {'fun': 'ScriptCommand_TextboxNoOverlapVar', 'params': ''}, {'fun': 'ScriptCommand_0807EB28', 'params': 's'}, @@ -143,9 +144,6 @@ commands = [ {'fun': 'ScriptCommand_0807F0C8', 'params': 'ss'} ] - - - # Functions that have already been renamed POINTER_MAP = { 'sub_08095458': 'nullsub_527', @@ -157,18 +155,22 @@ POINTER_MAP = { 'sub_080A2138': 'Windcrest_Unlock', 'sub_080A29BC': 'CreateDust' } -# tries to directly reference the function this is pointing to + + def get_pointer(barray): + # tries to directly reference the function this is pointing to integers = struct.unpack('I', barray) pointer = 'sub_' + (struct.pack('>I', integers[0]-1).hex()).upper() if pointer in POINTER_MAP: return POINTER_MAP[pointer] return pointer + # Data pointers that actually point to a script location DATA_MAP = { 'gUnk_08016384': 'script_08016384' } + def get_data_pointer(barray): integers = struct.unpack('I', barray) pointer = 'gUnk_' + (struct.pack('>I', integers[0]).hex()).upper() @@ -176,6 +178,7 @@ def get_data_pointer(barray): return DATA_MAP[pointer] return pointer + def get_script_pointer(barray): integers = struct.unpack('I', barray) return 'script_' + (struct.pack('>I', integers[0]).hex()).upper() @@ -184,7 +187,7 @@ def get_script_pointer(barray): def get_script_label(u32): return hex(u32).upper().replace('0X', 'script_0') - +# Collects a set of all the labels that were jumped to used_labels = set() def use_script_label(u32): global used_labels @@ -193,125 +196,121 @@ def use_script_label(u32): return label - - # definitions for parameter types parameters = { '': { - 'length':0, + 'length': 0, 'param': '', 'expr': '', 'read': lambda ctx: '' }, - 's': { + 's': { 'length': 1, 'param': 's', 'expr': ' .short \s', - 'read': lambda ctx: barray_to_u16_hex(ctx.data[ctx.ptr+2:ctx.ptr+4])[0] + 'read': lambda ctx: barray_to_u16_hex(ctx.data[ctx.ptr + 2:ctx.ptr + 4])[0] }, 'ss': { 'length': 2, 'param': 'a,b', 'expr': ' .short \\a\n .short \\b', - 'read': lambda ctx: ', '.join(barray_to_u16_hex(ctx.data[ctx.ptr+2:ctx.ptr+6])) + 'read': lambda ctx: ', '.join(barray_to_u16_hex(ctx.data[ctx.ptr + 2:ctx.ptr + 6])) }, 'sss': { 'length': 3, 'param': 'a,b,c', 'expr': ' .short \\a\n .short \\b\n .short \\c', - 'read': lambda ctx: ', '.join(barray_to_u16_hex(ctx.data[ctx.ptr+2:ctx.ptr+8])) + 'read': lambda ctx: ', '.join(barray_to_u16_hex(ctx.data[ctx.ptr + 2:ctx.ptr + 8])) }, 'ssss': { 'length': 4, 'param': 'a,b,c,d', 'expr': ' .short \\a\n .short \\b\n .short \\c\n .short \\d', - 'read': lambda ctx: ', '.join(barray_to_u16_hex(ctx.data[ctx.ptr+2:ctx.ptr+10])) + 'read': lambda ctx: ', '.join(barray_to_u16_hex(ctx.data[ctx.ptr + 2:ctx.ptr + 10])) }, 'w': { 'length': 2, 'param': 'w', 'expr': ' .word \w', - 'read': lambda ctx: barray_to_u32_hex(ctx.data[ctx.ptr+2:ctx.ptr+6])[0] + 'read': lambda ctx: barray_to_u32_hex(ctx.data[ctx.ptr + 2:ctx.ptr + 6])[0] }, 'ww': { 'length': 4, 'param': 'a,b', 'expr': ' .word \\a\n .word \\b', - 'read': lambda ctx: ', '.join(barray_to_u32_hex(ctx.data[ctx.ptr+2:ctx.ptr+10])) + 'read': lambda ctx: ', '.join(barray_to_u32_hex(ctx.data[ctx.ptr + 2:ctx.ptr + 10])) }, - - 'j': { # Relative jump target + 'j': { # Relative jump target 'length': 1, 'param': 's', 'expr': '1: .short \s - 1b', - 'read': lambda ctx: use_script_label(ctx.script_addr+ctx.ptr+ 2 +barray_to_s16(ctx.data[ctx.ptr+2:ctx.ptr+4])) - # hex(ctx.script_addr + barray_to_s16(ctx.data[ctx.ptr+2:ctx.ptr+4])).upper().replace('0X', 'script_0') + 'read': lambda ctx: use_script_label(ctx.script_addr + ctx.ptr + 2 + barray_to_s16(ctx.data[ctx.ptr + 2:ctx.ptr + 4])) }, 'jj': { 'length': 2, 'param': 'a,b', 'expr': '1: .short \\a - 1b\n .short \\b - 1b - 2', - 'read': lambda ctx: ', '.join([use_script_label(ctx.script_addr+ctx.ptr+ 2 +barray_to_s16(ctx.data[ctx.ptr+x*2+2:ctx.ptr+x*2+4]) + x*2) for x in range(0,2)]) - }, + 'read': lambda ctx: ', '.join([use_script_label(ctx.script_addr + ctx.ptr + 2 + barray_to_s16(ctx.data[ctx.ptr + x * 2 + 2:ctx.ptr + x * 2 + 4]) + x * 2) for x in range(0, 2)]) + }, 'jjj': { 'length': 3, 'param': 'a,b,c', 'expr': '1: .short \\a - 1b\n .short \\b - 1b - 2\n .short \\c - 1b - 4', - 'read': lambda ctx: ', '.join([use_script_label(ctx.script_addr+ctx.ptr+ 2 +barray_to_s16(ctx.data[ctx.ptr+x*2+2:ctx.ptr+x*2+4]) + x*2) for x in range(0,3)]) + 'read': lambda ctx: ', '.join([use_script_label(ctx.script_addr + ctx.ptr + 2 + barray_to_s16(ctx.data[ctx.ptr + x * 2 + 2:ctx.ptr + x * 2 + 4]) + x*2) for x in range(0, 3)]) }, 'jjjj': { 'length': 4, 'param': 'a,b,c,d', 'expr': '1: .short \\a - 1b\n .short \\b - 1b - 2\n .short \\c - 1b - 4\n .short \\d - 1b - 6', - 'read': lambda ctx: ', '.join([use_script_label(ctx.script_addr+ctx.ptr+ 2 +barray_to_s16(ctx.data[ctx.ptr+x*2+2:ctx.ptr+x*2+4]) + x*2) for x in range(0,4)]) + 'read': lambda ctx: ', '.join([use_script_label(ctx.script_addr + ctx.ptr + 2 + barray_to_s16(ctx.data[ctx.ptr + x * 2 + 2:ctx.ptr + x * 2 + 4]) + x*2) for x in range(0, 4)]) }, 'jjjjjjj': { 'length': 7, 'param': 'a,b,c,d,e,f,g', 'expr': '1: .short \\a - 1b\n .short \\b - 1b - 2\n .short \\c - 1b - 4\n .short \\d - 1b - 6\n .short \\e - 1b - 8\n .short \\f - 1b - 10\n .short \\g - 1b - 12', - 'read': lambda ctx: ', '.join([use_script_label(ctx.script_addr+ctx.ptr+ 2 +barray_to_s16(ctx.data[ctx.ptr+x*2+2:ctx.ptr+x*2+4]) + x*2) for x in range(0,7)]) - }, + 'read': lambda ctx: ', '.join([use_script_label(ctx.script_addr + ctx.ptr + 2 + barray_to_s16(ctx.data[ctx.ptr + x * 2 + 2:ctx.ptr + x * 2 + 4]) + x*2) for x in range(0, 7)]) + }, 'jjjjjjjjj': { 'length': 9, 'param': 'a,b,c,d,e,f,g,h,i', 'expr': '1: .short \\a - 1b\n .short \\b - 1b - 2\n .short \\c - 1b - 4\n .short \\d - 1b - 6\n .short \\e - 1b - 8\n .short \\f - 1b - 10\n .short \\g - 1b - 12\n .short \\h - 1b - 14\n .short \\i - 1b - 16', - 'read': lambda ctx: ', '.join([use_script_label(ctx.script_addr+ctx.ptr+ 2 +barray_to_s16(ctx.data[ctx.ptr+x*2+2:ctx.ptr+x*2+4]) + x*2) for x in range(0,9)]) - }, + 'read': lambda ctx: ', '.join([use_script_label(ctx.script_addr + ctx.ptr + 2 + barray_to_s16(ctx.data[ctx.ptr + x * 2 + 2:ctx.ptr + x * 2 + 4]) + x*2) for x in range(0, 9)]) + }, 'p': { 'length': 2, 'param': 'w', 'expr': ' .word \w', - 'read': lambda ctx: get_pointer(ctx.data[ctx.ptr+2:ctx.ptr+6]) + 'read': lambda ctx: get_pointer(ctx.data[ctx.ptr + 2:ctx.ptr + 6]) }, 'pw': { 'length': 4, 'param': 'a,b', 'expr': ' .word \\a\n .word \\b', - 'read': lambda ctx: get_pointer(ctx.data[ctx.ptr+2:ctx.ptr+6]) + ', ' + barray_to_u32_hex(ctx.data[ctx.ptr+6:ctx.ptr+14])[0] + 'read': lambda ctx: get_pointer(ctx.data[ctx.ptr + 2:ctx.ptr + 6]) + ', ' + barray_to_u32_hex(ctx.data[ctx.ptr + 6:ctx.ptr + 14])[0] }, - 'd': { # Data pointer + 'd': { # Data pointer 'length': 2, 'param': 'w', 'expr': ' .word \w', - 'read': lambda ctx: get_data_pointer(ctx.data[ctx.ptr+2:ctx.ptr+6]) + 'read': lambda ctx: get_data_pointer(ctx.data[ctx.ptr + 2:ctx.ptr + 6]) }, - 'x': { # Script pointer + 'x': { # Script pointer 'length': 2, 'param': 'w', 'expr': ' .word \w', - 'read': lambda ctx: get_script_pointer(ctx.data[ctx.ptr+2:ctx.ptr+6]) + 'read': lambda ctx: get_script_pointer(ctx.data[ctx.ptr + 2:ctx.ptr + 6]) }, 'xx': { 'length': 4, 'param': 'a, b', 'expr': ' .word \\a\n .word \\b', - 'read': lambda ctx: get_script_pointer(ctx.data[ctx.ptr+2:ctx.ptr+6]) + ', ' + get_script_pointer(ctx.data[ctx.ptr+6:ctx.ptr+10]) + 'read': lambda ctx: get_script_pointer(ctx.data[ctx.ptr + 2:ctx.ptr + 6]) + ', ' + get_script_pointer(ctx.data[ctx.ptr + 6:ctx.ptr + 10]) }, # Commands with variable parameter count are now handled by explicitely defining all used parameter configurations - # 'v': { + # 'v': { # 'length': -1, # 'param': '', # 'expr': '', @@ -323,4 +322,4 @@ parameters = { # 'expr': ' .word \w', # 'read': lambda ctx: '' # }, -} \ No newline at end of file +} diff --git a/tools/script_disassembler/incbin_parser.py b/tools/script_disassembler/incbin_parser.py deleted file mode 100644 index 70fe3da9..00000000 --- a/tools/script_disassembler/incbin_parser.py +++ /dev/null @@ -1,115 +0,0 @@ -# This python script reads the script.s file which contains all the .incbin macros -# Then it fetches the corresponding data of the baserom, o -TMC_FOLDER='../..' - -import subprocess -import sys -from script_disassembler import disassemble_script, generate_macros - -ROM_OFFSET=0x08000000 -SCRIPTS_START=0x08008B5C -SCRIPTS_END=0x08016984 - -# Create labels for these additional script instructions -# Currently done by splitting the script at that point -LABEL_BREAKS=[0x0800A088, 0x0800ACE0, 0x0800AD54, 0x0800B41C, 0x0800B7C4, 0x0800C8C8, 0x0800D190, 0x800D3EC, 0x0800E9F4, 0x0800FD80, 0x08012AC8, 0x08012F0C, 0x080130E4, 0x08013B70, 0x080142B0, 0x080147DC, 0x08014A80, 0x08014B10,0x0801635C, 0x08016384, 0x080165D8] - -# Whether to output a label for every line -PRINT_ALL_LABELS=False - -def read_baserom(): - # read baserom data - with open(f'{TMC_FOLDER}/baserom.gba', 'rb') as baserom: - return bytearray(baserom.read()) - -def get_label(addr): - return hex(addr).upper().replace('0X', 'script_0') - - -def disassemble_scripts(baserom_data): - script_start = SCRIPTS_START-ROM_OFFSET - - scripts = ''' .include "asm/macros.inc" - .include "constants/constants.inc" - - .include "asm/macros/scripts.inc" - - .syntax unified - - .text - -''' - label_break = 0 - - while script_start < SCRIPTS_END-ROM_OFFSET: - if label_break < len(LABEL_BREAKS) and script_start+ROM_OFFSET >=LABEL_BREAKS[label_break]: - label_break += 1 - - label = get_label(script_start+ROM_OFFSET) - print(f"Disassembling \033[1;34m{label}\033[0m ({script_start} / { SCRIPTS_END-ROM_OFFSET} bytes converted)...") - # find end of the script signified by 0xffff0000 - script_end = baserom_data.index(b'\xff\xff\x00\x00', script_start) + 4 - - if script_end > SCRIPTS_END-ROM_OFFSET: - script_end = SCRIPTS_END-ROM_OFFSET - - if label_break < len(LABEL_BREAKS) and script_end+ROM_OFFSET > LABEL_BREAKS[label_break]: - #print(f'break at {hex(LABEL_BREAKS[label_break])} instead of {hex(script_end)}') - script_end = LABEL_BREAKS[label_break]-ROM_OFFSET - - # read data from rom - data = baserom_data[script_start:script_end] - - - - scripts += f' .include "data/scripts/{label}.inc"\n' - stdout = sys.stdout - - with open(f'{TMC_FOLDER}/data/scripts/{label}.inc','w') as out: - sys.stdout = out - - if script_start == 0x1637C: # This function is actually assembly - print ('''thumb_func_start script_0801637C -script_0801637C: - push {lr} - bl CreateDust - pop {pc}''') - sys.stdout = stdout - script_start = script_end - continue - - print(f'SCRIPT_START {label}') - res = disassemble_script(data, script_start+ROM_OFFSET, PRINT_ALL_LABELS) - if res != 0: - # Script ended in the middle, need to create a new file - script_end = script_start + res - sys.stdout = stdout - - script_start = script_end - return scripts - -def main(): - baserom_data = read_baserom() - - # Do two passes, in the first pass not all labels that are jumped to are known, so those labels are recorded in the first pass - # This is not necessary when all labels are printed - if not PRINT_ALL_LABELS: - print('Collecting labels...') - disassemble_scripts(baserom_data) - print('Writing scripts with labels...') - scripts = disassemble_scripts(baserom_data) - - print('Writing scripts.s file...') - with open(f'{TMC_FOLDER}/data/scripts.s', 'w') as out: - out.write(scripts) - print('Generating asm macros...') - stdout = sys.stdout - with open(f'{TMC_FOLDER}/asm/macros/scripts.inc', 'w') as out: - sys.stdout = out - generate_macros() - sys.stdout = stdout - - print('\033[1;92mDone\033[0m\n') - -if __name__ == '__main__': - main() \ No newline at end of file diff --git a/tools/script_disassembler/script_disassembler.py b/tools/script_disassembler/script_disassembler.py index fe35bcd1..19b171c8 100644 --- a/tools/script_disassembler/script_disassembler.py +++ b/tools/script_disassembler/script_disassembler.py @@ -1,14 +1,14 @@ from dataclasses import dataclass import struct -from utils import barray_to_u16_hex, barray_to_u32_hex, u16_to_hex, u32_to_hex -from definitions import get_pointer, get_data_pointer, get_script_pointer, commands, parameters, get_script_label, used_labels +from utils import barray_to_u16_hex, u16_to_hex +from definitions import get_pointer, commands, parameters, get_script_label, used_labels # Disassembler for tmc scripts # Input 'macros' to generate the macros for the script commands # Input the script bytes as hex to disassemble the script -# Build macros: echo "macros" | python script_disassembler.py > ~/git/tmc/github/asm/macros/scripts.inc +# Build macros: echo "macros" | python script_disassembler.py > ~/git/tmc/github/asm/macros/scripts.inc @dataclass class Context: @@ -18,20 +18,23 @@ class Context: # Remove the ScriptCommand_ prefix for the asm macros -def build_script_command(name: str): +def build_script_command(name: str): name = name.replace("ScriptCommand_", "") - if name[0].isdigit(): # asm macros cannot start with an _ - return '_' + name + if name[0].isdigit(): # asm macros cannot start with an _ + return f'_{name}' return name + def print_rest_bytes(ctx): - print('\n'.join(['.byte ' + hex(x) for x in ctx.data[ctx.ptr:]])) + print('\n'.join(['.byte ' + hex(x) for x in ctx.data[ctx.ptr:]])) + def disassemble_command(ctx: Context, add_all_annotations=False): global used_labels if add_all_annotations or ctx.script_addr + ctx.ptr in used_labels: - print(f'{get_script_label(ctx.script_addr + ctx.ptr)}:') # print offsets to debug when manually inserting labels - cmd = struct.unpack('H', ctx.data[ctx.ptr:ctx.ptr+2])[0] + # print offsets to debug when manually inserting labels + print(f'{get_script_label(ctx.script_addr + ctx.ptr)}:') + cmd = struct.unpack('H', ctx.data[ctx.ptr:ctx.ptr + 2])[0] if cmd == 0: # this does not need to be the end of the script print('\t.short 0x0000') @@ -41,99 +44,88 @@ def disassemble_command(ctx: Context, add_all_annotations=False): if cmd == 0xffff: ctx.ptr += 2 print('SCRIPT_END') - cmd = struct.unpack('H', ctx.data[ctx.ptr:ctx.ptr+2])[0] + cmd = struct.unpack('H', ctx.data[ctx.ptr:ctx.ptr + 2])[0] if cmd == 0x0000: # This is actually the end of the script print('\t.short 0x0000') ctx.ptr += 2 return 2 - return 3 # There is a SCRIPT_END without 0x0000 afterwards, but still split into a new file, please + return 3 # There is a SCRIPT_END without 0x0000 afterwards, but still split into a new file, please commandSize = cmd >> 0xA if commandSize == 0: - raise Exception(f'Zero commandSize') - # TODO error - return 0 + raise Exception(f'Zero commandSize not allowed') commandId = cmd & 0x3FF if commandId >= len(commands): - #print_rest_bytes(ctx) - print(f'\t.short {u16_to_hex(cmd)}') - ctx.ptr += 2 - #raise Exception(f'Invalid commandId {commandId} / {len(commands)} {cmd}') - # TODO error - return 1 + raise Exception(f'Invalid commandId {commandId} / {len(commands)} {cmd}') command = commands[commandId] - param_length = commandSize - 1 + param_length = commandSize - 1 if commandSize > 1: - if ctx.ptr+2*commandSize > len(ctx.data): + if ctx.ptr + 2 * commandSize > len(ctx.data): raise Exception(f'Not enough data to fetch {commandSize-1} params') - return 0 - #meta = struct.unpack( - # 'H'*(unk_06-1), ctx.data[ctx.ptr+2:ctx.ptr+2*unk_06]) - #print('meta', meta) # Handle parameters if not 'params' in command: - raise Exception('Parameters not defined for ' + command['fun'] + ' Should be of length ' + str(param_length)) + raise Exception(f'Parameters not defined for {command["fun"]}. Should be of length {str(param_length)}') params = None - suffix = '' + suffix = '' # When there are multiple variants of parameters, choose the one with the correct count for this if isinstance(command['params'], list): - for i,param in enumerate(command['params']): + for i, param in enumerate(command['params']): if not param in parameters: raise Exception(f'Parameter configuration {param} not defined') candidate = parameters[param] - if candidate['length'] == commandSize -1: + if candidate['length'] == commandSize - 1: params = candidate if i != 0: - suffix = f'_{params["length"]}'# We need to add a suffix to distinguish the correct parameter variant + # We need to add a suffix to distinguish the correct parameter variant + suffix = f'_{params["length"]}' break if params is None: - raise Exception(f'No suitable parameter configuration with length {commandSize-1} found for {command["fun"]}') + raise Exception( + f'No suitable parameter configuration with length {commandSize-1} found for {command["fun"]}') else: if not command['params'] in parameters: - raise Exception('Parameter configuration ' + command['params'] + ' not defined') + raise Exception(f'Parameter configuration {command["params"]} not defined') params = parameters[command['params']] command_name = f'{command["fun"]}{suffix}' - if params['length'] == -1: # variable parameter length + if params['length'] == -1: # variable parameter length print(f'\t.short {u16_to_hex(cmd)} @ {build_script_command(command_name)} with {commandSize-1} parameters') if commandSize > 1: - print('\n'.join(['\t.short ' + x for x in barray_to_u16_hex(ctx.data[ctx.ptr+2:ctx.ptr+commandSize*2])])) + print('\n'.join(['\t.short ' + x for x in barray_to_u16_hex(ctx.data[ctx.ptr + 2:ctx.ptr + commandSize * 2])])) print(f'@ End of parameters') - ctx.ptr += commandSize*2 + ctx.ptr += commandSize * 2 return 1 - elif params['length'] == -2: # point and var + elif params['length'] == -2: # point and var print(f'\t.short {u16_to_hex(cmd)} @ {build_script_command(command_name)} with {commandSize-3} parameters') - print('\t.word '+ get_pointer(ctx.data[ctx.ptr+2:ctx.ptr+6])) + print('\t.word ' + get_pointer(ctx.data[ctx.ptr + 2:ctx.ptr + 6])) if commandSize > 3: - print('\n'.join(['\t.short ' + x for x in barray_to_u16_hex(ctx.data[ctx.ptr+6:ctx.ptr+commandSize*2])])) + print('\n'.join(['\t.short ' + x for x in barray_to_u16_hex(ctx.data[ctx.ptr + 6:ctx.ptr + commandSize * 2])])) print(f'@ End of parameters') - ctx.ptr += commandSize*2 + ctx.ptr += commandSize * 2 return 1 if commandSize-1 != params['length']: - raise Exception(f'Call {command_name} with ' + str(commandSize-1) +' length, while length of ' + str(params['length'])+' defined') + raise Exception(f'Call {command_name} with {commandSize-1} length, while length of {params["length"]} defined') print(f'\t{build_script_command(command_name)} {params["read"](ctx)}') # Execute script - ctx.ptr += commandSize*2 + ctx.ptr += commandSize * 2 return 1 - def disassemble_script(input_bytes, script_addr, add_all_annotations=False): - ctx = Context(0, input_bytes, script_addr) - foundEnd = False while True: - if ctx.ptr >= len(ctx.data) - 1: # End of file (there need to be at least two bytes remaining for the next operation id) + # End of file (there need to be at least two bytes remaining for the next operation id) + if ctx.ptr >= len(ctx.data) - 1: break res = disassemble_command(ctx, add_all_annotations) if res == 0: @@ -145,26 +137,21 @@ def disassemble_script(input_bytes, script_addr, add_all_annotations=False): # End in the middle of the script, please create a new file return ctx.ptr - - - # Print rest (did not manage to get there) if ctx.ptr < len(ctx.data): if (len(ctx.data) - ctx.ptr) % 2 != 0: print_rest_bytes(ctx) - # TODO error - raise Exception('DONT WANT EXTRA after '+ str(ctx.ptr) + ' / ' + str(len(ctx.data))) - return + raise Exception(f'There is extra data at the end {ctx.ptr} / {len(ctx.data)}') print('\n'.join(['.short ' + x for x in barray_to_u16_hex(ctx.data[ctx.ptr:])])) - raise Exception('DONT WANT EXTRA after '+ str(ctx.ptr) + ' / ' + str(len(ctx.data))) + raise Exception(f'There is extra data at the end {ctx.ptr} / {len(ctx.data)}') if not foundEnd: - # Seems to happen, sadly + # Sadly, there are script files without and end? return 0 #print('\033[93mNo end found\033[0m') - #raise Exception('No end found') return 0 + def generate_macros(): print('@ All the macro functions for scripts') print('@ Generated by disassemble_script.py') @@ -182,7 +169,7 @@ def generate_macros(): print('') for num, command in enumerate(commands): if not 'params' in command: - raise Exception('Parameters not defined for ' + command['fun'] + '!') + raise Exception(f'Parameters not defined for {command["fun"]}') def emit_macro(command_name, id, params): print(f'.macro {command_name} {params["param"]}') @@ -194,29 +181,29 @@ def generate_macros(): if isinstance(command['params'], list): # emit macros for all variants - for i,variant in enumerate(command['params']): + for i, variant in enumerate(command['params']): if not variant in parameters: - raise Exception('Parameter configuration ' + variant + ' not defined') + raise Exception(f'Parameter configuration {variant} not defined') params = parameters[variant] - id = ((params['length']+1) << 0xA) + num + id = ((params['length'] + 1) << 0xA) + num suffix = '' if i != 0: suffix = f'_{params["length"]}' emit_macro(f'{build_script_command(command["fun"])}{suffix}', id, params) else: if not command['params'] in parameters: - raise Exception('Parameter configuration ' + command['params'] + ' not defined') + raise Exception(f'Parameter configuration {command["params"]} not defined') params = parameters[command['params']] - id = ((params['length']+1) << 0xA) + num + id = ((params['length'] + 1) << 0xA) + num - if params['length'] < 0: # Don't emit anything for variable parameters + if params['length'] < 0: # Don't emit anything for variable parameters continue emit_macro(build_script_command(command['fun']), id, params) - #print('#define ' + command['fun'] + '(' + params['param'] + ') asm(".short '+u16_to_hex(id)+'");' + params['expr']) print('') + def main(): # Read input @@ -226,6 +213,7 @@ def main(): generate_macros() return disassemble_script(bytearray.fromhex(input_data)) - + + if __name__ == '__main__': - main() \ No newline at end of file + main() diff --git a/tools/script_disassembler/split_script_data.py b/tools/script_disassembler/split_script_data.py new file mode 100644 index 00000000..58ffa5a0 --- /dev/null +++ b/tools/script_disassembler/split_script_data.py @@ -0,0 +1,118 @@ +from script_disassembler import disassemble_script, generate_macros +import sys + +# Reads a section from the baserom, splits the residing scripts into seperate files and disassembles them +# Should only be run before any manual changes to the script files are done! + +TMC_FOLDER = '../..' + +ROM_OFFSET = 0x08000000 +SCRIPTS_START = 0x08008B5C +SCRIPTS_END = 0x08016984 + +# Create labels for these additional script instructions +# Currently done by splitting the script at that point +LABEL_BREAKS = [0x0800A088, 0x0800ACE0, 0x0800AD54, 0x0800B41C, 0x0800B7C4, 0x0800C8C8, 0x0800D190, 0x800D3EC, 0x0800E9F4, 0x0800FD80, + 0x08012AC8, 0x08012F0C, 0x080130E4, 0x08013B70, 0x080142B0, 0x080147DC, 0x08014A80, 0x08014B10, 0x0801635C, 0x08016384, 0x080165D8] + +# Whether to output a label for every line +PRINT_ALL_LABELS = False + + +def read_baserom(): + # read baserom data + with open(f'{TMC_FOLDER}/baserom.gba', 'rb') as baserom: + return bytearray(baserom.read()) + + +def get_label(addr): + return hex(addr).upper().replace('0X', 'script_0') + + +def disassemble_scripts(baserom_data): + script_start = SCRIPTS_START-ROM_OFFSET + + scripts = ''' .include "asm/macros.inc" + .include "constants/constants.inc" + + .include "asm/macros/scripts.inc" + + .syntax unified + + .text + +''' + label_break = 0 + + while script_start < SCRIPTS_END-ROM_OFFSET: + if label_break < len(LABEL_BREAKS) and script_start + ROM_OFFSET >= LABEL_BREAKS[label_break]: + label_break += 1 + + label = get_label(script_start + ROM_OFFSET) + print(f"Disassembling \033[1;34m{label}\033[0m ({script_start} / { SCRIPTS_END-ROM_OFFSET} bytes converted)...") + # find end of the script signified by 0xffff0000 + script_end = baserom_data.index(b'\xff\xff\x00\x00', script_start) + 4 + + if script_end > SCRIPTS_END-ROM_OFFSET: + script_end = SCRIPTS_END-ROM_OFFSET + + # break at a predefined label into a new file + if label_break < len(LABEL_BREAKS) and script_end + ROM_OFFSET > LABEL_BREAKS[label_break]: + script_end = LABEL_BREAKS[label_break]-ROM_OFFSET + + # read data from rom + data = baserom_data[script_start:script_end] + + scripts += f' .include "data/scripts/{label}.inc"\n' + stdout = sys.stdout + + with open(f'{TMC_FOLDER}/data/scripts/{label}.inc', 'w') as out: + sys.stdout = out + + if script_start == 0x1637C: # This function is actually assembly + print('''thumb_func_start script_0801637C +script_0801637C: + push {lr} + bl CreateDust + pop {pc}''') + sys.stdout = stdout + script_start = script_end + continue + + print(f'SCRIPT_START {label}') + res = disassemble_script(data, script_start + ROM_OFFSET, PRINT_ALL_LABELS) + if res != 0: + # Script ended in the middle, need to create a new file + script_end = script_start + res + sys.stdout = stdout + + script_start = script_end + return scripts + + +def main(): + baserom_data = read_baserom() + + # Do two passes, in the first pass not all labels that are jumped to are known, so those labels are recorded in the first pass + # This is not necessary when all labels are printed + if not PRINT_ALL_LABELS: + print('Collecting labels...') + disassemble_scripts(baserom_data) + print('Writing scripts with labels...') + scripts = disassemble_scripts(baserom_data) + + print('Writing scripts.s file...') + with open(f'{TMC_FOLDER}/data/scripts.s', 'w') as out: + out.write(scripts) + print('Generating asm macros...') + stdout = sys.stdout + with open(f'{TMC_FOLDER}/asm/macros/scripts.inc', 'w') as out: + sys.stdout = out + generate_macros() + sys.stdout = stdout + + print('\033[1;92mDone\033[0m\n') + + +if __name__ == '__main__': + main() -- cgit v1.2.3