summaryrefslogtreecommitdiff
path: root/tools/extract_yars.py
blob: 7f6b3363a5499a51eedc67566c179641031d660f (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
#!/usr/bin/env python3

# SPDX-FileCopyrightText: © 2023 ZeldaRET
# SPDX-License-Identifier: MIT

# yar (Yaz0 ARchive) decompressor
#
# This program decompresses every raw yar binary file listed in
# `tools/filelists/{version}/archives.csv` to a corresponding
# `extracted/{version}/baserom/{file}.unarchive` raw file.
#
# It works by decompressing every Yaz0 block and appending them one by one into
# a new raw binary file so it can be processed normally by other tools.


from __future__ import annotations

import argparse
import crunch64
import dataclasses
from pathlib import Path
import struct

from version import version_config

PRINT_XML = False


@dataclasses.dataclass
class ArchiveMeta:
    start: int
    end: int


def readFileAsBytearray(filepath: Path) -> bytearray:
    with filepath.open(mode="rb") as f:
        return bytearray(f.read())


def readBytesAsWord(array_of_bytes: bytearray, offset: int) -> int:
    return struct.unpack_from(f">I", array_of_bytes, offset)[0]


def getOffsetsList(archiveBytes: bytearray) -> list[ArchiveMeta]:
    archivesOffsets: list[ArchiveMeta] = []

    firstEntryOffset = readBytesAsWord(archiveBytes, 0)
    firstEntrySize = readBytesAsWord(archiveBytes, 4)

    archivesOffsets.append(
        ArchiveMeta(firstEntryOffset, firstEntryOffset + firstEntrySize)
    )

    offset = 4
    while offset < firstEntryOffset - 4:
        entry = readBytesAsWord(archiveBytes, offset)
        nextEntry = readBytesAsWord(archiveBytes, offset + 4)
        entryStart = entry + firstEntryOffset
        entryEnd = nextEntry + firstEntryOffset
        archivesOffsets.append(ArchiveMeta(entryStart, entryEnd))

        offset += 4

    return archivesOffsets


def extractArchive(archivePath: Path, outPath: Path):
    archiveBytes = readFileAsBytearray(archivePath)

    if readBytesAsWord(archiveBytes, 0) == 0:
        # Empty file, ignore it
        return

    print(f"Extracting '{archivePath}' -> '{outPath}'")
    archivesOffsets = getOffsetsList(archiveBytes)

    if PRINT_XML:
        print("<Root>")
        print(f'    <File Name="{outPath.stem}">')

    with outPath.open("wb") as out:
        currentOffset = 0
        for meta in archivesOffsets:
            decompressedBytes = crunch64.yaz0.decompress(
                archiveBytes[meta.start : meta.end]
            )
            decompressedSize = len(decompressedBytes)
            out.write(decompressedBytes)

            if PRINT_XML:
                print(
                    f'        <Blob Name="{archivePath.stem}_Blob_{currentOffset:06X}" Size="0x{decompressedSize:04X}" Offset="0x{currentOffset:X}" />'
                )

            currentOffset += decompressedSize

    if PRINT_XML:
        print(f"    </File>")
        print("</Root>")


def main():
    parser = argparse.ArgumentParser(description="MM archives extractor")
    parser.add_argument(
        "baserom_segments_dir",
        type=Path,
        help="Directory of uncompressed ROM segments",
    )
    parser.add_argument(
        "-v",
        "--version",
        help="version to process",
        default="n64-us",
    )
    parser.add_argument("--xml", help="Generate xml to stdout", action="store_true")

    args = parser.parse_args()

    baseromSegmentsDir: Path = args.baserom_segments_dir
    version: str = args.version

    global PRINT_XML
    PRINT_XML = args.xml

    config = version_config.load_version_config(version)

    for archiveName in config.archives:
        archivePath = baseromSegmentsDir / archiveName

        extractedPath = Path(str(archivePath) + ".unarchive")
        extractArchive(archivePath, extractedPath)


if __name__ == "__main__":
    main()