inital commit
This commit is contained in:
Executable
BIN
Binary file not shown.
Executable
BIN
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Executable
BIN
Binary file not shown.
Executable
+265
@@ -0,0 +1,265 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
# SPDX-FileCopyrightText: Copyright 2025 karas84 (https://github.com/karas84)
|
||||
# SPDX-License-Identifier: MIT
|
||||
#
|
||||
# This script inserts source line debug information (.loc directives) into assembly files
|
||||
# using data extracted from a JSON "stdump" file generated by the ccc tool (version 2.1),
|
||||
# available at https://github.com/chaoticgd/ccc.
|
||||
#
|
||||
# Original concept by Mc-muffin (https://github.com/Mc-muffin).
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import io
|
||||
import sys
|
||||
import json
|
||||
import argparse
|
||||
|
||||
from typing import Callable, Protocol, cast
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
class STDUMPJson:
|
||||
def __init__(self, json_path: Path):
|
||||
self._json = json.loads(json_path.read_text())
|
||||
self._line_cache = self._build_line_cache()
|
||||
self._validate_sub_source_files()
|
||||
|
||||
def _build_line_cache(self):
|
||||
line_cache: dict[int, int] = {}
|
||||
functions = self._json["functions"]
|
||||
|
||||
for fun in functions:
|
||||
if "line_numbers" not in fun:
|
||||
continue
|
||||
|
||||
for addr, num in fun["line_numbers"]:
|
||||
# there may be more lines for the same address. Ghidra seems to
|
||||
# keep the last one, so that's what we are also doing
|
||||
line_cache[addr] = num
|
||||
|
||||
return line_cache
|
||||
|
||||
def _validate_sub_source_files(self):
|
||||
functions = self._json["functions"]
|
||||
for fun in functions:
|
||||
if (line_numbers := fun.get("line_numbers")) is None:
|
||||
continue
|
||||
|
||||
if (sub_source_files := fun.get("sub_source_files")) is None:
|
||||
continue
|
||||
|
||||
line_numbers = cast(list[tuple[int, int]], line_numbers)
|
||||
sub_source_files = cast(list[tuple[int, str]], sub_source_files)
|
||||
|
||||
# # some functions may have "unterminated" inlines, but they usually have just one line number
|
||||
# if len(sub_source_files) % 2 != 0: # and len(line_numbers) != 1:
|
||||
# print(fun["name"], len(line_numbers))
|
||||
|
||||
# for addr, source in sub_source_files:
|
||||
|
||||
def find_function(self, name: str):
|
||||
functions = self._json["functions"]
|
||||
fun = next((f for f in functions if f["name"] == name), None)
|
||||
return fun
|
||||
|
||||
def get_line(self, addr: int):
|
||||
return self._line_cache.get(addr)
|
||||
|
||||
|
||||
def make_range_checker(lst: list[tuple[int, str]], delimiter: str) -> Callable[[int], bool]:
|
||||
"""
|
||||
Given a sorted list of (number, label) where number increases,
|
||||
build a checker that returns True if x is in a valid interval.
|
||||
|
||||
Interpretation:
|
||||
- Each entry (n, label) marks the interval starting at n and going
|
||||
up to the next entry's n (exclusive). The last entry's interval
|
||||
goes to +inf.
|
||||
- An interval starting at n is valid iff label == delimiter.
|
||||
- If the first entry's label != delimiter, everything before the first n is valid.
|
||||
"""
|
||||
if not lst:
|
||||
# no markers -> everything valid
|
||||
return lambda x: True
|
||||
|
||||
# ensure sorted by number
|
||||
lst_sorted = sorted(lst, key=lambda t: t[0])
|
||||
assert lst == lst_sorted
|
||||
|
||||
nums = [t[0] for t in lst_sorted]
|
||||
labels = [t[1] for t in lst_sorted]
|
||||
|
||||
# Precompute intervals as (start, end_exclusive, is_valid)
|
||||
intervals: list[tuple[int, int | None, bool]] = []
|
||||
n_items = len(nums)
|
||||
|
||||
for i in range(n_items):
|
||||
start = nums[i]
|
||||
end_exclusive: int | None
|
||||
|
||||
if i + 1 < n_items:
|
||||
end_exclusive = nums[i + 1]
|
||||
else:
|
||||
end_exclusive = None # means to +inf
|
||||
|
||||
is_valid = labels[i] == delimiter
|
||||
intervals.append((start, end_exclusive, is_valid))
|
||||
|
||||
first_before_is_valid = labels[0] != delimiter
|
||||
|
||||
def is_valid_fn(x: int) -> bool:
|
||||
# before first number
|
||||
if x < nums[0]:
|
||||
return first_before_is_valid
|
||||
|
||||
# find the interval that contains x
|
||||
for start, end_exclusive, valid_flag in intervals:
|
||||
if end_exclusive is None:
|
||||
if x >= start:
|
||||
return valid_flag
|
||||
else:
|
||||
if start <= x < end_exclusive:
|
||||
return valid_flag
|
||||
|
||||
# fallback (shouldn't happen)
|
||||
return False
|
||||
|
||||
return is_valid_fn
|
||||
|
||||
|
||||
def is_always_valid_fn(addr: int):
|
||||
return True
|
||||
|
||||
|
||||
def add_lines_to_asm(
|
||||
asm_path: Path,
|
||||
stdump_json_path: Path,
|
||||
fun_start_offset: int,
|
||||
keep_original_numbers: bool,
|
||||
asm_out: Path | None,
|
||||
):
|
||||
stdump_json = STDUMPJson(stdump_json_path)
|
||||
|
||||
function_name = asm_path.stem
|
||||
|
||||
if (fun := stdump_json.find_function(function_name)) is None:
|
||||
raise RuntimeError(f"Cannot find function '{function_name}' in ccc's JSON")
|
||||
|
||||
sub_source_files = fun.get("sub_source_files")
|
||||
# print(len(sub_source_files) if sub_source_files is not None else None)
|
||||
|
||||
asm_lines = asm_path.read_text().splitlines()
|
||||
re_instr = re.compile(r"^\s*\/\* [A-Z0-9]+ ([A-Z0-9]{8}) [A-Z0-9]{8} \*\/ .*$")
|
||||
line_dict: dict[int, int] = {}
|
||||
|
||||
relative_path: str = fun["relative_path"]
|
||||
non_func_addrs: list[int] = []
|
||||
|
||||
if sub_source_files:
|
||||
checker = make_range_checker(sub_source_files, relative_path)
|
||||
else:
|
||||
checker = is_always_valid_fn
|
||||
|
||||
start_line_num: int = sys.maxsize
|
||||
|
||||
for line in asm_lines:
|
||||
if m := re_instr.match(line):
|
||||
instr_addr = int(m.group(1), 16)
|
||||
line_num = stdump_json.get_line(instr_addr)
|
||||
|
||||
if line_num:
|
||||
line_dict[instr_addr] = line_num
|
||||
|
||||
if not checker(instr_addr):
|
||||
non_func_addrs.append(instr_addr)
|
||||
elif line_num:
|
||||
start_line_num = min(start_line_num, line_num)
|
||||
|
||||
if start_line_num == sys.maxsize:
|
||||
start_line_num = 1
|
||||
|
||||
min_line_num = 0 if keep_original_numbers else start_line_num - 1
|
||||
|
||||
new_asm_lines: list[str] = []
|
||||
asm_n: int = 0
|
||||
|
||||
for line in asm_lines:
|
||||
if m := re_instr.match(line):
|
||||
asm_n += 1
|
||||
|
||||
instr_addr = int(m.group(1), 16)
|
||||
line_num = line_dict.get(instr_addr)
|
||||
|
||||
if line_num is not None and instr_addr in non_func_addrs:
|
||||
new_asm_lines.append(f" .loc 1 {line_num} # inline")
|
||||
new_asm_lines.append(line)
|
||||
continue
|
||||
|
||||
if asm_n == 1 and line_num is None:
|
||||
# sometimes we don't have a number for the first line of assembly,
|
||||
# so we reuse the first known line among the ones we have
|
||||
line_num = start_line_num
|
||||
|
||||
if line_num is not None:
|
||||
new_line_num = (line_num - min_line_num) + fun_start_offset
|
||||
new_asm_lines.append(f" .loc 1 {new_line_num} ")
|
||||
|
||||
new_asm_lines.append(line)
|
||||
|
||||
stream = io.StringIO()
|
||||
stream.write(""".section .debug
|
||||
.previous
|
||||
.text
|
||||
.file 1 "source.c"
|
||||
|
||||
.set noat
|
||||
.set noreorder
|
||||
|
||||
""")
|
||||
stream.write("\n".join(new_asm_lines))
|
||||
|
||||
if asm_out is not None:
|
||||
asm_out.write_text(stream.getvalue())
|
||||
else:
|
||||
print(stream.getvalue())
|
||||
|
||||
|
||||
def main():
|
||||
class ArgProtocol(Protocol):
|
||||
asm_path: Path
|
||||
stdump_json_path: Path
|
||||
asm_out: Path | None
|
||||
offset: int
|
||||
keep_original: bool
|
||||
|
||||
parser = argparse.ArgumentParser(description="Add line debug info to assembly files")
|
||||
parser.add_argument("--asm-path", required=True, type=Path, help="Path to the assembly file to add lines to")
|
||||
parser.add_argument("--stdump-json-path", required=True, type=Path, help="Path to ccc's json stdump")
|
||||
parser.add_argument("--asm-out", type=Path, required=False, help="Path to output asm (defaults to stdout)")
|
||||
parser.add_argument(
|
||||
"--offset", type=int, required=False, default=0, help="Offset to apply to line numbers (default: 0)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--keep-original",
|
||||
action="store_true",
|
||||
help="Don't start line numbers from 1 (+ offset) but keep the original line numbers (+offset) instead",
|
||||
)
|
||||
|
||||
args = cast(ArgProtocol, parser.parse_args())
|
||||
|
||||
fun_start_num = args.offset
|
||||
|
||||
add_lines_to_asm(
|
||||
args.asm_path,
|
||||
args.stdump_json_path,
|
||||
fun_start_num,
|
||||
args.keep_original,
|
||||
args.asm_out,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,351 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import pydantic
|
||||
|
||||
from typing import Optional, Literal, Any
|
||||
|
||||
|
||||
class AddressRange(pydantic.BaseModel):
|
||||
low: int
|
||||
high: int
|
||||
|
||||
|
||||
class ValueType(pydantic.BaseModel):
|
||||
descriptor: Optional[str] = None
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
value_type: Optional[ValueType] = None
|
||||
return_type: Optional[ReturnType] = None
|
||||
modifier: Optional[str] = None
|
||||
vtable_index: Optional[int] = None
|
||||
is_constructor: Optional[bool] = None
|
||||
element_type: Optional[ElementType] = None
|
||||
element_count: Optional[int] = None
|
||||
|
||||
|
||||
class DeduplicatedTypeValueType(pydantic.BaseModel):
|
||||
descriptor: Literal["function_type", "type_name"]
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
value_type: Optional[ValueType] = None
|
||||
return_type: Optional[ReturnType] = None
|
||||
modifier: Optional[str] = None
|
||||
vtable_index: Optional[int] = None
|
||||
is_constructor: Optional[bool] = None
|
||||
element_type: Optional[ElementType] = None
|
||||
element_count: Optional[int] = None
|
||||
|
||||
|
||||
class ParameterType(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
value_type: Optional[ValueType] = None
|
||||
|
||||
|
||||
class Parameter(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
name: str
|
||||
class_: str = pydantic.Field(..., alias="class")
|
||||
storage: Storage
|
||||
block_low: int
|
||||
block_high: int
|
||||
type: ParameterType
|
||||
|
||||
|
||||
class ReturnType(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
value_type: Optional[ValueType] = None
|
||||
|
||||
|
||||
class FunctionType(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
parameters: list[Parameter]
|
||||
modifier: str
|
||||
vtable_index: int
|
||||
is_constructor: bool
|
||||
return_type: Optional[ReturnType] = None
|
||||
|
||||
|
||||
class Storage(pydantic.BaseModel):
|
||||
type: str
|
||||
register_: Optional[str] = pydantic.Field(None, alias="register")
|
||||
register_class: Optional[str] = None
|
||||
dbx_register_number: Optional[int] = None
|
||||
register_index: Optional[int] = None
|
||||
is_by_reference: Optional[bool] = None
|
||||
stack_offset: Optional[int] = None
|
||||
global_location: Optional[str] = None
|
||||
global_address: Optional[int] = None
|
||||
|
||||
|
||||
class Constant(pydantic.BaseModel):
|
||||
value: int
|
||||
name: str
|
||||
|
||||
|
||||
class ElementType(pydantic.BaseModel):
|
||||
descriptor: Literal["array", "pointer", "type_name", "enum"]
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
element_type: Optional[ElementType] = None
|
||||
element_count: Optional[int] = None
|
||||
value_type: Optional[ValueType] = None
|
||||
constants: Optional[list[Constant]] = None
|
||||
|
||||
def parsed_size(self) -> tuple[int, str]:
|
||||
if self.descriptor == "pointer":
|
||||
return 1, "pointer"
|
||||
|
||||
elif self.descriptor == "array":
|
||||
assert self.element_type
|
||||
assert self.element_count is not None
|
||||
if self.element_count == 0:
|
||||
# implicit size array
|
||||
_, type_name = self.element_type.parsed_size()
|
||||
return 0, type_name
|
||||
n, type_name = self.element_type.parsed_size()
|
||||
return self.element_count * n, type_name
|
||||
|
||||
elif self.descriptor == "enum":
|
||||
return 1, "enum"
|
||||
|
||||
else: # "type_name"
|
||||
assert self.type_name
|
||||
return 1, self.type_name
|
||||
|
||||
|
||||
class Local(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
name: str
|
||||
class_: str = pydantic.Field(..., alias="class")
|
||||
storage: Storage
|
||||
block_low: int
|
||||
block_high: int
|
||||
type: ElementType
|
||||
storage_class: Optional[str] = None
|
||||
|
||||
@property
|
||||
def parsed_size(self) -> tuple[int, str]:
|
||||
return self.type.parsed_size()
|
||||
|
||||
|
||||
class SubSourceFile(pydantic.BaseModel):
|
||||
address: int
|
||||
path: str
|
||||
|
||||
|
||||
class Function(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
name: str
|
||||
address_range: AddressRange
|
||||
type: FunctionType
|
||||
locals: list[Local]
|
||||
line_numbers: list[list[int]]
|
||||
sub_source_files: list[SubSourceFile]
|
||||
storage_class: Optional[str] = None
|
||||
relative_path: Optional[str] = None
|
||||
|
||||
|
||||
class Global(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
name: str
|
||||
class_: str = pydantic.Field(..., alias="class")
|
||||
storage: Storage
|
||||
block_low: int
|
||||
block_high: int
|
||||
type: ElementType
|
||||
storage_class: Optional[str] = None
|
||||
|
||||
@property
|
||||
def parsed_size(self) -> tuple[int, str]:
|
||||
return self.type.parsed_size()
|
||||
|
||||
|
||||
class File(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
path: str
|
||||
relative_path: str
|
||||
text_address: int
|
||||
types: list[Any]
|
||||
functions: list[Function]
|
||||
globals: list[Global]
|
||||
stabs_type_number_to_deduplicated_type_index: dict[str, int]
|
||||
|
||||
|
||||
class UnderlyingType(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
source: str
|
||||
type_name: str
|
||||
referenced_file_index: int
|
||||
referenced_stabs_type_number: int
|
||||
|
||||
|
||||
class Field(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
name: str
|
||||
relative_offset_bytes: int
|
||||
absolute_offset_bytes: int
|
||||
size_bits: int
|
||||
bitfield_offset_bits: Optional[int] = None
|
||||
underlying_type: Optional[UnderlyingType] = None
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
value_type: Optional[ValueType] = None
|
||||
base_classes: Optional[list[Any]] = None
|
||||
fields: Optional[list[Field]] = None
|
||||
member_functions: Optional[list[Any]] = None
|
||||
element_type: Optional[ElementType] = None
|
||||
element_count: Optional[int] = None
|
||||
|
||||
|
||||
class FieldModel(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
name: str
|
||||
relative_offset_bytes: int
|
||||
absolute_offset_bytes: int
|
||||
size_bits: int
|
||||
value_type: Optional[ValueType] = None
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
element_type: Optional[ElementType] = None
|
||||
element_count: Optional[int] = None
|
||||
bitfield_offset_bits: Optional[int] = None
|
||||
underlying_type: Optional[UnderlyingType] = None
|
||||
fields: Optional[list[Field]] = None
|
||||
member_functions: Optional[list[Any]] = None
|
||||
base_classes: Optional[list[Any]] = None
|
||||
|
||||
|
||||
class DeduplicatedType(pydantic.BaseModel):
|
||||
descriptor: Literal["array", "builtin", "enum", "pointer", "struct", "type_name", "union"]
|
||||
name: Optional[str] = None
|
||||
storage_class: Optional[Literal["typedef"]] = None
|
||||
stabs_type_number: int
|
||||
files: list[int]
|
||||
class_: Optional[str] = pydantic.Field(None, alias="class")
|
||||
size_bits: Optional[int] = None
|
||||
base_classes: Optional[list[Any]] = None
|
||||
fields: Optional[list[FieldModel]] = None
|
||||
member_functions: Optional[list[Any]] = None
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
value_type: Optional[DeduplicatedTypeValueType] = None
|
||||
conflict: Optional[bool] = None
|
||||
element_type: Optional[ElementType] = None
|
||||
element_count: Optional[int] = None
|
||||
constants: Optional[list[Constant]] = None
|
||||
|
||||
|
||||
class CCCJSONv7Model(pydantic.BaseModel):
|
||||
version: Literal[7]
|
||||
files: list[File]
|
||||
deduplicated_types: list[DeduplicatedType]
|
||||
|
||||
|
||||
# def test(stdump_json_path: str):
|
||||
# import json
|
||||
#
|
||||
# with open(stdump_json_path, mode="r") as fh:
|
||||
# json_data = fh.read()
|
||||
#
|
||||
# model = CCCJSONv7Model.model_validate_json(json_data)
|
||||
#
|
||||
# type_map: dict[str, int] = {}
|
||||
#
|
||||
# for n, dt in enumerate(model.deduplicated_types):
|
||||
# if dt.descriptor == "builtin":
|
||||
# assert dt.name and dt.name not in type_map
|
||||
# assert dt.class_ is not None
|
||||
# size_bits = int(dt.class_.split("-", maxsplit=1)[0])
|
||||
# assert size_bits % 8 == 0
|
||||
# type_map[dt.name] = size_bits // 8
|
||||
#
|
||||
# elif dt.descriptor == "type_name":
|
||||
# assert dt.name is not None
|
||||
# assert dt.type_name is not None
|
||||
# if dt.name in type_map:
|
||||
# if dt.size_bits:
|
||||
# assert type_map[dt.name] == dt.size_bits // 8
|
||||
# if dt.name == dt.type_name == "void":
|
||||
# type_map["void"] = type_map["int"]
|
||||
# continue
|
||||
# assert dt.type_name in type_map, (n, dt.type_name)
|
||||
# type_map[dt.name] = type_map[dt.type_name]
|
||||
#
|
||||
# elif dt.descriptor == "pointer":
|
||||
# assert dt.name is not None
|
||||
# assert dt.value_type is not None
|
||||
# if dt.value_type.descriptor == "function_type":
|
||||
# assert dt.name not in type_map
|
||||
# type_map[dt.name] = type_map["int"]
|
||||
# elif dt.value_type.descriptor == "type_name":
|
||||
# assert dt.value_type.type_name
|
||||
# assert dt.value_type.type_name in type_map
|
||||
# type_map[dt.name] = type_map[dt.value_type.type_name]
|
||||
#
|
||||
# elif dt.descriptor == "struct":
|
||||
# assert dt.name is not None
|
||||
# if not dt.conflict:
|
||||
# assert dt.name not in type_map, n
|
||||
# else:
|
||||
# if dt.name in type_map:
|
||||
# continue
|
||||
# assert dt.size_bits is not None
|
||||
# assert dt.size_bits % 8 == 0
|
||||
# type_map[dt.name] = dt.size_bits // 8
|
||||
#
|
||||
# elif dt.descriptor == "array":
|
||||
# assert dt.name is not None
|
||||
# element_count = 1
|
||||
# element_type = dt.element_type
|
||||
# type_name = None
|
||||
# while element_type:
|
||||
# if element_type.element_count is not None:
|
||||
# element_count *= element_type.element_count
|
||||
# if element_type.type_name is not None:
|
||||
# type_name = element_type.type_name
|
||||
# assert type_name in type_map
|
||||
# element_count *= type_map[type_name]
|
||||
# element_type = element_type.element_type
|
||||
# assert type_name and type_name in type_map, n
|
||||
# type_map[dt.name] = type_map[type_name] * element_count
|
||||
#
|
||||
# elif dt.descriptor == "enum":
|
||||
# if dt.name:
|
||||
# if not dt.conflict:
|
||||
# assert dt.name not in type_map, n
|
||||
# type_map[dt.name] = type_map["int"]
|
||||
#
|
||||
# elif dt.descriptor == "union":
|
||||
# assert dt.name
|
||||
# assert dt.size_bits
|
||||
# assert dt.size_bits % 8 == 0
|
||||
# type_map[dt.name] = dt.size_bits // 8
|
||||
#
|
||||
# else:
|
||||
# assert False, f"unknown {n}"
|
||||
#
|
||||
# print(json.dumps(type_map, indent=2))
|
||||
|
||||
|
||||
# if __name__ == "__main__":
|
||||
# test(path-to-stdump-json)
|
||||
@@ -0,0 +1,322 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import sys
|
||||
import math
|
||||
import ctypes
|
||||
import numpy as np
|
||||
import numpy.typing as npt
|
||||
from collections.abc import Sized, Sequence
|
||||
from typing import Type, TypeVar, Any, cast, TextIO, BinaryIO, Optional
|
||||
from ctypes import LittleEndianStructure, Union, Array, c_uint32
|
||||
|
||||
_G = TypeVar("_G")
|
||||
|
||||
CTypeType = (
|
||||
type[ctypes.c_int8]
|
||||
| type[ctypes.c_uint8]
|
||||
| type[ctypes.c_int16]
|
||||
| type[ctypes.c_uint16]
|
||||
| type[ctypes.c_int32]
|
||||
| type[ctypes.c_uint32]
|
||||
| type[ctypes.c_float]
|
||||
| type[ctypes.c_double]
|
||||
)
|
||||
|
||||
ctypes_types: dict[str, CTypeType] = {
|
||||
"char": ctypes.c_int8,
|
||||
"u_char": ctypes.c_uint8,
|
||||
"short": ctypes.c_int16,
|
||||
"u_short": ctypes.c_uint16,
|
||||
"int": ctypes.c_int32,
|
||||
"u_int": ctypes.c_uint32,
|
||||
"float": ctypes.c_float,
|
||||
"double": ctypes.c_double,
|
||||
}
|
||||
|
||||
|
||||
class c_addr(c_uint32):
|
||||
def __str__(self):
|
||||
# if self.value == 0:
|
||||
# return "NULL"
|
||||
|
||||
# return f"0x{self.value:08x}"
|
||||
return f"0x{self.value:x}"
|
||||
|
||||
|
||||
class c_str(c_uint32):
|
||||
def to_str(self, elf: BinaryIO):
|
||||
if self.value == 0:
|
||||
return "NULL"
|
||||
elf.seek(self.value)
|
||||
buf = io.BytesIO()
|
||||
while (c := elf.read(1)) != b"\0":
|
||||
buf.write(c)
|
||||
return '"' + buf.getvalue().decode("ASCII") + '"'
|
||||
|
||||
|
||||
class c_addr_ptr(c_uint32):
|
||||
_addresses: dict[int, str] | None = None
|
||||
|
||||
@classmethod
|
||||
def set_addresses(cls, addresses: dict[int, str] | None):
|
||||
cls._addresses = addresses
|
||||
|
||||
def __str__(self):
|
||||
if self.value == 0:
|
||||
return "NULL"
|
||||
|
||||
if self._addresses and self.value in self._addresses:
|
||||
return self._addresses[self.value]
|
||||
|
||||
print(f"warning: no address for pointer 0x{self.value:x}")
|
||||
|
||||
return f"0x{self.value:x}"
|
||||
|
||||
|
||||
def print_arr(arr: npt.NDArray[np.int_], lst: list[Any], file: TextIO):
|
||||
if arr.ndim > 1:
|
||||
for x in arr:
|
||||
file.write("{")
|
||||
print_arr(x, lst, file)
|
||||
file.write("},")
|
||||
else:
|
||||
file.write(",".join(str(lst[i]) for i in arr))
|
||||
|
||||
|
||||
def print_carr(arr: Sequence[Any], file: TextIO):
|
||||
if hasattr(arr[0], "_length_"):
|
||||
for a in arr:
|
||||
file.write("{")
|
||||
print_carr(a, file)
|
||||
file.write("},")
|
||||
else:
|
||||
v = str(str([x for x in arr])).replace("[", "").replace("]", "")
|
||||
file.write(v)
|
||||
|
||||
|
||||
def chunks(lst: Sequence[_G], n: int):
|
||||
"""Yield successive n-sized chunks from lst."""
|
||||
for i in range(0, len(lst), n):
|
||||
yield lst[i : i + n]
|
||||
|
||||
|
||||
def format_array(lst: Sequence[_G], dims: Sequence[int], file: TextIO):
|
||||
if len(dims) > 1:
|
||||
for ll in chunks(lst, len(lst) // dims[0]):
|
||||
file.write("{")
|
||||
format_array(ll, dims[1:], file)
|
||||
file.write("},")
|
||||
else:
|
||||
for n, x in enumerate(lst):
|
||||
sep = ", " if n < len(lst) - 1 else ""
|
||||
file.write(f"{x}{sep}")
|
||||
|
||||
|
||||
def resolve_annotations(namespace: dict[str, Any], annotations: dict[str, Any]):
|
||||
module = sys.modules.get(namespace.get("__module__", ""))
|
||||
globals_ = vars(module) if module else {}
|
||||
resolved: dict[str, Any] = {}
|
||||
|
||||
for k, v in annotations.items():
|
||||
if isinstance(v, str):
|
||||
try:
|
||||
resolved[k] = eval(v, globals_, namespace)
|
||||
except Exception:
|
||||
resolved[k] = v # keep as string if can't resolve
|
||||
else:
|
||||
resolved[k] = v
|
||||
|
||||
return resolved
|
||||
|
||||
|
||||
class LittleEndianStructureFieldsFromTypeHints(type(LittleEndianStructure)): # pyright: ignore
|
||||
def __new__(
|
||||
cls: Type[type],
|
||||
name: str,
|
||||
bases: tuple[type, ...],
|
||||
namespace: dict[str, Any],
|
||||
/,
|
||||
*,
|
||||
align: Optional[int] = None,
|
||||
pack: Optional[int] = None,
|
||||
) -> LittleEndianStructureFieldsFromTypeHints:
|
||||
annotations = namespace.get("__annotations__", {})
|
||||
annotations = resolve_annotations(namespace, annotations)
|
||||
if "__elf__" in annotations:
|
||||
annotations.pop("__elf__")
|
||||
if align is not None:
|
||||
namespace["_align_"] = align
|
||||
if pack is not None:
|
||||
namespace["_pack_"] = pack
|
||||
namespace["_layout_"] = "ms"
|
||||
if fields := list(annotations.items()):
|
||||
namespace["_fields_"] = fields
|
||||
return type(LittleEndianStructure).__new__(cls, name, bases, namespace) # pyright: ignore
|
||||
|
||||
|
||||
class CStructure(LittleEndianStructure, metaclass=LittleEndianStructureFieldsFromTypeHints):
|
||||
__elf__: BinaryIO
|
||||
|
||||
@classmethod
|
||||
def sizeof(cls) -> int:
|
||||
align = getattr(cls, "_align_", 0)
|
||||
c_size = ctypes.sizeof(cls)
|
||||
if align > 0:
|
||||
rem = c_size % align
|
||||
if rem:
|
||||
rem = align - rem
|
||||
return c_size + rem
|
||||
|
||||
return c_size
|
||||
|
||||
@classmethod
|
||||
def parse(cls, data: bytes):
|
||||
cstruct_size = sizeof(cls)
|
||||
assert len(data) % cstruct_size == 0, f"{len(data)}, {cstruct_size}"
|
||||
cstruct_num = len(data) // cstruct_size
|
||||
stream = io.BytesIO(data)
|
||||
cstructs = [cls.from_buffer_copy(stream.read(cstruct_size)) for _ in range(cstruct_num)]
|
||||
return cstructs
|
||||
|
||||
@classmethod
|
||||
def dumps(
|
||||
cls,
|
||||
name: str,
|
||||
data: bytes,
|
||||
numel: int | list[int],
|
||||
static: bool = False,
|
||||
nosize: bool = False,
|
||||
noarray: bool = False,
|
||||
):
|
||||
cstructs = cls.parse(data)
|
||||
stream = io.StringIO()
|
||||
if static:
|
||||
stream.write("static ")
|
||||
stream.write(f"{cls.__name__} {name}") # pyright: ignore
|
||||
assert not (noarray and isinstance(numel, list))
|
||||
if not noarray:
|
||||
if isinstance(numel, int):
|
||||
assert len(cstructs) == numel, (len(cstructs), numel)
|
||||
numel_str = f"{len(cstructs)}" if not nosize else ""
|
||||
stream.write(f"[{numel_str}]")
|
||||
else:
|
||||
numel_str = "".join([f"[{num if n > 0 or not nosize else ''}]" for n, num in enumerate(numel)])
|
||||
stream.write(numel_str)
|
||||
tot_numel = max(1, numel) if isinstance(numel, int) else math.prod(numel)
|
||||
# nmdim = 1 if isinstance(numel, int) else len(numel)
|
||||
assert len(cstructs) == tot_numel, (len(cstructs), tot_numel)
|
||||
|
||||
stream.write(" = ")
|
||||
if not noarray:
|
||||
stream.write("{\n")
|
||||
if isinstance(numel, int):
|
||||
for n, s in enumerate(cstructs):
|
||||
is_last = n == len(cstructs) - 1
|
||||
if is_last and noarray:
|
||||
stream.write(f"{s}\n")
|
||||
else:
|
||||
stream.write(f"{s},\n")
|
||||
|
||||
else:
|
||||
idxs = np.arange(tot_numel).reshape(numel)
|
||||
print_arr(idxs, cstructs, stream)
|
||||
if not noarray:
|
||||
stream.write("}")
|
||||
stream.write(";\n\n")
|
||||
return stream.getvalue()
|
||||
|
||||
# def to_str(self, elf: BinaryIO):
|
||||
def __str__(self):
|
||||
stream = io.StringIO()
|
||||
stream.write(" {\n")
|
||||
for f, *_ in self._fields_: # pyright: ignore
|
||||
if f.startswith("_pad_"):
|
||||
continue
|
||||
v = getattr(self, f) # pyright: ignore
|
||||
if f in ("_in", "_pass"):
|
||||
f = f[1:] # pyright: ignore
|
||||
if isinstance(v, c_str):
|
||||
stream.write(f" .{f} = {v.to_str(self.__elf__)},\n")
|
||||
elif not isinstance(v, Array):
|
||||
stream.write(f" .{f} = {v},\n")
|
||||
else:
|
||||
arr = cast(Sized, v)
|
||||
if len(arr) and isinstance(v[0], CStructure):
|
||||
arr = cast(list[CStructure], arr)
|
||||
stream.write(f" .{f} = {{\n")
|
||||
for elem in arr:
|
||||
stream.write(f" {elem},\n")
|
||||
stream.write(" },\n")
|
||||
else:
|
||||
arr = cast(Sequence[Any], arr)
|
||||
if isinstance(arr[0], Array):
|
||||
# multidimensional ctypes array
|
||||
stream.write(f" .{f} = {{")
|
||||
print_carr(arr, stream)
|
||||
stream.write("},\n")
|
||||
else:
|
||||
stream.write(f" .{f} = {{")
|
||||
dims = [len(arr)]
|
||||
format_array(arr, dims, stream)
|
||||
|
||||
stream.write("},\n")
|
||||
stream.write(" }")
|
||||
return stream.getvalue()
|
||||
|
||||
|
||||
class UnionFieldsFromTypeHints(type(Union)): # pyright: ignore
|
||||
def __new__(
|
||||
cls: Type[type],
|
||||
name: str,
|
||||
bases: tuple[type, ...],
|
||||
namespace: dict[str, Any],
|
||||
/,
|
||||
*,
|
||||
align: Optional[int] = None,
|
||||
pack: Optional[int] = None,
|
||||
) -> UnionFieldsFromTypeHints:
|
||||
annotations = namespace.get("__annotations__", {})
|
||||
if align is not None:
|
||||
namespace["_align_"] = align
|
||||
if pack is not None:
|
||||
namespace["_pack_"] = pack
|
||||
if fields := list(annotations.items()):
|
||||
namespace["_fields_"] = fields
|
||||
return type(Union).__new__(cls, name, bases, namespace) # pyright: ignore
|
||||
|
||||
|
||||
class CUnion(Union, metaclass=UnionFieldsFromTypeHints):
|
||||
pass
|
||||
|
||||
|
||||
_T = TypeVar("_T", bound=CStructure)
|
||||
|
||||
|
||||
def sizeof(cstruct_type: Type[_T]) -> int:
|
||||
align = getattr(cstruct_type, "_align_", 0)
|
||||
c_size = ctypes.sizeof(cstruct_type)
|
||||
if align > 0:
|
||||
rem = c_size % align
|
||||
if rem:
|
||||
rem = align - rem
|
||||
return c_size + rem
|
||||
|
||||
return c_size
|
||||
|
||||
|
||||
def parse_cstruct(cstruct_type: Type[_T], data: bytes) -> list[_T]:
|
||||
cstruct_size = sizeof(cstruct_type)
|
||||
assert len(data) % cstruct_size == 0, f"{len(data)}, {cstruct_size}"
|
||||
cstruct_num = len(data) // cstruct_size
|
||||
stream = io.BytesIO(data)
|
||||
cstructs = [cstruct_type.from_buffer_copy(stream.read(cstruct_size)) for _ in range(cstruct_num)]
|
||||
return cstructs
|
||||
|
||||
|
||||
def print_cstruct(name: str, cstruct_type: Type[CStructure], data: bytes):
|
||||
cstructs = parse_cstruct(cstruct_type, data)
|
||||
print(f"{cstruct_type.__name__} {name}[{len(cstructs)}] = {{") # pyright: ignore
|
||||
for s in cstructs:
|
||||
print(s)
|
||||
print("};\n")
|
||||
@@ -0,0 +1,56 @@
|
||||
import argparse
|
||||
|
||||
|
||||
registers = {
|
||||
"$0": "zero",
|
||||
"$1": "at",
|
||||
"$2": "v0",
|
||||
"$3": "v1",
|
||||
"$4": "a0",
|
||||
"$5": "a1",
|
||||
"$6": "a2",
|
||||
"$7": "a3",
|
||||
"$8": "t0",
|
||||
"$9": "t1",
|
||||
"$10": "t2",
|
||||
"$11": "t3",
|
||||
"$12": "t4",
|
||||
"$13": "t5",
|
||||
"$14": "t6",
|
||||
"$15": "t7",
|
||||
"$16": "s0",
|
||||
"$17": "s1",
|
||||
"$18": "s2",
|
||||
"$19": "s3",
|
||||
"$20": "s4",
|
||||
"$21": "s5",
|
||||
"$22": "s6",
|
||||
"$23": "s7",
|
||||
"$24": "t8",
|
||||
"$25": "t9",
|
||||
"$26": "k0",
|
||||
"$27": "k1",
|
||||
"$28": "gp",
|
||||
"$29": "sp",
|
||||
"$30": "fp",
|
||||
"$31": "ra",
|
||||
}
|
||||
|
||||
|
||||
def fix_asm(asm_file: str):
|
||||
with open(asm_file, mode="r") as fh:
|
||||
for line in fh:
|
||||
for reg_num, reg_mnem in reversed(registers.items()):
|
||||
line = line.replace(reg_num, reg_mnem)
|
||||
print(line, end="")
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("asm_file", help="source assembly file path")
|
||||
args = parser.parse_args()
|
||||
fix_asm(args.asm_file)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,125 @@
|
||||
import os
|
||||
import re
|
||||
import argparse
|
||||
import pydantic
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Protocol, cast
|
||||
|
||||
|
||||
CONFIG_ROOT = Path(__file__).parent.parent.parent.resolve() / "config"
|
||||
|
||||
# enumerate languages from config dir
|
||||
LANGUAGES = [Path(f.path).name for f in os.scandir(CONFIG_ROOT) if f.is_dir()]
|
||||
|
||||
RE_INSTR = re.compile(r"^\t(?:[^.]|\.p2align \d+)")
|
||||
|
||||
|
||||
class InstructionPatch(pydantic.RootModel[tuple[int, str, str]]):
|
||||
root: tuple[int, str, str]
|
||||
|
||||
@property
|
||||
def instr_no(self):
|
||||
return self.root[0]
|
||||
|
||||
@property
|
||||
def instr_org(self):
|
||||
return self.root[1]
|
||||
|
||||
@property
|
||||
def instr_patch(self):
|
||||
return self.root[2]
|
||||
|
||||
|
||||
class ASMPatch(pydantic.RootModel[dict[str, list[InstructionPatch]]]):
|
||||
root: dict[str, list[InstructionPatch]]
|
||||
|
||||
|
||||
class PatchDB(pydantic.RootModel[dict[str, ASMPatch]]):
|
||||
root: dict[str, ASMPatch]
|
||||
|
||||
|
||||
def fix_asm(asm_file: Path, asm_patch: ASMPatch):
|
||||
lines = asm_file.read_text().splitlines()
|
||||
|
||||
def find_func(func: str):
|
||||
for i, line in enumerate(lines):
|
||||
if re.match(rf"^{func}:", line):
|
||||
return i
|
||||
|
||||
return -1
|
||||
|
||||
def find_line(start: int, num: int):
|
||||
offset = 0
|
||||
instr_no = -1
|
||||
for line in lines[start:]:
|
||||
if RE_INSTR.match(line):
|
||||
instr_no += 1
|
||||
|
||||
if instr_no == num:
|
||||
return start + offset
|
||||
|
||||
offset += 1
|
||||
|
||||
raise RuntimeError("cannot find instruction!")
|
||||
|
||||
for func, patch_lst in asm_patch.root.items():
|
||||
n = find_func(func)
|
||||
|
||||
if n == -1:
|
||||
print(f"WARNING: cannot find function {func} in {asm_file.name}")
|
||||
continue
|
||||
|
||||
for patch in patch_lst:
|
||||
line_no = find_line(n, patch.instr_no)
|
||||
line_org = lines[line_no]
|
||||
|
||||
line_org_clean = re.sub(r"\s+", " ", line_org).strip()
|
||||
instr_org_clean = re.sub(r"\s+", " ", patch.instr_org).strip()
|
||||
instr_patch_clean = re.sub(r"\s+", " ", patch.instr_patch).strip()
|
||||
|
||||
if line_org_clean != instr_org_clean:
|
||||
print(f"WARNING: wrong line: {asm_file.name}:{func}:{line_no + 1}: {line_org} != {patch.instr_org}")
|
||||
continue
|
||||
|
||||
lines[line_no] = f"\t{instr_patch_clean}"
|
||||
|
||||
asm_file.write_text("\n".join(lines))
|
||||
|
||||
|
||||
def main():
|
||||
class ArgsProtocol(Protocol):
|
||||
language: str
|
||||
asm_file: Path
|
||||
|
||||
parser = argparse.ArgumentParser(description="apply asm patches to assembly files")
|
||||
parser.add_argument("language", type=str, choices=LANGUAGES, help="language of the asm that is being patched")
|
||||
parser.add_argument("asm_file", type=Path, help="generated assembly file to patch (relative to build dir)")
|
||||
|
||||
args = cast(ArgsProtocol, parser.parse_args())
|
||||
|
||||
asm_file = CONFIG_ROOT / args.language / args.asm_file
|
||||
|
||||
if not asm_file.exists():
|
||||
print(f"ERROR: cannot find assembly file {asm_file}")
|
||||
exit(1)
|
||||
|
||||
patch_db_path = CONFIG_ROOT / args.language / "asm_patches.json"
|
||||
|
||||
if not patch_db_path.exists():
|
||||
print("ERROR: patch db file doesn't exist. aborting patching asm")
|
||||
exit(1)
|
||||
|
||||
patch_db = PatchDB.model_validate_json(patch_db_path.read_text())
|
||||
|
||||
if asm_file.name not in patch_db.root:
|
||||
print(f"ERROR: no patches found for file {asm_file.name}")
|
||||
exit(1)
|
||||
|
||||
asm_patch = patch_db.root[asm_file.name]
|
||||
|
||||
fix_asm(asm_file, asm_patch)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,38 @@
|
||||
import re
|
||||
import glob
|
||||
import tqdm
|
||||
import argparse
|
||||
|
||||
from typing import Protocol, cast
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def fix_assets(asm_data_path: Path, asset_rel_path: Path):
|
||||
asm_files = glob.glob(str(asm_data_path / "**/*.s"), recursive=True)
|
||||
for asm_file in tqdm.tqdm(asm_files, desc="Fixing data asm"):
|
||||
n: int = 0
|
||||
with open(asm_file, mode="r") as fh:
|
||||
data_asm: str = fh.read()
|
||||
data_asm, n = re.subn(rf'\.incbin "{asset_rel_path}/', '.incbin "assets/', data_asm)
|
||||
|
||||
if n > 0:
|
||||
with open(asm_file, mode="w") as wh:
|
||||
wh.write(data_asm)
|
||||
|
||||
|
||||
def main():
|
||||
class ArgsProtocol(Protocol):
|
||||
asm_data_path: Path
|
||||
asset_rel_path: Path
|
||||
|
||||
parser = argparse.ArgumentParser(description="fixes data asm include path")
|
||||
parser.add_argument("asm_path", metavar="asm-path", type=Path, help="data path in assembly root to patch")
|
||||
parser.add_argument("asset_rel_path", metavar="asset-path", type=Path, help="relative asset path")
|
||||
|
||||
args = cast(ArgsProtocol, parser.parse_args())
|
||||
|
||||
fix_assets(args.asm_data_path, args.asset_rel_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,79 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import argparse
|
||||
import pydantic
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Protocol, cast
|
||||
|
||||
|
||||
CONFIG_ROOT = Path(__file__).parent.parent.parent.resolve() / "config"
|
||||
|
||||
# enumerate languages from config dir
|
||||
LANGUAGES = [Path(f.path).name for f in os.scandir(CONFIG_ROOT) if f.is_dir()]
|
||||
|
||||
|
||||
class BytePatch(pydantic.RootModel[tuple[int, int]]):
|
||||
root: tuple[int, int]
|
||||
|
||||
@property
|
||||
def address(self):
|
||||
return self.root[0]
|
||||
|
||||
@property
|
||||
def size(self):
|
||||
return self.root[1]
|
||||
|
||||
|
||||
class BINPatch(pydantic.RootModel[dict[str, BytePatch]]):
|
||||
root: dict[str, BytePatch]
|
||||
|
||||
|
||||
class PatchDB(pydantic.RootModel[dict[str, BINPatch]]):
|
||||
root: dict[str, BINPatch]
|
||||
|
||||
|
||||
def fix_elf(orig_elf_path: Path, built_elf_path: Path, bin_patch: BytePatch):
|
||||
with orig_elf_path.open("rb") as fh:
|
||||
fh.seek(bin_patch.address)
|
||||
data = fh.read(bin_patch.size)
|
||||
|
||||
with built_elf_path.open(mode="r+b") as wh:
|
||||
wh.seek(bin_patch.address)
|
||||
wh.write(data)
|
||||
|
||||
|
||||
def main():
|
||||
class ArgsProtocol(Protocol):
|
||||
language: str
|
||||
elf_file: Path
|
||||
|
||||
parser = argparse.ArgumentParser(description="apply asm patches to assembly files")
|
||||
parser.add_argument("language", type=str, choices=LANGUAGES, help="language of the elf that is being patched")
|
||||
parser.add_argument("elf_file", type=Path, help="elf file to patch (relative to build dir)")
|
||||
|
||||
args = cast(ArgsProtocol, parser.parse_args())
|
||||
|
||||
built_elf_file = CONFIG_ROOT / args.language / args.elf_file
|
||||
|
||||
if not built_elf_file.exists():
|
||||
print(f"ERROR: cannot find elf file {built_elf_file}")
|
||||
exit(1)
|
||||
|
||||
orig_elf_path = CONFIG_ROOT / args.language / args.elf_file.name
|
||||
|
||||
patch_db_path = CONFIG_ROOT / args.language / "bin_patches.json"
|
||||
|
||||
if not patch_db_path.exists():
|
||||
exit(0)
|
||||
|
||||
patch_db = PatchDB.model_validate_json(patch_db_path.read_text())
|
||||
|
||||
for _tu_name, bin_patch in patch_db.root.items():
|
||||
for _func_name, byte_patch in bin_patch.root.items():
|
||||
fix_elf(orig_elf_path, built_elf_file, byte_patch)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,75 @@
|
||||
import re
|
||||
import glob
|
||||
import tqdm
|
||||
import argparse
|
||||
import functools
|
||||
|
||||
from typing import Protocol, cast
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
@functools.lru_cache(maxsize=1024) # pyright: ignore[reportUntypedFunctionDecorator]
|
||||
def get_symbol_address(symbol_addrs: Path, symbol: str):
|
||||
if match := re.match(r"^ *([^+ ]+) *\+ *(0x[0-9a-fA-F]+)$", symbol):
|
||||
symbol = match.group(1)
|
||||
offset = int(match.group(2), 16)
|
||||
else:
|
||||
offset = 0
|
||||
|
||||
with open(symbol_addrs, mode="r") as fh:
|
||||
for line in fh:
|
||||
if line.startswith(f"{symbol} "):
|
||||
addr = int(line.split("=")[1].split(";")[0].strip(), 16)
|
||||
return addr + offset
|
||||
|
||||
assert False, f"{symbol} not found"
|
||||
|
||||
|
||||
def fix_gp(asm_path: Path, gp_value: int, symbol_addrs: Path):
|
||||
if gp_value <= 0:
|
||||
return
|
||||
|
||||
asm_files = glob.glob(str(asm_path / "**/*.s"), recursive=True)
|
||||
for asm_file in tqdm.tqdm(asm_files, desc="Fixing gp_rel"):
|
||||
lines: list[str] = []
|
||||
with open(asm_file, mode="r") as fh:
|
||||
for line in fh:
|
||||
if match := re.match(r"^(.*)%gp_rel\(([^)]+)\)(.*)$", line):
|
||||
# ol = line
|
||||
instr_pre = match.group(1)
|
||||
instr_post = match.group(3)
|
||||
address_str = match.group(2)
|
||||
if address_str.startswith("D_"):
|
||||
address_str = address_str.replace("D_", "0x")
|
||||
res = eval(address_str)
|
||||
address = res
|
||||
else:
|
||||
address = get_symbol_address(symbol_addrs, address_str)
|
||||
gp_rel = address - gp_value
|
||||
line = f"{instr_pre}{hex(gp_rel)}{instr_post}\n"
|
||||
lines.append(line)
|
||||
with open(asm_file, mode="w") as wh:
|
||||
wh.writelines(lines)
|
||||
|
||||
|
||||
def main():
|
||||
class ArgsProtocol(Protocol):
|
||||
asm_path: Path
|
||||
gp_value: int
|
||||
symbol_addrs: Path
|
||||
|
||||
def hex_int(x: str):
|
||||
return int(x, 16)
|
||||
|
||||
parser = argparse.ArgumentParser(description="fixes asm removing gp_rel macro")
|
||||
parser.add_argument("asm_path", metavar="asm-path", type=Path, help="assembly root path to patch")
|
||||
parser.add_argument("gp_value", metavar="gp", type=hex_int, help="gp value in hex")
|
||||
parser.add_argument("symbol_addrs", metavar="symbol-addrs", type=Path, help="path of symbol_addrs.txt")
|
||||
|
||||
args = cast(ArgsProtocol, parser.parse_args())
|
||||
|
||||
fix_gp(args.asm_path, args.gp_value, args.symbol_addrs)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,139 @@
|
||||
import re
|
||||
import sys
|
||||
import yaml
|
||||
import tqdm
|
||||
|
||||
from typing import cast, Any
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
# e.g.: build/src/main/main.c.o(.text);
|
||||
# build/asm/data/rodata_4.rodata.s.o(.rodata);
|
||||
# ...
|
||||
# NOTE: also account for optional '/data/' in path as chunks of data
|
||||
# not belonging to a c file are put into 'build/asm/data/' folder.
|
||||
re_subsegment_line = re.compile(
|
||||
r"^(?P<indent> +)build/(?:asm|src)(?:/data)?/(?P<name>.*)\.[sc]\.o\(\.(?P<section>.+)\);$"
|
||||
)
|
||||
|
||||
# e.g.: .main 0x100000 : AT(main_ROM_START) SUBALIGN(2)
|
||||
# .main_bss (NOLOAD) : SUBALIGN(4)
|
||||
# ...
|
||||
re_section_line = re.compile(r"^(?P<indent> +)\.(?P<section>[^ ]+) .* SUBALIGN\((?P<subalign>[0-9]+)\)$")
|
||||
|
||||
|
||||
def get_align(address: int):
|
||||
return (
|
||||
128 * ((address % 128) == 0)
|
||||
or 64 * ((address % 64) == 0)
|
||||
or 32 * ((address % 32) == 0)
|
||||
or 16 * ((address % 16) == 0)
|
||||
or 8 * ((address % 8) == 0)
|
||||
or 4 * ((address % 4) == 0)
|
||||
or 2 * ((address % 2) == 0)
|
||||
or 1
|
||||
)
|
||||
|
||||
|
||||
def make_align_map(config: dict[str, Any]):
|
||||
segments = cast(list[dict[str, Any] | list[Any]] | None, config["segments"])
|
||||
assert segments
|
||||
main_segment = next(
|
||||
(segment for segment in segments if isinstance(segment, dict) and segment.get("name") == "main"), None
|
||||
)
|
||||
assert main_segment, "cannot find main segment"
|
||||
|
||||
subsegments = cast(list[dict[str, Any] | list[Any]] | None, main_segment["subsegments"])
|
||||
assert subsegments, "cannot find main subsegments"
|
||||
|
||||
align_map: dict[str, int] = {}
|
||||
|
||||
for subsegment in subsegments:
|
||||
if not isinstance(subsegment, dict):
|
||||
continue
|
||||
|
||||
s_type = cast(str | None, subsegment.get("type"))
|
||||
vram = cast(int | None, subsegment.get("vram"))
|
||||
name = cast(str | None, subsegment.get("name"))
|
||||
if not s_type or not vram or not name:
|
||||
continue
|
||||
|
||||
if name.endswith("bin"):
|
||||
continue
|
||||
|
||||
align = get_align(vram)
|
||||
|
||||
if not s_type.startswith("."):
|
||||
name = f"{s_type}#{name}.{s_type}"
|
||||
else:
|
||||
name = f"{s_type[1:]}#{name}"
|
||||
|
||||
align_map[name] = align
|
||||
|
||||
return align_map
|
||||
|
||||
|
||||
def fix_linkerscript(config: dict[str, Any], linkerscript_path: Path):
|
||||
align_map = make_align_map(config)
|
||||
|
||||
section_subalign = cast(dict[str, int], config["_section_subalign"])
|
||||
|
||||
line_count = 0
|
||||
with open(linkerscript_path, mode="r") as fh:
|
||||
for line in fh:
|
||||
line_count += 1
|
||||
|
||||
patched_lines: list[str] = []
|
||||
|
||||
with open(linkerscript_path, mode="r") as fh:
|
||||
for line in tqdm.tqdm(fh, desc="Fixing linker script", total=line_count):
|
||||
if match := re_subsegment_line.match(line):
|
||||
indent = cast(str, match["indent"])
|
||||
name = cast(str, match["name"])
|
||||
section = cast(str, match["section"])
|
||||
|
||||
# force each subsegment in the following sections to have align 8
|
||||
if section == "text":
|
||||
patched_lines.append(f"{indent}. = ALIGN(., 8);\n")
|
||||
|
||||
key = f"{section}#{name}"
|
||||
if align := align_map.get(key):
|
||||
patched_lines.append(f"{indent}. = ALIGN(., {align});\n")
|
||||
|
||||
if match := re_section_line.match(line):
|
||||
indent = cast(str, match["indent"])
|
||||
section = cast(str, match["section"])
|
||||
subalign = cast(str, match["subalign"])
|
||||
|
||||
# force each section to have the subalign specified in the yaml
|
||||
if section in section_subalign:
|
||||
current_subalign = f"SUBALIGN({subalign})"
|
||||
fixed_subalign = f"SUBALIGN({section_subalign[section]})"
|
||||
line = line.replace(current_subalign, fixed_subalign)
|
||||
|
||||
patched_lines.append(line)
|
||||
|
||||
with open(linkerscript_path, mode="w") as fh:
|
||||
fh.writelines(patched_lines)
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) != 3:
|
||||
print("usage: fix_linkerscript.py CONFIG_YAML_PATH LINKERSCRIPT_PATH")
|
||||
exit(1)
|
||||
|
||||
config_path = Path(sys.argv[1])
|
||||
linkerscript_path = Path(sys.argv[2])
|
||||
|
||||
with open(config_path, mode="r") as fh:
|
||||
try:
|
||||
config = cast(dict[str, Any], yaml.safe_load(fh))
|
||||
except yaml.YAMLError as e:
|
||||
print(e)
|
||||
raise e
|
||||
|
||||
fix_linkerscript(config, linkerscript_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,153 @@
|
||||
import json
|
||||
import argparse
|
||||
|
||||
from typing import Protocol, Iterable, Any, cast
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def fix_unit(unit: dict[str, Any]):
|
||||
# name: str = unit["name"]
|
||||
fuzzy_match_percent: float | None = unit["measures"].get("fuzzy_match_percent")
|
||||
|
||||
assert fuzzy_match_percent and 0 < fuzzy_match_percent < 100
|
||||
|
||||
text_section = next(section for section in unit["sections"] if section["name"] == ".text")
|
||||
# text_size: int = int(text_section["size"])
|
||||
text_section_fuzzy_match_percent: float = text_section["fuzzy_match_percent"]
|
||||
|
||||
if 0 < text_section_fuzzy_match_percent < 100:
|
||||
# fix text section fuzzy match percent
|
||||
text_section["fuzzy_match_percent"] = 100.0
|
||||
|
||||
text_fuzzy_match_percent: float = unit["measures"]["fuzzy_match_percent"]
|
||||
|
||||
assert text_fuzzy_match_percent == fuzzy_match_percent
|
||||
|
||||
functions = unit["functions"]
|
||||
|
||||
total_code: int = int(unit["measures"]["total_code"])
|
||||
matched_code: int = int(unit["measures"]["matched_code"])
|
||||
matched_functions = unit["measures"]["matched_functions"]
|
||||
|
||||
computed_total_code: int = 0
|
||||
computed_matched_code: int = 0
|
||||
computed_matched_functions: int = 0
|
||||
|
||||
duplicate_functions = (
|
||||
"Tim2CalcBufWidth__2",
|
||||
"_ftoi0__2",
|
||||
"ItemGetMain__2",
|
||||
"setD3_CHCR__2",
|
||||
"setD4_CHCR__2",
|
||||
"setD4_CHCR__3",
|
||||
"_fpadd_parts__2",
|
||||
)
|
||||
|
||||
for function in functions:
|
||||
function_size: int = int(function["size"])
|
||||
try:
|
||||
function_fuzzy_match_percent: float = function["fuzzy_match_percent"]
|
||||
except KeyError:
|
||||
# fix known function duplicates
|
||||
if function["name"] in duplicate_functions:
|
||||
function["fuzzy_match_percent"] = 100.0
|
||||
function_fuzzy_match_percent = 100.0
|
||||
matched_code += function_size
|
||||
matched_functions += 1
|
||||
else:
|
||||
raise
|
||||
|
||||
if function_fuzzy_match_percent == 100.0:
|
||||
computed_matched_code += function_size
|
||||
computed_matched_functions += 1
|
||||
|
||||
# fix function fuzzy match percent
|
||||
function["fuzzy_match_percent"] = 100.0
|
||||
|
||||
computed_total_code += function_size
|
||||
|
||||
assert total_code == computed_total_code
|
||||
assert matched_code == computed_matched_code
|
||||
assert matched_functions == computed_matched_functions
|
||||
|
||||
# fix unit measures
|
||||
unit["measures"]["fuzzy_match_percent"] = 100.0
|
||||
unit["measures"]["matched_code"] = unit["measures"]["total_code"]
|
||||
unit["measures"]["matched_code_percent"] = 100.0
|
||||
unit["measures"]["matched_functions"] = unit["measures"]["total_functions"]
|
||||
unit["measures"]["matched_functions_percent"] = 100.0
|
||||
|
||||
|
||||
def fix_report(report_path: Path):
|
||||
report = json.loads(report_path.read_text())
|
||||
|
||||
units: Iterable[Any] = report["units"]
|
||||
|
||||
computed_total_code: int = 0
|
||||
computed_matched_code: int = 0
|
||||
|
||||
computed_total_functions: int = 0
|
||||
computed_matched_functions: int = 0
|
||||
|
||||
total_code: int = int(report["measures"]["total_code"])
|
||||
total_functions: int = report["measures"]["total_functions"]
|
||||
|
||||
for unit in units:
|
||||
# name: str = unit["name"]
|
||||
unit_total_code: int = int(unit["measures"]["total_code"])
|
||||
unit_total_functions: int = unit["measures"]["total_functions"]
|
||||
fuzzy_match_percent: float | None = unit["measures"].get("fuzzy_match_percent")
|
||||
|
||||
if fuzzy_match_percent and 0 < fuzzy_match_percent < 100:
|
||||
fix_unit(unit)
|
||||
|
||||
if fuzzy_match_percent and fuzzy_match_percent > 0:
|
||||
computed_matched_code += unit_total_code
|
||||
computed_matched_functions += unit_total_functions
|
||||
|
||||
computed_total_functions += unit_total_functions
|
||||
computed_total_code += unit_total_code
|
||||
|
||||
assert total_code == computed_total_code
|
||||
assert total_functions == computed_total_functions
|
||||
|
||||
# fix report measures
|
||||
report["measures"]["fuzzy_match_percent"] = 100.0 * computed_matched_code / computed_total_code
|
||||
report["measures"]["matched_code"] = str(computed_matched_code)
|
||||
report["measures"]["matched_code_percent"] = report["measures"]["fuzzy_match_percent"]
|
||||
report["measures"]["matched_functions"] = computed_matched_functions
|
||||
report["measures"]["matched_functions_percent"] = 100.0 * computed_matched_functions / computed_total_functions
|
||||
|
||||
categories = report["categories"]
|
||||
assert len(categories) == 1
|
||||
assert categories[0]["measures"]["total_code"] == report["measures"]["total_code"]
|
||||
assert categories[0]["measures"]["total_units"] == report["measures"]["total_units"]
|
||||
|
||||
categories[0]["measures"]["fuzzy_match_percent"] = report["measures"]["fuzzy_match_percent"]
|
||||
categories[0]["measures"]["matched_code"] = report["measures"]["matched_code"]
|
||||
categories[0]["measures"]["matched_code_percent"] = report["measures"]["matched_code_percent"]
|
||||
categories[0]["measures"]["matched_functions"] = report["measures"]["matched_functions"]
|
||||
categories[0]["measures"]["matched_functions_percent"] = report["measures"]["matched_functions_percent"]
|
||||
|
||||
# /path/to/report.json -> /path/to/report_fixed.json
|
||||
# fixed_report_path = report_path.with_name(f"{report_path.stem}_fixed{report_path.suffix}")
|
||||
# fixed_report_path.write_text(json.dumps(report))
|
||||
report_path.write_text(json.dumps(report))
|
||||
|
||||
print(f"Wrote fixed report to {report_path}")
|
||||
|
||||
|
||||
def main():
|
||||
class ArgsProtocol(Protocol):
|
||||
report_path: Path
|
||||
|
||||
parser = argparse.ArgumentParser(description="fixes objdiff report")
|
||||
parser.add_argument("report_path", metavar="report-path", type=Path, help="path to the report generated by objdiff")
|
||||
|
||||
args = cast(ArgsProtocol, parser.parse_args())
|
||||
|
||||
fix_report(args.report_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,32 @@
|
||||
import re
|
||||
import argparse
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--language", required=True, choices=["us", "eu"])
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.language == "us":
|
||||
map_path = Path("config/us/build/SLUS_203.88.map")
|
||||
else:
|
||||
map_path = Path("config/eu/build/SLES_508.21.map")
|
||||
|
||||
with open(map_path, mode="r") as fh:
|
||||
for n, line in enumerate(fh):
|
||||
line = line.rstrip("\n")
|
||||
if match := re.match(r"^\s*0x([0-9a-fA-F]+)\s+[^ ]+?([0-9a-fA-F]{6,8})\s*$", line):
|
||||
addr = match.group(1)
|
||||
label = match.group(2)
|
||||
if addr.upper() != label.upper():
|
||||
print(f"{map_path}:{n+1} {line}")
|
||||
return
|
||||
|
||||
print("no mismatches found")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,288 @@
|
||||
# pyright: reportUnknownMemberType=false
|
||||
|
||||
import re
|
||||
import argparse
|
||||
|
||||
from typing import cast
|
||||
from pathlib import Path
|
||||
from elftools.elf.elffile import ELFFile
|
||||
from elftools.elf.sections import SymbolTableSection
|
||||
|
||||
|
||||
"""
|
||||
matches variable declaration:
|
||||
/* SECTION ADDRESS */ VAR_TYPE NAME[NUMEL];
|
||||
"""
|
||||
re_glob = re.compile(
|
||||
r"^/\* (?P<section>.*?) (?P<address>.*?) \*/ (?P<var_type>[^(]*) (?P<name>.*?)(?:\[(?P<numel>.*?)\])?;"
|
||||
)
|
||||
|
||||
"""
|
||||
matches structs (or unions) with no typedef:
|
||||
struct NAME { // SIZE
|
||||
...
|
||||
};
|
||||
"""
|
||||
re_struct = re.compile(
|
||||
r"^(?:struct|union) (?P<name>.*?) \{ // (?P<size>0x[0-9a-f]+)\n.*?^\};", flags=re.MULTILINE | re.DOTALL
|
||||
)
|
||||
|
||||
"""
|
||||
matches structs (or unions) with typedef:
|
||||
typedef struct { // SIZE
|
||||
...
|
||||
} NAME;
|
||||
"""
|
||||
re_typedef_struct = re.compile(
|
||||
r"^typedef (?:struct|union) \{ // (?P<size>0x[0-9a-f]+)\n.*?^\} (?P<name>.*?);", flags=re.MULTILINE | re.DOTALL
|
||||
)
|
||||
|
||||
sizes: dict[str, int] = {
|
||||
"char": 1,
|
||||
"u_char": 1,
|
||||
"short": 2,
|
||||
"short int": 2,
|
||||
"u_short": 2,
|
||||
"int": 4,
|
||||
"u_int": 4,
|
||||
"float": 4,
|
||||
"u_long128": 8,
|
||||
"sceVu0FMATRIX": 4 * 4 * 4,
|
||||
"sceVu0FVECTOR": 4 * 4,
|
||||
"sceSifClientData": 0x2C,
|
||||
}
|
||||
|
||||
command_script_keywords = (
|
||||
"VERSION",
|
||||
"SECTIONS",
|
||||
"ABSOLUTE",
|
||||
"LOADADDR",
|
||||
"ALIGN",
|
||||
"DEFINED",
|
||||
"NEXT",
|
||||
"SIZEOF",
|
||||
"SIZEOF_HEADERS",
|
||||
"MAX",
|
||||
"MIN",
|
||||
"PHDRS",
|
||||
"CREATE_OBJECT_SYMBOLS",
|
||||
"BYTE",
|
||||
"SHORT",
|
||||
"LONG",
|
||||
"SQUAD",
|
||||
"FILL",
|
||||
"BLOCK",
|
||||
"NOLOAD",
|
||||
"AT",
|
||||
"OVERLAY",
|
||||
"NOCROSSREFS",
|
||||
"PT_NULL",
|
||||
"PT_LOAD",
|
||||
"PT_DYNAMIC",
|
||||
"PT_INTERP",
|
||||
"PT_NOTE",
|
||||
"PT_SHLIB",
|
||||
"PT_PHDR",
|
||||
"ENTRY",
|
||||
"FLOAT",
|
||||
"NOFLOAT",
|
||||
"FORCE_COMMON_ALLOCATION",
|
||||
"INCLUDE",
|
||||
"INPUT",
|
||||
"GROUP",
|
||||
"OUTPUT",
|
||||
"OUTPUT_ARCH",
|
||||
"OUTPUT_FORMAT",
|
||||
"SEARCH_DIR",
|
||||
"STARTUP",
|
||||
"TARGET",
|
||||
"NOCROSSREFS",
|
||||
)
|
||||
|
||||
|
||||
class GlobalVarLineMatch:
|
||||
max_section_len: int = 0
|
||||
max_address_len: int = 0
|
||||
max_var_type_len: int = 0
|
||||
max_name_len: int = 0
|
||||
max_numel_len: int = 0
|
||||
|
||||
_name_cache: list[str] = []
|
||||
|
||||
@classmethod
|
||||
def _get_unique_name(cls, name: str):
|
||||
i = 1
|
||||
unique_name = f"{name}__local_{i}"
|
||||
while unique_name in cls._name_cache:
|
||||
i += 1
|
||||
unique_name = f"{name}__local_{i}"
|
||||
cls._name_cache.append(unique_name)
|
||||
return unique_name
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
symtab: SymbolTableSection,
|
||||
/,
|
||||
*,
|
||||
section: str,
|
||||
address: str | int,
|
||||
var_type: str,
|
||||
name: str,
|
||||
numel: str | int | None = None,
|
||||
):
|
||||
self.symtab: SymbolTableSection = symtab
|
||||
|
||||
self.section: str = section
|
||||
self.address: int = address if isinstance(address, int) else int(address, 16)
|
||||
self.var_type: str = var_type
|
||||
self.name: str = name
|
||||
self.numel: int
|
||||
|
||||
while self.name.startswith("*"):
|
||||
self.name = self.name[1:]
|
||||
self.var_type += "*"
|
||||
|
||||
# resolve function pointers (e.g., 'void (*SpecialEventInitTbl[0])...')
|
||||
if match := re.match(r"\(\*(.*?)\[.*?\]\).*", self.name):
|
||||
self.name = match.group(1)
|
||||
|
||||
self.name = self.name.replace("*", "")
|
||||
|
||||
if numel is not None:
|
||||
if isinstance(numel, int):
|
||||
self.numel = numel
|
||||
else:
|
||||
prod_numel = 1
|
||||
numels = numel.split("][")
|
||||
for n in numels:
|
||||
prod_numel *= int(n)
|
||||
self.numel = prod_numel
|
||||
|
||||
if numel is None:
|
||||
self.numel = 1
|
||||
|
||||
# get symbol
|
||||
symbol = self.symtab.get_symbol_by_name(self.name)
|
||||
|
||||
assert isinstance(symbol, list)
|
||||
|
||||
symbol = next(
|
||||
(sym for sym in symbol if sym.name == self.name and sym.entry["st_value"] == self.address),
|
||||
None,
|
||||
)
|
||||
|
||||
assert symbol
|
||||
|
||||
self.symbol = symbol
|
||||
|
||||
if not self.is_global:
|
||||
self.name = self._get_unique_name(self.name)
|
||||
|
||||
escaped_name_len = len(str(self.name))
|
||||
if self.name in command_script_keywords:
|
||||
escaped_name_len += 2
|
||||
|
||||
GlobalVarLineMatch.max_section_len = max(GlobalVarLineMatch.max_section_len, len(str(self.section)))
|
||||
GlobalVarLineMatch.max_address_len = max(GlobalVarLineMatch.max_address_len, len(str(self.address)))
|
||||
GlobalVarLineMatch.max_var_type_len = max(GlobalVarLineMatch.max_var_type_len, len(str(self.var_type)))
|
||||
GlobalVarLineMatch.max_name_len = max(GlobalVarLineMatch.max_name_len, escaped_name_len)
|
||||
GlobalVarLineMatch.max_numel_len = max(GlobalVarLineMatch.max_numel_len, len(str(self.numel)))
|
||||
|
||||
@property
|
||||
def is_global(self):
|
||||
return cast(str, self.symbol["st_info"]["bind"]) == "STB_GLOBAL"
|
||||
|
||||
@property
|
||||
def is_hidden(self):
|
||||
return cast(str, self.symbol["st_other"]["visibility"]) == "STV_HIDDEN"
|
||||
|
||||
@property
|
||||
def size(self):
|
||||
size = cast(int, self.symbol["st_size"])
|
||||
if size == 0:
|
||||
if "*" in self.var_type:
|
||||
size = 4
|
||||
elif self.var_type in sizes:
|
||||
size = sizes[self.var_type] * self.numel
|
||||
return size
|
||||
|
||||
def to_string(self, as_linker_command_file: bool):
|
||||
name = self.name if self.name not in command_script_keywords else f'"{self.name}"'
|
||||
cls_str = f"{name:{GlobalVarLineMatch.max_name_len}s} = 0x{self.address:08x};"
|
||||
|
||||
if not as_linker_command_file:
|
||||
if self.size > 0:
|
||||
cls_str = f"{cls_str} // size:0x{self.size:x}"
|
||||
else:
|
||||
cls_str = f"{cls_str} //0 {self.var_type} * {self.numel}"
|
||||
|
||||
# cls_str += f" bind:{'global' if self.is_global else 'local'}"
|
||||
# cls_str += f" visibility:{'hidden' if self.is_hidden else 'visible'}"
|
||||
|
||||
return cls_str
|
||||
|
||||
def __str__(self):
|
||||
return self.to_string(as_linker_command_file=False)
|
||||
|
||||
|
||||
def parse_globals(elf_path: Path, globals_path: Path, types_path: Path, as_linker_command_file: bool):
|
||||
with open(elf_path, mode="rb") as fh:
|
||||
elf = ELFFile(fh)
|
||||
|
||||
# Find the symbol table.
|
||||
symtab = elf.get_section_by_name(".symtab")
|
||||
assert isinstance(symtab, SymbolTableSection)
|
||||
|
||||
with open(types_path, mode="r") as f:
|
||||
types_data = f.read()
|
||||
|
||||
for struct_type, size_hex_str in re_struct.findall(types_data):
|
||||
struct_type = cast(str, struct_type)
|
||||
struct_size = int(cast(str, size_hex_str), 16)
|
||||
# assert not (struct_type in sizes and sizes[struct_type] != struct_size)
|
||||
sizes[struct_type] = struct_size
|
||||
|
||||
for size_hex_str, struct_type in re_typedef_struct.findall(types_data):
|
||||
struct_type = cast(str, struct_type)
|
||||
struct_size = int(cast(str, size_hex_str), 16)
|
||||
# assert not (struct_type in sizes and sizes[struct_type] != struct_size), (
|
||||
# struct_type,
|
||||
# sizes[struct_type],
|
||||
# struct_size,
|
||||
# )
|
||||
sizes[struct_type] = struct_size
|
||||
|
||||
if "tagSE_WRK" in sizes:
|
||||
sizes["SE_WRK"] = sizes["tagSE_WRK"]
|
||||
|
||||
with open(globals_path, mode="r") as f:
|
||||
lines = [line.strip() for line in f.readlines() if line.startswith("/*") and line.strip().endswith(";")]
|
||||
|
||||
matches = [GlobalVarLineMatch(symtab, **match.groupdict()) for line in lines if (match := re_glob.match(line))]
|
||||
matches.sort(key=lambda match: match.address)
|
||||
|
||||
for match in matches:
|
||||
print(match.to_string(as_linker_command_file=as_linker_command_file))
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--language", required=True, choices=["us", "eu"])
|
||||
parser.add_argument("--as-linker-command-file", action="store_true")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.language == "us":
|
||||
elf_path = Path("config/us/SLUS_203.88")
|
||||
globals_path = Path("ccc/SLUS_203.88/globals.h")
|
||||
types_path = Path("ccc/SLUS_203.88/types.h")
|
||||
else:
|
||||
elf_path = Path("config/eu/SLES_508.21")
|
||||
globals_path = Path("ccc/SLES_508.21/globals.h")
|
||||
types_path = Path("ccc/SLES_508.21/types.h")
|
||||
|
||||
parse_globals(elf_path, globals_path, types_path, args.as_linker_command_file)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,378 @@
|
||||
# pyright: reportUnknownMemberType=false
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import json
|
||||
import argparse
|
||||
import itertools
|
||||
|
||||
from typing import Protocol, TextIO, cast
|
||||
from pathlib import Path
|
||||
from elftools.elf.elffile import ELFFile
|
||||
from elftools.elf.sections import SymbolTableSection, Symbol
|
||||
|
||||
import ccc_json_v7
|
||||
|
||||
Range = tuple[int, int]
|
||||
|
||||
command_script_keywords = (
|
||||
"VERSION",
|
||||
"SECTIONS",
|
||||
"ABSOLUTE",
|
||||
"LOADADDR",
|
||||
"ALIGN",
|
||||
"DEFINED",
|
||||
"NEXT",
|
||||
"SIZEOF",
|
||||
"SIZEOF_HEADERS",
|
||||
"MAX",
|
||||
"MIN",
|
||||
"PHDRS",
|
||||
"CREATE_OBJECT_SYMBOLS",
|
||||
"BYTE",
|
||||
"SHORT",
|
||||
"LONG",
|
||||
"SQUAD",
|
||||
"FILL",
|
||||
"BLOCK",
|
||||
"NOLOAD",
|
||||
"AT",
|
||||
"OVERLAY",
|
||||
"NOCROSSREFS",
|
||||
"PT_NULL",
|
||||
"PT_LOAD",
|
||||
"PT_DYNAMIC",
|
||||
"PT_INTERP",
|
||||
"PT_NOTE",
|
||||
"PT_SHLIB",
|
||||
"PT_PHDR",
|
||||
"ENTRY",
|
||||
"FLOAT",
|
||||
"NOFLOAT",
|
||||
"FORCE_COMMON_ALLOCATION",
|
||||
"INCLUDE",
|
||||
"INPUT",
|
||||
"GROUP",
|
||||
"OUTPUT",
|
||||
"OUTPUT_ARCH",
|
||||
"OUTPUT_FORMAT",
|
||||
"SEARCH_DIR",
|
||||
"STARTUP",
|
||||
"TARGET",
|
||||
"NOCROSSREFS",
|
||||
)
|
||||
|
||||
skip_symbols = (
|
||||
"_fbss",
|
||||
"_gp",
|
||||
)
|
||||
|
||||
size_exceptions = {
|
||||
"dorcon": 0x1C,
|
||||
}
|
||||
|
||||
|
||||
def parse_types(ccc_model_v7: ccc_json_v7.CCCJSONv7Model):
|
||||
type_map: dict[str, int] = {}
|
||||
|
||||
for n, dt in enumerate(ccc_model_v7.deduplicated_types):
|
||||
if dt.descriptor == "builtin":
|
||||
assert dt.name and dt.name not in type_map
|
||||
assert dt.class_ is not None
|
||||
size_bits = int(dt.class_.split("-", maxsplit=1)[0])
|
||||
assert size_bits % 8 == 0
|
||||
type_map[dt.name] = size_bits // 8
|
||||
|
||||
elif dt.descriptor == "type_name":
|
||||
assert dt.name is not None
|
||||
assert dt.type_name is not None
|
||||
if dt.name in type_map:
|
||||
if dt.size_bits:
|
||||
assert type_map[dt.name] == dt.size_bits // 8
|
||||
if dt.name == dt.type_name == "void":
|
||||
type_map["void"] = type_map["int"]
|
||||
continue
|
||||
assert dt.type_name in type_map, (n, dt.type_name)
|
||||
type_map[dt.name] = type_map[dt.type_name]
|
||||
|
||||
elif dt.descriptor == "pointer":
|
||||
assert dt.name is not None
|
||||
assert dt.value_type is not None
|
||||
if dt.value_type.descriptor == "function_type":
|
||||
assert dt.name not in type_map
|
||||
type_map[dt.name] = type_map["int"]
|
||||
elif dt.value_type.descriptor == "type_name":
|
||||
assert dt.value_type.type_name
|
||||
assert dt.value_type.type_name in type_map
|
||||
type_map[dt.name] = type_map[dt.value_type.type_name]
|
||||
|
||||
elif dt.descriptor == "struct":
|
||||
assert dt.name is not None
|
||||
if not dt.conflict:
|
||||
assert dt.name not in type_map, n
|
||||
else:
|
||||
if dt.name in type_map:
|
||||
continue
|
||||
assert dt.size_bits is not None
|
||||
assert dt.size_bits % 8 == 0
|
||||
type_map[dt.name] = dt.size_bits // 8
|
||||
|
||||
elif dt.descriptor == "array":
|
||||
assert dt.name is not None
|
||||
element_count = 1
|
||||
element_type = dt.element_type
|
||||
type_name = None
|
||||
while element_type:
|
||||
if element_type.element_count is not None:
|
||||
element_count *= element_type.element_count
|
||||
if element_type.type_name is not None:
|
||||
type_name = element_type.type_name
|
||||
assert type_name in type_map
|
||||
element_count *= type_map[type_name]
|
||||
element_type = element_type.element_type
|
||||
assert type_name and type_name in type_map, n
|
||||
type_map[dt.name] = type_map[type_name] * element_count
|
||||
|
||||
elif dt.descriptor == "enum":
|
||||
if dt.name:
|
||||
if not dt.conflict:
|
||||
assert dt.name not in type_map, n
|
||||
type_map[dt.name] = type_map["int"]
|
||||
|
||||
elif dt.descriptor == "union":
|
||||
assert dt.name
|
||||
assert dt.size_bits
|
||||
assert dt.size_bits % 8 == 0
|
||||
type_map[dt.name] = dt.size_bits // 8
|
||||
|
||||
else:
|
||||
assert False, f"unknown {n}"
|
||||
|
||||
return type_map
|
||||
|
||||
|
||||
def in_range(ranges: list[Range], address: int):
|
||||
if not ranges:
|
||||
return True
|
||||
|
||||
in_range = any((ra[0] <= address <= ra[1]) for ra in ranges)
|
||||
|
||||
return in_range
|
||||
|
||||
|
||||
def parse_ccc_model_v7(ccc_model_v7: ccc_json_v7.CCCJSONv7Model, ranges: list[Range]):
|
||||
type_sizes = parse_types(ccc_model_v7)
|
||||
|
||||
if "pointer" not in type_sizes:
|
||||
type_sizes["pointer"] = type_sizes["int"]
|
||||
|
||||
static_locals: list[ccc_json_v7.Local] = []
|
||||
global_vars: list[ccc_json_v7.Global] = []
|
||||
|
||||
for file in ccc_model_v7.files:
|
||||
for global_var in file.globals:
|
||||
assert global_var.storage.global_address
|
||||
if in_range(ranges, global_var.storage.global_address):
|
||||
global_vars.append(global_var)
|
||||
|
||||
for function in file.functions:
|
||||
for local in function.locals:
|
||||
if local.storage_class == "static":
|
||||
assert local.storage.global_address
|
||||
if in_range(ranges, local.storage.global_address):
|
||||
static_locals.append(local)
|
||||
|
||||
return type_sizes, global_vars, static_locals
|
||||
|
||||
|
||||
class SymbolWithNoNameException(Exception): ...
|
||||
|
||||
|
||||
class ParsedSymbol:
|
||||
_name_map: dict[str, list[ParsedSymbol]] = {}
|
||||
_max_name_len: int = 0
|
||||
|
||||
def __init__(self, symtab: SymbolTableSection, symbol: Symbol) -> None:
|
||||
self.symtab = symtab
|
||||
self.symbol = symbol
|
||||
|
||||
if not self.name:
|
||||
raise SymbolWithNoNameException
|
||||
|
||||
if self.name not in ParsedSymbol._name_map:
|
||||
ParsedSymbol._name_map[self.name] = []
|
||||
|
||||
ParsedSymbol._name_map[self.name].append(self)
|
||||
ParsedSymbol._max_name_len = max(ParsedSymbol._max_name_len, len(self.name))
|
||||
|
||||
@property
|
||||
def name(self) -> str:
|
||||
return re.sub(r"^(.*?)(\.\d+)?$", r"\1", self.symbol.name)
|
||||
|
||||
@property
|
||||
def size(self) -> int:
|
||||
return cast(int, self.symbol["st_size"])
|
||||
|
||||
@property
|
||||
def address(self) -> int:
|
||||
return cast(int, self.symbol.entry["st_value"])
|
||||
|
||||
def in_range(self, range: Range):
|
||||
return range[0] <= self.address <= range[1]
|
||||
|
||||
def to_undefined_syms(self) -> str:
|
||||
homonyms = ParsedSymbol._name_map[self.name]
|
||||
assert len(homonyms) > 0
|
||||
if len(homonyms) == 1:
|
||||
assert homonyms[0] == self
|
||||
name = self.name
|
||||
else:
|
||||
idx = homonyms.index(self)
|
||||
name = f"{self.name}__local_{idx + 1}"
|
||||
|
||||
name = name if name not in command_script_keywords else f'"{name}"'
|
||||
|
||||
# name_len_fmt = ParsedSymbol._max_name_len + len("__local_9999")
|
||||
name_len_fmt = 40
|
||||
|
||||
return f"{name:{name_len_fmt}s} = 0x{self.address:08x};"
|
||||
|
||||
def to_symbol_addrs(
|
||||
self,
|
||||
type_sizes: dict[str, int],
|
||||
global_vars: list[ccc_json_v7.Global],
|
||||
static_locals: list[ccc_json_v7.Local],
|
||||
) -> str:
|
||||
homonyms = ParsedSymbol._name_map[self.name]
|
||||
assert len(homonyms) > 0
|
||||
if len(homonyms) == 1:
|
||||
assert homonyms[0] == self
|
||||
name = self.name
|
||||
else:
|
||||
idx = homonyms.index(self)
|
||||
name = f"{self.name}__local_{idx + 1}"
|
||||
|
||||
# name_len_fmt = ParsedSymbol._max_name_len + len("__local_9999")
|
||||
name_len_fmt = 40
|
||||
|
||||
size = size_exceptions.get(self.name, self.size)
|
||||
if size == 0:
|
||||
var = next(
|
||||
(global_ for global_ in global_vars if global_.storage.global_address == self.address), None
|
||||
) or next((local for local in static_locals if local.storage.global_address == self.address), None)
|
||||
|
||||
if not var:
|
||||
print(f"cannot find {self.name} with size 0 in json")
|
||||
else:
|
||||
element_count, type_name = var.parsed_size
|
||||
type_size = type_sizes.get(type_name)
|
||||
assert type_size is not None, type_name
|
||||
size = element_count * type_size
|
||||
|
||||
if not size:
|
||||
print(f"size of {self.name} is also 0 using json")
|
||||
|
||||
size_str = f"size:0x{size:x}" if size else ""
|
||||
|
||||
return f"{name:{name_len_fmt}s} = 0x{self.address:08x}; // {size_str}"
|
||||
|
||||
|
||||
def parse_symbols_safe(elf_path: Path, dest_path: Path, json_path: Path, ranges: list[Range]):
|
||||
if not dest_path.is_dir():
|
||||
raise RuntimeError(f"{dest_path} is not a directory")
|
||||
|
||||
symbol_addrs_path = dest_path / "symbols_addrs.txt"
|
||||
undefined_syms_path = dest_path / "undefined_syms.txt"
|
||||
|
||||
if symbol_addrs_path.exists() or undefined_syms_path.exists():
|
||||
raise RuntimeError("symbols_addrs.txt or undefined_syms.txt already exist in dest folder")
|
||||
|
||||
with open(elf_path, mode="rb") as elf_fh, open(json_path, mode="r") as json_fh:
|
||||
elf = ELFFile(elf_fh)
|
||||
|
||||
json_data = json.load(json_fh)
|
||||
ccc_model_v7 = ccc_json_v7.CCCJSONv7Model.model_validate(json_data)
|
||||
type_sizes, global_vars, static_locals = parse_ccc_model_v7(ccc_model_v7, ranges)
|
||||
|
||||
with open(symbol_addrs_path, mode="w") as symbol_addrs, open(undefined_syms_path, mode="w") as undefined_syms:
|
||||
parse_symbols(elf, symbol_addrs, undefined_syms, ranges, type_sizes, global_vars, static_locals)
|
||||
|
||||
|
||||
def parse_symbols(
|
||||
elf: ELFFile,
|
||||
symbol_addrs: TextIO,
|
||||
undefined_syms: TextIO,
|
||||
ranges: list[Range],
|
||||
type_sizes: dict[str, int],
|
||||
global_vars: list[ccc_json_v7.Global],
|
||||
static_locals: list[ccc_json_v7.Local],
|
||||
):
|
||||
# find symbol table
|
||||
symtab = elf.get_section_by_name(".symtab")
|
||||
assert isinstance(symtab, SymbolTableSection)
|
||||
|
||||
parsed_symbols: list[ParsedSymbol] = []
|
||||
|
||||
for symbol in symtab.iter_symbols():
|
||||
try:
|
||||
parsed_symbol = ParsedSymbol(symtab, symbol)
|
||||
if parsed_symbol.name in skip_symbols:
|
||||
continue
|
||||
parsed_symbols.append(parsed_symbol)
|
||||
except SymbolWithNoNameException:
|
||||
pass
|
||||
|
||||
parsed_symbols.sort(key=lambda x: x.address)
|
||||
|
||||
for parsed_symbol in parsed_symbols:
|
||||
if ranges:
|
||||
in_range = any(parsed_symbol.in_range(ra) for ra in ranges)
|
||||
if not in_range:
|
||||
continue
|
||||
|
||||
parsed_symbol.address
|
||||
symbol_addrs.write(parsed_symbol.to_symbol_addrs(type_sizes, global_vars, static_locals))
|
||||
symbol_addrs.write("\n")
|
||||
|
||||
undefined_syms.write(parsed_symbol.to_undefined_syms())
|
||||
undefined_syms.write("\n")
|
||||
|
||||
|
||||
def main():
|
||||
class ArgsProtocol(Protocol):
|
||||
elf_path: Path
|
||||
dest_path: Path
|
||||
json_path: Path
|
||||
ranges: list[Range]
|
||||
|
||||
def range_type(arg: str) -> Range:
|
||||
if m := re.match(r"^([0-9a-f]+)-([0-9a-f]+)$", arg):
|
||||
range_start, range_end = int(m.group(1), 16), int(m.group(2), 16)
|
||||
if not (range_end > range_start):
|
||||
raise argparse.ArgumentTypeError("non monotonic range")
|
||||
return range_start, range_end
|
||||
raise argparse.ArgumentTypeError("invalid range")
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("elf_path", metavar="ELF", type=Path, help="path to ELF file")
|
||||
parser.add_argument("--json", dest="json_path", type=Path, help="path to CCC json (v7)")
|
||||
parser.add_argument(
|
||||
"--dest",
|
||||
dest="dest_path",
|
||||
type=Path,
|
||||
required=True,
|
||||
help="folder where symbol_addrs.txt and undefined_syms.txt will be created "
|
||||
"(existing files will not be overwritten)",
|
||||
)
|
||||
parser.add_argument("--range", action="append", dest="ranges", nargs="+", type=range_type)
|
||||
|
||||
args = parser.parse_args()
|
||||
args.ranges = list(itertools.chain(*args.ranges))
|
||||
args = cast(ArgsProtocol, args)
|
||||
|
||||
parse_symbols_safe(args.elf_path, args.dest_path, args.json_path, args.ranges)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,414 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Literal, Optional
|
||||
from pydantic import RootModel, BaseModel, Field
|
||||
|
||||
|
||||
class Compiler(BaseModel):
|
||||
name: str
|
||||
asm_function_macro: str = Field("glabel", description="")
|
||||
asm_function_alt_macro: str = Field("glabel", description="")
|
||||
asm_jtbl_label_macro: str = Field("glabel", description="")
|
||||
asm_data_macro: str = Field("glabel", description="")
|
||||
asm_end_label: str = Field("", description="")
|
||||
c_newline: str = Field("\n", description="")
|
||||
asm_inc_header: str = Field("", description="")
|
||||
include_macro_inc: bool = Field(True, description="")
|
||||
asm_emit_size_directive: Optional[bool] = Field(None, description="")
|
||||
|
||||
|
||||
class SplatOpts(BaseModel):
|
||||
# Debug / logging
|
||||
verbose: Optional[bool] = Field(None, description="Verbose")
|
||||
dump_symbols: Optional[bool] = Field(None, description="Dump symbols")
|
||||
modes: Optional[list[str]] = Field(None, description="modes")
|
||||
|
||||
# Project configuration
|
||||
base_path: Path = Field(
|
||||
None, description="Determines the base path of the project. Everything is relative to this path"
|
||||
)
|
||||
target_path: Optional[Path] = Field(None, description="Determines the path to the target binary")
|
||||
elf_path: Optional[Path] = Field(None, description="Path to the final elf target")
|
||||
platform: Optional[str] = Field(None, description="Determines the platform of the target binary")
|
||||
compiler: Optional[Compiler | str] = Field(
|
||||
None, description="Determines the compiler used to compile the target binary"
|
||||
)
|
||||
endianness: Optional[Literal["big", "little"]] = Field(
|
||||
None, description="Determines the endianness of the target binary"
|
||||
)
|
||||
section_order: Optional[list[str]] = Field(
|
||||
None,
|
||||
description="Determines the default section order of the target binary. This can be overridden per-segment",
|
||||
)
|
||||
generated_c_preamble: Optional[str] = Field(
|
||||
None, description="Determines the code that is inserted by default in generated .c files"
|
||||
)
|
||||
generated_s_preamble: Optional[str] = Field(
|
||||
None, description="Determines the code that is inserted by default in generated .s files"
|
||||
)
|
||||
use_o_as_suffix: Optional[bool] = Field(
|
||||
None, description="Determines whether to use .o as the suffix for all binary files?... TODO document"
|
||||
)
|
||||
gp_value: Optional[int] = Field(
|
||||
None, description="the value of the $gp register to correctly calculate offset to %gp_rel relocs"
|
||||
)
|
||||
check_consecutive_segment_types: Optional[bool] = Field(
|
||||
None, description="Checks and errors if there are any non consecutive segment types"
|
||||
)
|
||||
|
||||
# Paths
|
||||
asset_path: Optional[Path] = Field(None, description="")
|
||||
symbol_addrs_paths: Optional[list[Path]] = Field(
|
||||
None,
|
||||
description="""Determines the path to the symbol addresses file(s)
|
||||
A symbol_addrs file is to be updated/curated manually and contains addresses of symbols
|
||||
as well as optional metadata such as rom address, type, and more
|
||||
|
||||
It's possible to use more than one file by supplying a list instead of a string""",
|
||||
)
|
||||
reloc_addrs_paths: Optional[list[Path]] = Field(None, description="")
|
||||
build_path: Optional[Path] = Field(None, description="Determines the path to the project build directory")
|
||||
src_path: Optional[Path] = Field(None, description="Determines the path to the source code directory")
|
||||
asm_path: Optional[Path] = Field(None, description="Determines the path to the asm code directory")
|
||||
data_path: Optional[Path] = Field(None, description="Determines the path to the asm data directory")
|
||||
nonmatchings_path: Optional[Path] = Field(None, description="Determines the path to the asm nonmatchings directory")
|
||||
cache_path: Optional[Path] = Field(
|
||||
None, description="Determines the path to the cache file (used when supplied --use-cache via the CLI)"
|
||||
)
|
||||
hasm_in_src_path: Optional[bool] = Field(
|
||||
None, description="Tells splat to consider `hasm` files to be relative to `src_path` instead of `asm_path`."
|
||||
)
|
||||
create_undefined_funcs_auto: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to create an automatically-generated undefined functions fil. "
|
||||
"This file stores all functions that are referenced in the code but are not defined as seen by splat",
|
||||
)
|
||||
undefined_funcs_auto_path: Optional[Path] = Field(
|
||||
None, description="Determines the path to the undefined_funcs_auto file"
|
||||
)
|
||||
|
||||
create_undefined_syms_auto: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to create an automatically-generated undefined symbols file. "
|
||||
"This file stores all symbols that are referenced in the code but are not defined as seen by splat",
|
||||
)
|
||||
undefined_syms_auto_path: Optional[Path] = Field(
|
||||
None, description="Determines the path to the undefined_symbols_auto file"
|
||||
)
|
||||
|
||||
extensions_path: Optional[Path] = Field(
|
||||
None, description="Determines the path in which to search for custom splat extensions"
|
||||
)
|
||||
|
||||
lib_path: Optional[Path] = Field(
|
||||
None, description="Determines the path to library files that are to be linked into the target binary"
|
||||
)
|
||||
|
||||
# TODO document
|
||||
elf_section_list_path: Optional[Path] = Field(None, description="")
|
||||
|
||||
# Linker script
|
||||
subalign: Optional[int] = Field(
|
||||
None, description="Determines the default subalign value to be specified in the generated linker script"
|
||||
)
|
||||
|
||||
auto_all_sections: Optional[list[str]] = Field(
|
||||
None,
|
||||
description="The following option determines whether to automatically configure the linker script to link "
|
||||
'against specified sections for all "base" (asm/c) files when the yaml doesn\'t have manual configurations '
|
||||
"for these sections.",
|
||||
)
|
||||
ld_script_path: Optional[Path] = Field(
|
||||
None, description="Determines the desired path to the linker script that splat will generate"
|
||||
)
|
||||
ld_symbol_header_path: Optional[Path] = Field(
|
||||
None,
|
||||
description="Determines the desired path to the linker symbol header, which exposes externed definitions "
|
||||
"for all segment ram/rom start/end locations",
|
||||
)
|
||||
ld_discard_section: Optional[bool] = Field(
|
||||
None, description="Determines whether to add a discard section with a wildcard to the linker script"
|
||||
)
|
||||
ld_sections_allowlist: Optional[list[str]] = Field(
|
||||
None,
|
||||
description="A list of sections to preserve during link time. It can be useful to preserve debugging sections",
|
||||
)
|
||||
ld_sections_denylist: Optional[list[str]] = Field(
|
||||
None,
|
||||
description="A list of sections to discard during link time. It can be useful to avoid using the wildcard "
|
||||
"discard. Note that this option does not turn off `ld_discard_section`",
|
||||
)
|
||||
ld_wildcard_sections: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to add wildcards for section linking in the linker script "
|
||||
"(.rodata* for example)",
|
||||
)
|
||||
ld_use_symbolic_vram_addresses: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to use `follows_vram` (segment option) and `vram_symbol` / `follows_classes` "
|
||||
"(vram_class options) to calculate vram addresses in the linker script. If disabled, this uses the plain "
|
||||
"integer values for vram addresses defined in the yaml.",
|
||||
)
|
||||
ld_partial_linking: Optional[bool] = Field(
|
||||
None,
|
||||
description="Change linker script generation to allow partially linking segments. Requires both "
|
||||
"`ld_partial_scripts_path` and `ld_partial_build_segments_path` to be set.",
|
||||
)
|
||||
ld_partial_scripts_path: Optional[Path] = Field(
|
||||
None, description="Folder were each intermediary linker script will be written to."
|
||||
)
|
||||
ld_partial_build_segments_path: Optional[Path] = Field(
|
||||
None, description="Folder where the built partially linked segments will be placed by the build system."
|
||||
)
|
||||
ld_dependencies: Optional[bool] = Field(
|
||||
None,
|
||||
description="Generate a dependency file for every linker script generated. Dependency files will have the "
|
||||
"same path and name as the corresponding linker script, but changing the extension to `.d`. Requires "
|
||||
"`elf_path` to be set.",
|
||||
)
|
||||
ld_legacy_generation: Optional[bool] = Field(
|
||||
None,
|
||||
description="Legacy linker script generation does not impose the section_order specified in the yaml "
|
||||
"options or per-segment options.",
|
||||
)
|
||||
segment_end_before_align: Optional[bool] = Field(
|
||||
None,
|
||||
description="If enabled, the end symbol for each segment will be placed before the alignment directive "
|
||||
"for the segment",
|
||||
)
|
||||
segment_symbols_style: Optional[str] = Field(
|
||||
None,
|
||||
description="Controls the style of the auto-generated segment symbols in the linker script. "
|
||||
"Possible values: Optional[splat, makerom",
|
||||
)
|
||||
ld_rom_start: Optional[int] = Field(
|
||||
None, description="Specifies the starting offset for rom address symbols in the linker script."
|
||||
)
|
||||
ld_fill_value: Optional[int] = Field(
|
||||
None,
|
||||
description="The value passed to the FILL statement on each segment. `None` disables using FILL "
|
||||
"statements on the linker script. Defaults to a fill value of 0.",
|
||||
)
|
||||
ld_bss_is_noload: Optional[bool] = Field(
|
||||
None,
|
||||
description="Allows to control if `bss` sections (and derivative sections) will be put on a `NOLOAD` "
|
||||
"segment on the generated linker script or not.",
|
||||
)
|
||||
ld_align_segment_vram_end: Optional[bool] = Field(
|
||||
None, description="Allows to toggle aligning the `*_VRAM_END` linker symbol for each segment."
|
||||
)
|
||||
ld_align_section_vram_end: Optional[bool] = Field(
|
||||
None, description="Allows to toggle aligning the `*_END` linker symbol for each section of each section."
|
||||
)
|
||||
ld_generate_symbol_per_data_segment: Optional[bool] = Field(
|
||||
None, description="If enabled, the generated linker script will have a linker symbol for each data file"
|
||||
)
|
||||
ld_bss_contains_common: Optional[bool] = Field(
|
||||
None, description="Sets the default option for the `bss_contains_common` attribute of all segments."
|
||||
)
|
||||
|
||||
################################################################################
|
||||
# C file options
|
||||
################################################################################
|
||||
create_c_files: Optional[bool] = Field(
|
||||
None, description="Determines whether to create new c files if they don't exist"
|
||||
)
|
||||
auto_decompile_empty_functions: Optional[bool] = Field(
|
||||
None, description='Determines whether to "auto-decompile" empty functions'
|
||||
)
|
||||
do_c_func_detection: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to detect matched/unmatched functions in existing c files so we can avoid "
|
||||
"creating .s files for already-decompiled functions",
|
||||
)
|
||||
c_newline: Optional[str] = Field(None, description="Determines the newline char(s) to be used in c files")
|
||||
|
||||
################################################################################
|
||||
# (Dis)assembly-related options
|
||||
################################################################################
|
||||
symbol_name_format: Optional[str] = Field(
|
||||
None, description="The following options determine the format that symbols should be named by default"
|
||||
)
|
||||
symbol_name_format_no_rom: Optional[str] = Field(
|
||||
None, description="Same as above but for symbols with no rom address"
|
||||
)
|
||||
find_file_boundaries: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to detect and hint to the user about likely file splits " "when disassembling",
|
||||
)
|
||||
pair_rodata_to_text: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to detect and hint to the user about possible rodata sections "
|
||||
"corresponding to a text section",
|
||||
)
|
||||
migrate_rodata_to_functions: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to attempt to automatically migrate rodata into functions "
|
||||
"(only works in certain circumstances)",
|
||||
)
|
||||
asm_inc_header: Optional[str] = Field(
|
||||
None, description="Determines the header to be used in every asm file that's included from c files"
|
||||
)
|
||||
asm_function_macro: Optional[str] = Field(
|
||||
None, description="Determines the macro used to declare functions in asm files"
|
||||
)
|
||||
asm_function_alt_macro: Optional[str] = Field(
|
||||
None,
|
||||
description="Determines the macro used to declare symbols in the middle of functions in asm files "
|
||||
"(which may be alternative entries)",
|
||||
)
|
||||
asm_jtbl_label_macro: Optional[str] = Field(
|
||||
None, description="Determines the macro used to declare jumptable labels in asm files"
|
||||
)
|
||||
asm_data_macro: Optional[str] = Field(
|
||||
None, description="Determines the macro used to declare data symbols in asm files"
|
||||
)
|
||||
asm_end_label: Optional[str] = Field(
|
||||
None, description="Determines the macro used at the end of a function, such as endlabel or .end"
|
||||
)
|
||||
asm_emit_size_directive: Optional[bool] = Field(
|
||||
None, description="Toggles the .size directive emitted by the disassembler"
|
||||
)
|
||||
include_macro_inc: Optional[bool] = Field(
|
||||
None, description="Determines including the macro.inc file on non-migrated rodata variables"
|
||||
)
|
||||
mnemonic_ljust: Optional[int] = Field(
|
||||
None, description="Determines the number of characters to left align before the TODO finish documenting"
|
||||
)
|
||||
rom_address_padding: Optional[bool] = Field(None, description="Determines whether to pad the rom address")
|
||||
mips_abi_gpr: Optional[str] = Field(
|
||||
None, description="Determines which ABI names to use for general purpose registers"
|
||||
)
|
||||
mips_abi_float_regs: Optional[str] = Field(
|
||||
None,
|
||||
description="""Determines which ABI names to use for floating point registers
|
||||
Valid values: 'numeric', 'o32', 'n32', 'n64'
|
||||
o32 is highly recommended, as it provides logically named registers for floating point instructions
|
||||
For more info, see https://gist.github.com/EllipticEllipsis/27eef11205c7a59d8ea85632bc49224d""",
|
||||
)
|
||||
named_regs_for_c_funcs: Optional[bool] = Field(
|
||||
None, description="Determines whether functions inside c files should have named registers"
|
||||
)
|
||||
add_set_gp_64: Optional[bool] = Field(None, description='Determines whether to add ".set gp=64" to asm/hasm files')
|
||||
create_asm_dependencies: Optional[bool] = Field(
|
||||
None,
|
||||
description="Generate .asmproc.d dependency files for each C file which still reference functions "
|
||||
"in assembly files",
|
||||
)
|
||||
string_encoding: Optional[str] = Field(
|
||||
None, description="Global option for rodata string encoding. This can be overridden per segment"
|
||||
)
|
||||
data_string_encoding: Optional[str] = Field(
|
||||
None, description="Global option for data string encoding. This can be overridden per segment"
|
||||
)
|
||||
rodata_string_guesser_level: Optional[int] = Field(
|
||||
None, description="Global option for the rodata string guesser. 0 disables the guesser completely."
|
||||
)
|
||||
data_string_guesser_level: Optional[int] = Field(
|
||||
None, description="Global option for the data string guesser. 0 disables the guesser completely."
|
||||
)
|
||||
allow_data_addends: Optional[bool] = Field(
|
||||
None,
|
||||
description="Global option for allowing data symbols using addends on symbol references. "
|
||||
"It can be overridden per symbol",
|
||||
)
|
||||
disasm_unknown: Optional[bool] = Field(
|
||||
None,
|
||||
description="Tells the disassembler to try disassembling functions with unknown instructions instead of "
|
||||
"falling back to disassembling as raw data",
|
||||
)
|
||||
detect_redundant_function_end: Optional[bool] = Field(
|
||||
None,
|
||||
description="Tries to detect redundant and unreferenced functions ends and merge them together. "
|
||||
"This option is ignored if the compiler is not set to IDO.",
|
||||
)
|
||||
disassemble_all: Optional[bool] = Field(
|
||||
None, description="Don't skip disassembling already matched functions and migrated sections"
|
||||
)
|
||||
|
||||
################################################################################
|
||||
# N64-specific options
|
||||
################################################################################
|
||||
header_encoding: Optional[str] = Field(None, description="Determines the encoding of the header")
|
||||
gfx_ucode: Optional[str] = Field(
|
||||
None,
|
||||
description="""Determines the type gfx ucode (used by gfx segments)
|
||||
Valid options are ['f3d', 'f3db', 'f3dex', 'f3dexb', 'f3dex2']""",
|
||||
)
|
||||
libultra_symbols: Optional[bool] = Field(
|
||||
None,
|
||||
description="Use named libultra symbols by default. Those will need to be added to a linker script "
|
||||
"manually by the user",
|
||||
)
|
||||
ique_symbols: Optional[bool] = Field(
|
||||
None,
|
||||
description="Use named libultra symbols by default. Those will need to be added to a linker script "
|
||||
"manually by the user",
|
||||
)
|
||||
hardware_regs: Optional[bool] = Field(
|
||||
None,
|
||||
description="Use named hardware register symbols by default. Those will need to be added to a linker script "
|
||||
"manually by the user",
|
||||
)
|
||||
image_type_in_extension: Optional[bool] = Field(
|
||||
None, description="Append the image type to the output file extension"
|
||||
)
|
||||
|
||||
################################################################################
|
||||
# Compiler-specific options
|
||||
################################################################################
|
||||
use_legacy_include_asm: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to use a legacy INCLUDE_ASM macro "
|
||||
"format in c files only applies to GCC/SN64",
|
||||
)
|
||||
|
||||
|
||||
class VramClass(BaseModel):
|
||||
name: str = Field(..., description="")
|
||||
vram: Optional[int] = Field(None, description="")
|
||||
|
||||
|
||||
class DictSegment(BaseModel):
|
||||
start: int = Field(None, description="")
|
||||
rom_start: Optional[int] = Field(None, description="")
|
||||
rom_start: Optional[int] = Field(None, description="")
|
||||
rom_end: Optional[int] = Field(None, description="")
|
||||
type: str = Field(..., description="")
|
||||
name: str = Field(..., description="")
|
||||
vram: Optional[int] = Field(None, description="")
|
||||
vram_start: Optional[int] = Field(None, description="")
|
||||
vram_symbol: Optional[str] = Field(None, description="")
|
||||
vram_class: Optional[VramClass] = Field(None, description="")
|
||||
follows_vram: Optional[str] = Field(None, description="")
|
||||
align: Optional[int] = Field(None, description="")
|
||||
subalign: Optional[int] = Field(None, description="")
|
||||
section_order: Optional[list[str]] = Field(None, description="")
|
||||
subsegments: Optional[list[Segment]] = Field(None, description="")
|
||||
bss_size: Optional[int] = Field(None, description="")
|
||||
args: Optional[list[str]] = Field(None, description="")
|
||||
|
||||
|
||||
class ListSegment(RootModel[tuple[int, str, str] | tuple[int, str] | tuple[int]]): ...
|
||||
|
||||
|
||||
Segment = DictSegment | ListSegment
|
||||
|
||||
|
||||
class Config(BaseModel):
|
||||
name: str
|
||||
sha1: str
|
||||
options: SplatOpts
|
||||
segments: list[Segment]
|
||||
|
||||
|
||||
def main():
|
||||
import json
|
||||
|
||||
with open("../../.vscode/schema/splat_config.schema.json", "w") as fh:
|
||||
fh.write(json.dumps(Config.model_json_schema(), indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user