inital commit
This commit is contained in:
Executable
+265
@@ -0,0 +1,265 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
# SPDX-FileCopyrightText: Copyright 2025 karas84 (https://github.com/karas84)
|
||||
# SPDX-License-Identifier: MIT
|
||||
#
|
||||
# This script inserts source line debug information (.loc directives) into assembly files
|
||||
# using data extracted from a JSON "stdump" file generated by the ccc tool (version 2.1),
|
||||
# available at https://github.com/chaoticgd/ccc.
|
||||
#
|
||||
# Original concept by Mc-muffin (https://github.com/Mc-muffin).
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import io
|
||||
import sys
|
||||
import json
|
||||
import argparse
|
||||
|
||||
from typing import Callable, Protocol, cast
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
class STDUMPJson:
|
||||
def __init__(self, json_path: Path):
|
||||
self._json = json.loads(json_path.read_text())
|
||||
self._line_cache = self._build_line_cache()
|
||||
self._validate_sub_source_files()
|
||||
|
||||
def _build_line_cache(self):
|
||||
line_cache: dict[int, int] = {}
|
||||
functions = self._json["functions"]
|
||||
|
||||
for fun in functions:
|
||||
if "line_numbers" not in fun:
|
||||
continue
|
||||
|
||||
for addr, num in fun["line_numbers"]:
|
||||
# there may be more lines for the same address. Ghidra seems to
|
||||
# keep the last one, so that's what we are also doing
|
||||
line_cache[addr] = num
|
||||
|
||||
return line_cache
|
||||
|
||||
def _validate_sub_source_files(self):
|
||||
functions = self._json["functions"]
|
||||
for fun in functions:
|
||||
if (line_numbers := fun.get("line_numbers")) is None:
|
||||
continue
|
||||
|
||||
if (sub_source_files := fun.get("sub_source_files")) is None:
|
||||
continue
|
||||
|
||||
line_numbers = cast(list[tuple[int, int]], line_numbers)
|
||||
sub_source_files = cast(list[tuple[int, str]], sub_source_files)
|
||||
|
||||
# # some functions may have "unterminated" inlines, but they usually have just one line number
|
||||
# if len(sub_source_files) % 2 != 0: # and len(line_numbers) != 1:
|
||||
# print(fun["name"], len(line_numbers))
|
||||
|
||||
# for addr, source in sub_source_files:
|
||||
|
||||
def find_function(self, name: str):
|
||||
functions = self._json["functions"]
|
||||
fun = next((f for f in functions if f["name"] == name), None)
|
||||
return fun
|
||||
|
||||
def get_line(self, addr: int):
|
||||
return self._line_cache.get(addr)
|
||||
|
||||
|
||||
def make_range_checker(lst: list[tuple[int, str]], delimiter: str) -> Callable[[int], bool]:
|
||||
"""
|
||||
Given a sorted list of (number, label) where number increases,
|
||||
build a checker that returns True if x is in a valid interval.
|
||||
|
||||
Interpretation:
|
||||
- Each entry (n, label) marks the interval starting at n and going
|
||||
up to the next entry's n (exclusive). The last entry's interval
|
||||
goes to +inf.
|
||||
- An interval starting at n is valid iff label == delimiter.
|
||||
- If the first entry's label != delimiter, everything before the first n is valid.
|
||||
"""
|
||||
if not lst:
|
||||
# no markers -> everything valid
|
||||
return lambda x: True
|
||||
|
||||
# ensure sorted by number
|
||||
lst_sorted = sorted(lst, key=lambda t: t[0])
|
||||
assert lst == lst_sorted
|
||||
|
||||
nums = [t[0] for t in lst_sorted]
|
||||
labels = [t[1] for t in lst_sorted]
|
||||
|
||||
# Precompute intervals as (start, end_exclusive, is_valid)
|
||||
intervals: list[tuple[int, int | None, bool]] = []
|
||||
n_items = len(nums)
|
||||
|
||||
for i in range(n_items):
|
||||
start = nums[i]
|
||||
end_exclusive: int | None
|
||||
|
||||
if i + 1 < n_items:
|
||||
end_exclusive = nums[i + 1]
|
||||
else:
|
||||
end_exclusive = None # means to +inf
|
||||
|
||||
is_valid = labels[i] == delimiter
|
||||
intervals.append((start, end_exclusive, is_valid))
|
||||
|
||||
first_before_is_valid = labels[0] != delimiter
|
||||
|
||||
def is_valid_fn(x: int) -> bool:
|
||||
# before first number
|
||||
if x < nums[0]:
|
||||
return first_before_is_valid
|
||||
|
||||
# find the interval that contains x
|
||||
for start, end_exclusive, valid_flag in intervals:
|
||||
if end_exclusive is None:
|
||||
if x >= start:
|
||||
return valid_flag
|
||||
else:
|
||||
if start <= x < end_exclusive:
|
||||
return valid_flag
|
||||
|
||||
# fallback (shouldn't happen)
|
||||
return False
|
||||
|
||||
return is_valid_fn
|
||||
|
||||
|
||||
def is_always_valid_fn(addr: int):
|
||||
return True
|
||||
|
||||
|
||||
def add_lines_to_asm(
|
||||
asm_path: Path,
|
||||
stdump_json_path: Path,
|
||||
fun_start_offset: int,
|
||||
keep_original_numbers: bool,
|
||||
asm_out: Path | None,
|
||||
):
|
||||
stdump_json = STDUMPJson(stdump_json_path)
|
||||
|
||||
function_name = asm_path.stem
|
||||
|
||||
if (fun := stdump_json.find_function(function_name)) is None:
|
||||
raise RuntimeError(f"Cannot find function '{function_name}' in ccc's JSON")
|
||||
|
||||
sub_source_files = fun.get("sub_source_files")
|
||||
# print(len(sub_source_files) if sub_source_files is not None else None)
|
||||
|
||||
asm_lines = asm_path.read_text().splitlines()
|
||||
re_instr = re.compile(r"^\s*\/\* [A-Z0-9]+ ([A-Z0-9]{8}) [A-Z0-9]{8} \*\/ .*$")
|
||||
line_dict: dict[int, int] = {}
|
||||
|
||||
relative_path: str = fun["relative_path"]
|
||||
non_func_addrs: list[int] = []
|
||||
|
||||
if sub_source_files:
|
||||
checker = make_range_checker(sub_source_files, relative_path)
|
||||
else:
|
||||
checker = is_always_valid_fn
|
||||
|
||||
start_line_num: int = sys.maxsize
|
||||
|
||||
for line in asm_lines:
|
||||
if m := re_instr.match(line):
|
||||
instr_addr = int(m.group(1), 16)
|
||||
line_num = stdump_json.get_line(instr_addr)
|
||||
|
||||
if line_num:
|
||||
line_dict[instr_addr] = line_num
|
||||
|
||||
if not checker(instr_addr):
|
||||
non_func_addrs.append(instr_addr)
|
||||
elif line_num:
|
||||
start_line_num = min(start_line_num, line_num)
|
||||
|
||||
if start_line_num == sys.maxsize:
|
||||
start_line_num = 1
|
||||
|
||||
min_line_num = 0 if keep_original_numbers else start_line_num - 1
|
||||
|
||||
new_asm_lines: list[str] = []
|
||||
asm_n: int = 0
|
||||
|
||||
for line in asm_lines:
|
||||
if m := re_instr.match(line):
|
||||
asm_n += 1
|
||||
|
||||
instr_addr = int(m.group(1), 16)
|
||||
line_num = line_dict.get(instr_addr)
|
||||
|
||||
if line_num is not None and instr_addr in non_func_addrs:
|
||||
new_asm_lines.append(f" .loc 1 {line_num} # inline")
|
||||
new_asm_lines.append(line)
|
||||
continue
|
||||
|
||||
if asm_n == 1 and line_num is None:
|
||||
# sometimes we don't have a number for the first line of assembly,
|
||||
# so we reuse the first known line among the ones we have
|
||||
line_num = start_line_num
|
||||
|
||||
if line_num is not None:
|
||||
new_line_num = (line_num - min_line_num) + fun_start_offset
|
||||
new_asm_lines.append(f" .loc 1 {new_line_num} ")
|
||||
|
||||
new_asm_lines.append(line)
|
||||
|
||||
stream = io.StringIO()
|
||||
stream.write(""".section .debug
|
||||
.previous
|
||||
.text
|
||||
.file 1 "source.c"
|
||||
|
||||
.set noat
|
||||
.set noreorder
|
||||
|
||||
""")
|
||||
stream.write("\n".join(new_asm_lines))
|
||||
|
||||
if asm_out is not None:
|
||||
asm_out.write_text(stream.getvalue())
|
||||
else:
|
||||
print(stream.getvalue())
|
||||
|
||||
|
||||
def main():
|
||||
class ArgProtocol(Protocol):
|
||||
asm_path: Path
|
||||
stdump_json_path: Path
|
||||
asm_out: Path | None
|
||||
offset: int
|
||||
keep_original: bool
|
||||
|
||||
parser = argparse.ArgumentParser(description="Add line debug info to assembly files")
|
||||
parser.add_argument("--asm-path", required=True, type=Path, help="Path to the assembly file to add lines to")
|
||||
parser.add_argument("--stdump-json-path", required=True, type=Path, help="Path to ccc's json stdump")
|
||||
parser.add_argument("--asm-out", type=Path, required=False, help="Path to output asm (defaults to stdout)")
|
||||
parser.add_argument(
|
||||
"--offset", type=int, required=False, default=0, help="Offset to apply to line numbers (default: 0)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--keep-original",
|
||||
action="store_true",
|
||||
help="Don't start line numbers from 1 (+ offset) but keep the original line numbers (+offset) instead",
|
||||
)
|
||||
|
||||
args = cast(ArgProtocol, parser.parse_args())
|
||||
|
||||
fun_start_num = args.offset
|
||||
|
||||
add_lines_to_asm(
|
||||
args.asm_path,
|
||||
args.stdump_json_path,
|
||||
fun_start_num,
|
||||
args.keep_original,
|
||||
args.asm_out,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,351 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import pydantic
|
||||
|
||||
from typing import Optional, Literal, Any
|
||||
|
||||
|
||||
class AddressRange(pydantic.BaseModel):
|
||||
low: int
|
||||
high: int
|
||||
|
||||
|
||||
class ValueType(pydantic.BaseModel):
|
||||
descriptor: Optional[str] = None
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
value_type: Optional[ValueType] = None
|
||||
return_type: Optional[ReturnType] = None
|
||||
modifier: Optional[str] = None
|
||||
vtable_index: Optional[int] = None
|
||||
is_constructor: Optional[bool] = None
|
||||
element_type: Optional[ElementType] = None
|
||||
element_count: Optional[int] = None
|
||||
|
||||
|
||||
class DeduplicatedTypeValueType(pydantic.BaseModel):
|
||||
descriptor: Literal["function_type", "type_name"]
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
value_type: Optional[ValueType] = None
|
||||
return_type: Optional[ReturnType] = None
|
||||
modifier: Optional[str] = None
|
||||
vtable_index: Optional[int] = None
|
||||
is_constructor: Optional[bool] = None
|
||||
element_type: Optional[ElementType] = None
|
||||
element_count: Optional[int] = None
|
||||
|
||||
|
||||
class ParameterType(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
value_type: Optional[ValueType] = None
|
||||
|
||||
|
||||
class Parameter(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
name: str
|
||||
class_: str = pydantic.Field(..., alias="class")
|
||||
storage: Storage
|
||||
block_low: int
|
||||
block_high: int
|
||||
type: ParameterType
|
||||
|
||||
|
||||
class ReturnType(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
value_type: Optional[ValueType] = None
|
||||
|
||||
|
||||
class FunctionType(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
parameters: list[Parameter]
|
||||
modifier: str
|
||||
vtable_index: int
|
||||
is_constructor: bool
|
||||
return_type: Optional[ReturnType] = None
|
||||
|
||||
|
||||
class Storage(pydantic.BaseModel):
|
||||
type: str
|
||||
register_: Optional[str] = pydantic.Field(None, alias="register")
|
||||
register_class: Optional[str] = None
|
||||
dbx_register_number: Optional[int] = None
|
||||
register_index: Optional[int] = None
|
||||
is_by_reference: Optional[bool] = None
|
||||
stack_offset: Optional[int] = None
|
||||
global_location: Optional[str] = None
|
||||
global_address: Optional[int] = None
|
||||
|
||||
|
||||
class Constant(pydantic.BaseModel):
|
||||
value: int
|
||||
name: str
|
||||
|
||||
|
||||
class ElementType(pydantic.BaseModel):
|
||||
descriptor: Literal["array", "pointer", "type_name", "enum"]
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
element_type: Optional[ElementType] = None
|
||||
element_count: Optional[int] = None
|
||||
value_type: Optional[ValueType] = None
|
||||
constants: Optional[list[Constant]] = None
|
||||
|
||||
def parsed_size(self) -> tuple[int, str]:
|
||||
if self.descriptor == "pointer":
|
||||
return 1, "pointer"
|
||||
|
||||
elif self.descriptor == "array":
|
||||
assert self.element_type
|
||||
assert self.element_count is not None
|
||||
if self.element_count == 0:
|
||||
# implicit size array
|
||||
_, type_name = self.element_type.parsed_size()
|
||||
return 0, type_name
|
||||
n, type_name = self.element_type.parsed_size()
|
||||
return self.element_count * n, type_name
|
||||
|
||||
elif self.descriptor == "enum":
|
||||
return 1, "enum"
|
||||
|
||||
else: # "type_name"
|
||||
assert self.type_name
|
||||
return 1, self.type_name
|
||||
|
||||
|
||||
class Local(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
name: str
|
||||
class_: str = pydantic.Field(..., alias="class")
|
||||
storage: Storage
|
||||
block_low: int
|
||||
block_high: int
|
||||
type: ElementType
|
||||
storage_class: Optional[str] = None
|
||||
|
||||
@property
|
||||
def parsed_size(self) -> tuple[int, str]:
|
||||
return self.type.parsed_size()
|
||||
|
||||
|
||||
class SubSourceFile(pydantic.BaseModel):
|
||||
address: int
|
||||
path: str
|
||||
|
||||
|
||||
class Function(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
name: str
|
||||
address_range: AddressRange
|
||||
type: FunctionType
|
||||
locals: list[Local]
|
||||
line_numbers: list[list[int]]
|
||||
sub_source_files: list[SubSourceFile]
|
||||
storage_class: Optional[str] = None
|
||||
relative_path: Optional[str] = None
|
||||
|
||||
|
||||
class Global(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
name: str
|
||||
class_: str = pydantic.Field(..., alias="class")
|
||||
storage: Storage
|
||||
block_low: int
|
||||
block_high: int
|
||||
type: ElementType
|
||||
storage_class: Optional[str] = None
|
||||
|
||||
@property
|
||||
def parsed_size(self) -> tuple[int, str]:
|
||||
return self.type.parsed_size()
|
||||
|
||||
|
||||
class File(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
path: str
|
||||
relative_path: str
|
||||
text_address: int
|
||||
types: list[Any]
|
||||
functions: list[Function]
|
||||
globals: list[Global]
|
||||
stabs_type_number_to_deduplicated_type_index: dict[str, int]
|
||||
|
||||
|
||||
class UnderlyingType(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
source: str
|
||||
type_name: str
|
||||
referenced_file_index: int
|
||||
referenced_stabs_type_number: int
|
||||
|
||||
|
||||
class Field(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
name: str
|
||||
relative_offset_bytes: int
|
||||
absolute_offset_bytes: int
|
||||
size_bits: int
|
||||
bitfield_offset_bits: Optional[int] = None
|
||||
underlying_type: Optional[UnderlyingType] = None
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
value_type: Optional[ValueType] = None
|
||||
base_classes: Optional[list[Any]] = None
|
||||
fields: Optional[list[Field]] = None
|
||||
member_functions: Optional[list[Any]] = None
|
||||
element_type: Optional[ElementType] = None
|
||||
element_count: Optional[int] = None
|
||||
|
||||
|
||||
class FieldModel(pydantic.BaseModel):
|
||||
descriptor: str
|
||||
name: str
|
||||
relative_offset_bytes: int
|
||||
absolute_offset_bytes: int
|
||||
size_bits: int
|
||||
value_type: Optional[ValueType] = None
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
element_type: Optional[ElementType] = None
|
||||
element_count: Optional[int] = None
|
||||
bitfield_offset_bits: Optional[int] = None
|
||||
underlying_type: Optional[UnderlyingType] = None
|
||||
fields: Optional[list[Field]] = None
|
||||
member_functions: Optional[list[Any]] = None
|
||||
base_classes: Optional[list[Any]] = None
|
||||
|
||||
|
||||
class DeduplicatedType(pydantic.BaseModel):
|
||||
descriptor: Literal["array", "builtin", "enum", "pointer", "struct", "type_name", "union"]
|
||||
name: Optional[str] = None
|
||||
storage_class: Optional[Literal["typedef"]] = None
|
||||
stabs_type_number: int
|
||||
files: list[int]
|
||||
class_: Optional[str] = pydantic.Field(None, alias="class")
|
||||
size_bits: Optional[int] = None
|
||||
base_classes: Optional[list[Any]] = None
|
||||
fields: Optional[list[FieldModel]] = None
|
||||
member_functions: Optional[list[Any]] = None
|
||||
source: Optional[str] = None
|
||||
type_name: Optional[str] = None
|
||||
referenced_file_index: Optional[int] = None
|
||||
referenced_stabs_type_number: Optional[int] = None
|
||||
value_type: Optional[DeduplicatedTypeValueType] = None
|
||||
conflict: Optional[bool] = None
|
||||
element_type: Optional[ElementType] = None
|
||||
element_count: Optional[int] = None
|
||||
constants: Optional[list[Constant]] = None
|
||||
|
||||
|
||||
class CCCJSONv7Model(pydantic.BaseModel):
|
||||
version: Literal[7]
|
||||
files: list[File]
|
||||
deduplicated_types: list[DeduplicatedType]
|
||||
|
||||
|
||||
# def test(stdump_json_path: str):
|
||||
# import json
|
||||
#
|
||||
# with open(stdump_json_path, mode="r") as fh:
|
||||
# json_data = fh.read()
|
||||
#
|
||||
# model = CCCJSONv7Model.model_validate_json(json_data)
|
||||
#
|
||||
# type_map: dict[str, int] = {}
|
||||
#
|
||||
# for n, dt in enumerate(model.deduplicated_types):
|
||||
# if dt.descriptor == "builtin":
|
||||
# assert dt.name and dt.name not in type_map
|
||||
# assert dt.class_ is not None
|
||||
# size_bits = int(dt.class_.split("-", maxsplit=1)[0])
|
||||
# assert size_bits % 8 == 0
|
||||
# type_map[dt.name] = size_bits // 8
|
||||
#
|
||||
# elif dt.descriptor == "type_name":
|
||||
# assert dt.name is not None
|
||||
# assert dt.type_name is not None
|
||||
# if dt.name in type_map:
|
||||
# if dt.size_bits:
|
||||
# assert type_map[dt.name] == dt.size_bits // 8
|
||||
# if dt.name == dt.type_name == "void":
|
||||
# type_map["void"] = type_map["int"]
|
||||
# continue
|
||||
# assert dt.type_name in type_map, (n, dt.type_name)
|
||||
# type_map[dt.name] = type_map[dt.type_name]
|
||||
#
|
||||
# elif dt.descriptor == "pointer":
|
||||
# assert dt.name is not None
|
||||
# assert dt.value_type is not None
|
||||
# if dt.value_type.descriptor == "function_type":
|
||||
# assert dt.name not in type_map
|
||||
# type_map[dt.name] = type_map["int"]
|
||||
# elif dt.value_type.descriptor == "type_name":
|
||||
# assert dt.value_type.type_name
|
||||
# assert dt.value_type.type_name in type_map
|
||||
# type_map[dt.name] = type_map[dt.value_type.type_name]
|
||||
#
|
||||
# elif dt.descriptor == "struct":
|
||||
# assert dt.name is not None
|
||||
# if not dt.conflict:
|
||||
# assert dt.name not in type_map, n
|
||||
# else:
|
||||
# if dt.name in type_map:
|
||||
# continue
|
||||
# assert dt.size_bits is not None
|
||||
# assert dt.size_bits % 8 == 0
|
||||
# type_map[dt.name] = dt.size_bits // 8
|
||||
#
|
||||
# elif dt.descriptor == "array":
|
||||
# assert dt.name is not None
|
||||
# element_count = 1
|
||||
# element_type = dt.element_type
|
||||
# type_name = None
|
||||
# while element_type:
|
||||
# if element_type.element_count is not None:
|
||||
# element_count *= element_type.element_count
|
||||
# if element_type.type_name is not None:
|
||||
# type_name = element_type.type_name
|
||||
# assert type_name in type_map
|
||||
# element_count *= type_map[type_name]
|
||||
# element_type = element_type.element_type
|
||||
# assert type_name and type_name in type_map, n
|
||||
# type_map[dt.name] = type_map[type_name] * element_count
|
||||
#
|
||||
# elif dt.descriptor == "enum":
|
||||
# if dt.name:
|
||||
# if not dt.conflict:
|
||||
# assert dt.name not in type_map, n
|
||||
# type_map[dt.name] = type_map["int"]
|
||||
#
|
||||
# elif dt.descriptor == "union":
|
||||
# assert dt.name
|
||||
# assert dt.size_bits
|
||||
# assert dt.size_bits % 8 == 0
|
||||
# type_map[dt.name] = dt.size_bits // 8
|
||||
#
|
||||
# else:
|
||||
# assert False, f"unknown {n}"
|
||||
#
|
||||
# print(json.dumps(type_map, indent=2))
|
||||
|
||||
|
||||
# if __name__ == "__main__":
|
||||
# test(path-to-stdump-json)
|
||||
@@ -0,0 +1,322 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import sys
|
||||
import math
|
||||
import ctypes
|
||||
import numpy as np
|
||||
import numpy.typing as npt
|
||||
from collections.abc import Sized, Sequence
|
||||
from typing import Type, TypeVar, Any, cast, TextIO, BinaryIO, Optional
|
||||
from ctypes import LittleEndianStructure, Union, Array, c_uint32
|
||||
|
||||
_G = TypeVar("_G")
|
||||
|
||||
CTypeType = (
|
||||
type[ctypes.c_int8]
|
||||
| type[ctypes.c_uint8]
|
||||
| type[ctypes.c_int16]
|
||||
| type[ctypes.c_uint16]
|
||||
| type[ctypes.c_int32]
|
||||
| type[ctypes.c_uint32]
|
||||
| type[ctypes.c_float]
|
||||
| type[ctypes.c_double]
|
||||
)
|
||||
|
||||
ctypes_types: dict[str, CTypeType] = {
|
||||
"char": ctypes.c_int8,
|
||||
"u_char": ctypes.c_uint8,
|
||||
"short": ctypes.c_int16,
|
||||
"u_short": ctypes.c_uint16,
|
||||
"int": ctypes.c_int32,
|
||||
"u_int": ctypes.c_uint32,
|
||||
"float": ctypes.c_float,
|
||||
"double": ctypes.c_double,
|
||||
}
|
||||
|
||||
|
||||
class c_addr(c_uint32):
|
||||
def __str__(self):
|
||||
# if self.value == 0:
|
||||
# return "NULL"
|
||||
|
||||
# return f"0x{self.value:08x}"
|
||||
return f"0x{self.value:x}"
|
||||
|
||||
|
||||
class c_str(c_uint32):
|
||||
def to_str(self, elf: BinaryIO):
|
||||
if self.value == 0:
|
||||
return "NULL"
|
||||
elf.seek(self.value)
|
||||
buf = io.BytesIO()
|
||||
while (c := elf.read(1)) != b"\0":
|
||||
buf.write(c)
|
||||
return '"' + buf.getvalue().decode("ASCII") + '"'
|
||||
|
||||
|
||||
class c_addr_ptr(c_uint32):
|
||||
_addresses: dict[int, str] | None = None
|
||||
|
||||
@classmethod
|
||||
def set_addresses(cls, addresses: dict[int, str] | None):
|
||||
cls._addresses = addresses
|
||||
|
||||
def __str__(self):
|
||||
if self.value == 0:
|
||||
return "NULL"
|
||||
|
||||
if self._addresses and self.value in self._addresses:
|
||||
return self._addresses[self.value]
|
||||
|
||||
print(f"warning: no address for pointer 0x{self.value:x}")
|
||||
|
||||
return f"0x{self.value:x}"
|
||||
|
||||
|
||||
def print_arr(arr: npt.NDArray[np.int_], lst: list[Any], file: TextIO):
|
||||
if arr.ndim > 1:
|
||||
for x in arr:
|
||||
file.write("{")
|
||||
print_arr(x, lst, file)
|
||||
file.write("},")
|
||||
else:
|
||||
file.write(",".join(str(lst[i]) for i in arr))
|
||||
|
||||
|
||||
def print_carr(arr: Sequence[Any], file: TextIO):
|
||||
if hasattr(arr[0], "_length_"):
|
||||
for a in arr:
|
||||
file.write("{")
|
||||
print_carr(a, file)
|
||||
file.write("},")
|
||||
else:
|
||||
v = str(str([x for x in arr])).replace("[", "").replace("]", "")
|
||||
file.write(v)
|
||||
|
||||
|
||||
def chunks(lst: Sequence[_G], n: int):
|
||||
"""Yield successive n-sized chunks from lst."""
|
||||
for i in range(0, len(lst), n):
|
||||
yield lst[i : i + n]
|
||||
|
||||
|
||||
def format_array(lst: Sequence[_G], dims: Sequence[int], file: TextIO):
|
||||
if len(dims) > 1:
|
||||
for ll in chunks(lst, len(lst) // dims[0]):
|
||||
file.write("{")
|
||||
format_array(ll, dims[1:], file)
|
||||
file.write("},")
|
||||
else:
|
||||
for n, x in enumerate(lst):
|
||||
sep = ", " if n < len(lst) - 1 else ""
|
||||
file.write(f"{x}{sep}")
|
||||
|
||||
|
||||
def resolve_annotations(namespace: dict[str, Any], annotations: dict[str, Any]):
|
||||
module = sys.modules.get(namespace.get("__module__", ""))
|
||||
globals_ = vars(module) if module else {}
|
||||
resolved: dict[str, Any] = {}
|
||||
|
||||
for k, v in annotations.items():
|
||||
if isinstance(v, str):
|
||||
try:
|
||||
resolved[k] = eval(v, globals_, namespace)
|
||||
except Exception:
|
||||
resolved[k] = v # keep as string if can't resolve
|
||||
else:
|
||||
resolved[k] = v
|
||||
|
||||
return resolved
|
||||
|
||||
|
||||
class LittleEndianStructureFieldsFromTypeHints(type(LittleEndianStructure)): # pyright: ignore
|
||||
def __new__(
|
||||
cls: Type[type],
|
||||
name: str,
|
||||
bases: tuple[type, ...],
|
||||
namespace: dict[str, Any],
|
||||
/,
|
||||
*,
|
||||
align: Optional[int] = None,
|
||||
pack: Optional[int] = None,
|
||||
) -> LittleEndianStructureFieldsFromTypeHints:
|
||||
annotations = namespace.get("__annotations__", {})
|
||||
annotations = resolve_annotations(namespace, annotations)
|
||||
if "__elf__" in annotations:
|
||||
annotations.pop("__elf__")
|
||||
if align is not None:
|
||||
namespace["_align_"] = align
|
||||
if pack is not None:
|
||||
namespace["_pack_"] = pack
|
||||
namespace["_layout_"] = "ms"
|
||||
if fields := list(annotations.items()):
|
||||
namespace["_fields_"] = fields
|
||||
return type(LittleEndianStructure).__new__(cls, name, bases, namespace) # pyright: ignore
|
||||
|
||||
|
||||
class CStructure(LittleEndianStructure, metaclass=LittleEndianStructureFieldsFromTypeHints):
|
||||
__elf__: BinaryIO
|
||||
|
||||
@classmethod
|
||||
def sizeof(cls) -> int:
|
||||
align = getattr(cls, "_align_", 0)
|
||||
c_size = ctypes.sizeof(cls)
|
||||
if align > 0:
|
||||
rem = c_size % align
|
||||
if rem:
|
||||
rem = align - rem
|
||||
return c_size + rem
|
||||
|
||||
return c_size
|
||||
|
||||
@classmethod
|
||||
def parse(cls, data: bytes):
|
||||
cstruct_size = sizeof(cls)
|
||||
assert len(data) % cstruct_size == 0, f"{len(data)}, {cstruct_size}"
|
||||
cstruct_num = len(data) // cstruct_size
|
||||
stream = io.BytesIO(data)
|
||||
cstructs = [cls.from_buffer_copy(stream.read(cstruct_size)) for _ in range(cstruct_num)]
|
||||
return cstructs
|
||||
|
||||
@classmethod
|
||||
def dumps(
|
||||
cls,
|
||||
name: str,
|
||||
data: bytes,
|
||||
numel: int | list[int],
|
||||
static: bool = False,
|
||||
nosize: bool = False,
|
||||
noarray: bool = False,
|
||||
):
|
||||
cstructs = cls.parse(data)
|
||||
stream = io.StringIO()
|
||||
if static:
|
||||
stream.write("static ")
|
||||
stream.write(f"{cls.__name__} {name}") # pyright: ignore
|
||||
assert not (noarray and isinstance(numel, list))
|
||||
if not noarray:
|
||||
if isinstance(numel, int):
|
||||
assert len(cstructs) == numel, (len(cstructs), numel)
|
||||
numel_str = f"{len(cstructs)}" if not nosize else ""
|
||||
stream.write(f"[{numel_str}]")
|
||||
else:
|
||||
numel_str = "".join([f"[{num if n > 0 or not nosize else ''}]" for n, num in enumerate(numel)])
|
||||
stream.write(numel_str)
|
||||
tot_numel = max(1, numel) if isinstance(numel, int) else math.prod(numel)
|
||||
# nmdim = 1 if isinstance(numel, int) else len(numel)
|
||||
assert len(cstructs) == tot_numel, (len(cstructs), tot_numel)
|
||||
|
||||
stream.write(" = ")
|
||||
if not noarray:
|
||||
stream.write("{\n")
|
||||
if isinstance(numel, int):
|
||||
for n, s in enumerate(cstructs):
|
||||
is_last = n == len(cstructs) - 1
|
||||
if is_last and noarray:
|
||||
stream.write(f"{s}\n")
|
||||
else:
|
||||
stream.write(f"{s},\n")
|
||||
|
||||
else:
|
||||
idxs = np.arange(tot_numel).reshape(numel)
|
||||
print_arr(idxs, cstructs, stream)
|
||||
if not noarray:
|
||||
stream.write("}")
|
||||
stream.write(";\n\n")
|
||||
return stream.getvalue()
|
||||
|
||||
# def to_str(self, elf: BinaryIO):
|
||||
def __str__(self):
|
||||
stream = io.StringIO()
|
||||
stream.write(" {\n")
|
||||
for f, *_ in self._fields_: # pyright: ignore
|
||||
if f.startswith("_pad_"):
|
||||
continue
|
||||
v = getattr(self, f) # pyright: ignore
|
||||
if f in ("_in", "_pass"):
|
||||
f = f[1:] # pyright: ignore
|
||||
if isinstance(v, c_str):
|
||||
stream.write(f" .{f} = {v.to_str(self.__elf__)},\n")
|
||||
elif not isinstance(v, Array):
|
||||
stream.write(f" .{f} = {v},\n")
|
||||
else:
|
||||
arr = cast(Sized, v)
|
||||
if len(arr) and isinstance(v[0], CStructure):
|
||||
arr = cast(list[CStructure], arr)
|
||||
stream.write(f" .{f} = {{\n")
|
||||
for elem in arr:
|
||||
stream.write(f" {elem},\n")
|
||||
stream.write(" },\n")
|
||||
else:
|
||||
arr = cast(Sequence[Any], arr)
|
||||
if isinstance(arr[0], Array):
|
||||
# multidimensional ctypes array
|
||||
stream.write(f" .{f} = {{")
|
||||
print_carr(arr, stream)
|
||||
stream.write("},\n")
|
||||
else:
|
||||
stream.write(f" .{f} = {{")
|
||||
dims = [len(arr)]
|
||||
format_array(arr, dims, stream)
|
||||
|
||||
stream.write("},\n")
|
||||
stream.write(" }")
|
||||
return stream.getvalue()
|
||||
|
||||
|
||||
class UnionFieldsFromTypeHints(type(Union)): # pyright: ignore
|
||||
def __new__(
|
||||
cls: Type[type],
|
||||
name: str,
|
||||
bases: tuple[type, ...],
|
||||
namespace: dict[str, Any],
|
||||
/,
|
||||
*,
|
||||
align: Optional[int] = None,
|
||||
pack: Optional[int] = None,
|
||||
) -> UnionFieldsFromTypeHints:
|
||||
annotations = namespace.get("__annotations__", {})
|
||||
if align is not None:
|
||||
namespace["_align_"] = align
|
||||
if pack is not None:
|
||||
namespace["_pack_"] = pack
|
||||
if fields := list(annotations.items()):
|
||||
namespace["_fields_"] = fields
|
||||
return type(Union).__new__(cls, name, bases, namespace) # pyright: ignore
|
||||
|
||||
|
||||
class CUnion(Union, metaclass=UnionFieldsFromTypeHints):
|
||||
pass
|
||||
|
||||
|
||||
_T = TypeVar("_T", bound=CStructure)
|
||||
|
||||
|
||||
def sizeof(cstruct_type: Type[_T]) -> int:
|
||||
align = getattr(cstruct_type, "_align_", 0)
|
||||
c_size = ctypes.sizeof(cstruct_type)
|
||||
if align > 0:
|
||||
rem = c_size % align
|
||||
if rem:
|
||||
rem = align - rem
|
||||
return c_size + rem
|
||||
|
||||
return c_size
|
||||
|
||||
|
||||
def parse_cstruct(cstruct_type: Type[_T], data: bytes) -> list[_T]:
|
||||
cstruct_size = sizeof(cstruct_type)
|
||||
assert len(data) % cstruct_size == 0, f"{len(data)}, {cstruct_size}"
|
||||
cstruct_num = len(data) // cstruct_size
|
||||
stream = io.BytesIO(data)
|
||||
cstructs = [cstruct_type.from_buffer_copy(stream.read(cstruct_size)) for _ in range(cstruct_num)]
|
||||
return cstructs
|
||||
|
||||
|
||||
def print_cstruct(name: str, cstruct_type: Type[CStructure], data: bytes):
|
||||
cstructs = parse_cstruct(cstruct_type, data)
|
||||
print(f"{cstruct_type.__name__} {name}[{len(cstructs)}] = {{") # pyright: ignore
|
||||
for s in cstructs:
|
||||
print(s)
|
||||
print("};\n")
|
||||
@@ -0,0 +1,56 @@
|
||||
import argparse
|
||||
|
||||
|
||||
registers = {
|
||||
"$0": "zero",
|
||||
"$1": "at",
|
||||
"$2": "v0",
|
||||
"$3": "v1",
|
||||
"$4": "a0",
|
||||
"$5": "a1",
|
||||
"$6": "a2",
|
||||
"$7": "a3",
|
||||
"$8": "t0",
|
||||
"$9": "t1",
|
||||
"$10": "t2",
|
||||
"$11": "t3",
|
||||
"$12": "t4",
|
||||
"$13": "t5",
|
||||
"$14": "t6",
|
||||
"$15": "t7",
|
||||
"$16": "s0",
|
||||
"$17": "s1",
|
||||
"$18": "s2",
|
||||
"$19": "s3",
|
||||
"$20": "s4",
|
||||
"$21": "s5",
|
||||
"$22": "s6",
|
||||
"$23": "s7",
|
||||
"$24": "t8",
|
||||
"$25": "t9",
|
||||
"$26": "k0",
|
||||
"$27": "k1",
|
||||
"$28": "gp",
|
||||
"$29": "sp",
|
||||
"$30": "fp",
|
||||
"$31": "ra",
|
||||
}
|
||||
|
||||
|
||||
def fix_asm(asm_file: str):
|
||||
with open(asm_file, mode="r") as fh:
|
||||
for line in fh:
|
||||
for reg_num, reg_mnem in reversed(registers.items()):
|
||||
line = line.replace(reg_num, reg_mnem)
|
||||
print(line, end="")
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("asm_file", help="source assembly file path")
|
||||
args = parser.parse_args()
|
||||
fix_asm(args.asm_file)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,125 @@
|
||||
import os
|
||||
import re
|
||||
import argparse
|
||||
import pydantic
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Protocol, cast
|
||||
|
||||
|
||||
CONFIG_ROOT = Path(__file__).parent.parent.parent.resolve() / "config"
|
||||
|
||||
# enumerate languages from config dir
|
||||
LANGUAGES = [Path(f.path).name for f in os.scandir(CONFIG_ROOT) if f.is_dir()]
|
||||
|
||||
RE_INSTR = re.compile(r"^\t(?:[^.]|\.p2align \d+)")
|
||||
|
||||
|
||||
class InstructionPatch(pydantic.RootModel[tuple[int, str, str]]):
|
||||
root: tuple[int, str, str]
|
||||
|
||||
@property
|
||||
def instr_no(self):
|
||||
return self.root[0]
|
||||
|
||||
@property
|
||||
def instr_org(self):
|
||||
return self.root[1]
|
||||
|
||||
@property
|
||||
def instr_patch(self):
|
||||
return self.root[2]
|
||||
|
||||
|
||||
class ASMPatch(pydantic.RootModel[dict[str, list[InstructionPatch]]]):
|
||||
root: dict[str, list[InstructionPatch]]
|
||||
|
||||
|
||||
class PatchDB(pydantic.RootModel[dict[str, ASMPatch]]):
|
||||
root: dict[str, ASMPatch]
|
||||
|
||||
|
||||
def fix_asm(asm_file: Path, asm_patch: ASMPatch):
|
||||
lines = asm_file.read_text().splitlines()
|
||||
|
||||
def find_func(func: str):
|
||||
for i, line in enumerate(lines):
|
||||
if re.match(rf"^{func}:", line):
|
||||
return i
|
||||
|
||||
return -1
|
||||
|
||||
def find_line(start: int, num: int):
|
||||
offset = 0
|
||||
instr_no = -1
|
||||
for line in lines[start:]:
|
||||
if RE_INSTR.match(line):
|
||||
instr_no += 1
|
||||
|
||||
if instr_no == num:
|
||||
return start + offset
|
||||
|
||||
offset += 1
|
||||
|
||||
raise RuntimeError("cannot find instruction!")
|
||||
|
||||
for func, patch_lst in asm_patch.root.items():
|
||||
n = find_func(func)
|
||||
|
||||
if n == -1:
|
||||
print(f"WARNING: cannot find function {func} in {asm_file.name}")
|
||||
continue
|
||||
|
||||
for patch in patch_lst:
|
||||
line_no = find_line(n, patch.instr_no)
|
||||
line_org = lines[line_no]
|
||||
|
||||
line_org_clean = re.sub(r"\s+", " ", line_org).strip()
|
||||
instr_org_clean = re.sub(r"\s+", " ", patch.instr_org).strip()
|
||||
instr_patch_clean = re.sub(r"\s+", " ", patch.instr_patch).strip()
|
||||
|
||||
if line_org_clean != instr_org_clean:
|
||||
print(f"WARNING: wrong line: {asm_file.name}:{func}:{line_no + 1}: {line_org} != {patch.instr_org}")
|
||||
continue
|
||||
|
||||
lines[line_no] = f"\t{instr_patch_clean}"
|
||||
|
||||
asm_file.write_text("\n".join(lines))
|
||||
|
||||
|
||||
def main():
|
||||
class ArgsProtocol(Protocol):
|
||||
language: str
|
||||
asm_file: Path
|
||||
|
||||
parser = argparse.ArgumentParser(description="apply asm patches to assembly files")
|
||||
parser.add_argument("language", type=str, choices=LANGUAGES, help="language of the asm that is being patched")
|
||||
parser.add_argument("asm_file", type=Path, help="generated assembly file to patch (relative to build dir)")
|
||||
|
||||
args = cast(ArgsProtocol, parser.parse_args())
|
||||
|
||||
asm_file = CONFIG_ROOT / args.language / args.asm_file
|
||||
|
||||
if not asm_file.exists():
|
||||
print(f"ERROR: cannot find assembly file {asm_file}")
|
||||
exit(1)
|
||||
|
||||
patch_db_path = CONFIG_ROOT / args.language / "asm_patches.json"
|
||||
|
||||
if not patch_db_path.exists():
|
||||
print("ERROR: patch db file doesn't exist. aborting patching asm")
|
||||
exit(1)
|
||||
|
||||
patch_db = PatchDB.model_validate_json(patch_db_path.read_text())
|
||||
|
||||
if asm_file.name not in patch_db.root:
|
||||
print(f"ERROR: no patches found for file {asm_file.name}")
|
||||
exit(1)
|
||||
|
||||
asm_patch = patch_db.root[asm_file.name]
|
||||
|
||||
fix_asm(asm_file, asm_patch)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,38 @@
|
||||
import re
|
||||
import glob
|
||||
import tqdm
|
||||
import argparse
|
||||
|
||||
from typing import Protocol, cast
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def fix_assets(asm_data_path: Path, asset_rel_path: Path):
|
||||
asm_files = glob.glob(str(asm_data_path / "**/*.s"), recursive=True)
|
||||
for asm_file in tqdm.tqdm(asm_files, desc="Fixing data asm"):
|
||||
n: int = 0
|
||||
with open(asm_file, mode="r") as fh:
|
||||
data_asm: str = fh.read()
|
||||
data_asm, n = re.subn(rf'\.incbin "{asset_rel_path}/', '.incbin "assets/', data_asm)
|
||||
|
||||
if n > 0:
|
||||
with open(asm_file, mode="w") as wh:
|
||||
wh.write(data_asm)
|
||||
|
||||
|
||||
def main():
|
||||
class ArgsProtocol(Protocol):
|
||||
asm_data_path: Path
|
||||
asset_rel_path: Path
|
||||
|
||||
parser = argparse.ArgumentParser(description="fixes data asm include path")
|
||||
parser.add_argument("asm_path", metavar="asm-path", type=Path, help="data path in assembly root to patch")
|
||||
parser.add_argument("asset_rel_path", metavar="asset-path", type=Path, help="relative asset path")
|
||||
|
||||
args = cast(ArgsProtocol, parser.parse_args())
|
||||
|
||||
fix_assets(args.asm_data_path, args.asset_rel_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,79 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import argparse
|
||||
import pydantic
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Protocol, cast
|
||||
|
||||
|
||||
CONFIG_ROOT = Path(__file__).parent.parent.parent.resolve() / "config"
|
||||
|
||||
# enumerate languages from config dir
|
||||
LANGUAGES = [Path(f.path).name for f in os.scandir(CONFIG_ROOT) if f.is_dir()]
|
||||
|
||||
|
||||
class BytePatch(pydantic.RootModel[tuple[int, int]]):
|
||||
root: tuple[int, int]
|
||||
|
||||
@property
|
||||
def address(self):
|
||||
return self.root[0]
|
||||
|
||||
@property
|
||||
def size(self):
|
||||
return self.root[1]
|
||||
|
||||
|
||||
class BINPatch(pydantic.RootModel[dict[str, BytePatch]]):
|
||||
root: dict[str, BytePatch]
|
||||
|
||||
|
||||
class PatchDB(pydantic.RootModel[dict[str, BINPatch]]):
|
||||
root: dict[str, BINPatch]
|
||||
|
||||
|
||||
def fix_elf(orig_elf_path: Path, built_elf_path: Path, bin_patch: BytePatch):
|
||||
with orig_elf_path.open("rb") as fh:
|
||||
fh.seek(bin_patch.address)
|
||||
data = fh.read(bin_patch.size)
|
||||
|
||||
with built_elf_path.open(mode="r+b") as wh:
|
||||
wh.seek(bin_patch.address)
|
||||
wh.write(data)
|
||||
|
||||
|
||||
def main():
|
||||
class ArgsProtocol(Protocol):
|
||||
language: str
|
||||
elf_file: Path
|
||||
|
||||
parser = argparse.ArgumentParser(description="apply asm patches to assembly files")
|
||||
parser.add_argument("language", type=str, choices=LANGUAGES, help="language of the elf that is being patched")
|
||||
parser.add_argument("elf_file", type=Path, help="elf file to patch (relative to build dir)")
|
||||
|
||||
args = cast(ArgsProtocol, parser.parse_args())
|
||||
|
||||
built_elf_file = CONFIG_ROOT / args.language / args.elf_file
|
||||
|
||||
if not built_elf_file.exists():
|
||||
print(f"ERROR: cannot find elf file {built_elf_file}")
|
||||
exit(1)
|
||||
|
||||
orig_elf_path = CONFIG_ROOT / args.language / args.elf_file.name
|
||||
|
||||
patch_db_path = CONFIG_ROOT / args.language / "bin_patches.json"
|
||||
|
||||
if not patch_db_path.exists():
|
||||
exit(0)
|
||||
|
||||
patch_db = PatchDB.model_validate_json(patch_db_path.read_text())
|
||||
|
||||
for _tu_name, bin_patch in patch_db.root.items():
|
||||
for _func_name, byte_patch in bin_patch.root.items():
|
||||
fix_elf(orig_elf_path, built_elf_file, byte_patch)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,75 @@
|
||||
import re
|
||||
import glob
|
||||
import tqdm
|
||||
import argparse
|
||||
import functools
|
||||
|
||||
from typing import Protocol, cast
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
@functools.lru_cache(maxsize=1024) # pyright: ignore[reportUntypedFunctionDecorator]
|
||||
def get_symbol_address(symbol_addrs: Path, symbol: str):
|
||||
if match := re.match(r"^ *([^+ ]+) *\+ *(0x[0-9a-fA-F]+)$", symbol):
|
||||
symbol = match.group(1)
|
||||
offset = int(match.group(2), 16)
|
||||
else:
|
||||
offset = 0
|
||||
|
||||
with open(symbol_addrs, mode="r") as fh:
|
||||
for line in fh:
|
||||
if line.startswith(f"{symbol} "):
|
||||
addr = int(line.split("=")[1].split(";")[0].strip(), 16)
|
||||
return addr + offset
|
||||
|
||||
assert False, f"{symbol} not found"
|
||||
|
||||
|
||||
def fix_gp(asm_path: Path, gp_value: int, symbol_addrs: Path):
|
||||
if gp_value <= 0:
|
||||
return
|
||||
|
||||
asm_files = glob.glob(str(asm_path / "**/*.s"), recursive=True)
|
||||
for asm_file in tqdm.tqdm(asm_files, desc="Fixing gp_rel"):
|
||||
lines: list[str] = []
|
||||
with open(asm_file, mode="r") as fh:
|
||||
for line in fh:
|
||||
if match := re.match(r"^(.*)%gp_rel\(([^)]+)\)(.*)$", line):
|
||||
# ol = line
|
||||
instr_pre = match.group(1)
|
||||
instr_post = match.group(3)
|
||||
address_str = match.group(2)
|
||||
if address_str.startswith("D_"):
|
||||
address_str = address_str.replace("D_", "0x")
|
||||
res = eval(address_str)
|
||||
address = res
|
||||
else:
|
||||
address = get_symbol_address(symbol_addrs, address_str)
|
||||
gp_rel = address - gp_value
|
||||
line = f"{instr_pre}{hex(gp_rel)}{instr_post}\n"
|
||||
lines.append(line)
|
||||
with open(asm_file, mode="w") as wh:
|
||||
wh.writelines(lines)
|
||||
|
||||
|
||||
def main():
|
||||
class ArgsProtocol(Protocol):
|
||||
asm_path: Path
|
||||
gp_value: int
|
||||
symbol_addrs: Path
|
||||
|
||||
def hex_int(x: str):
|
||||
return int(x, 16)
|
||||
|
||||
parser = argparse.ArgumentParser(description="fixes asm removing gp_rel macro")
|
||||
parser.add_argument("asm_path", metavar="asm-path", type=Path, help="assembly root path to patch")
|
||||
parser.add_argument("gp_value", metavar="gp", type=hex_int, help="gp value in hex")
|
||||
parser.add_argument("symbol_addrs", metavar="symbol-addrs", type=Path, help="path of symbol_addrs.txt")
|
||||
|
||||
args = cast(ArgsProtocol, parser.parse_args())
|
||||
|
||||
fix_gp(args.asm_path, args.gp_value, args.symbol_addrs)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,139 @@
|
||||
import re
|
||||
import sys
|
||||
import yaml
|
||||
import tqdm
|
||||
|
||||
from typing import cast, Any
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
# e.g.: build/src/main/main.c.o(.text);
|
||||
# build/asm/data/rodata_4.rodata.s.o(.rodata);
|
||||
# ...
|
||||
# NOTE: also account for optional '/data/' in path as chunks of data
|
||||
# not belonging to a c file are put into 'build/asm/data/' folder.
|
||||
re_subsegment_line = re.compile(
|
||||
r"^(?P<indent> +)build/(?:asm|src)(?:/data)?/(?P<name>.*)\.[sc]\.o\(\.(?P<section>.+)\);$"
|
||||
)
|
||||
|
||||
# e.g.: .main 0x100000 : AT(main_ROM_START) SUBALIGN(2)
|
||||
# .main_bss (NOLOAD) : SUBALIGN(4)
|
||||
# ...
|
||||
re_section_line = re.compile(r"^(?P<indent> +)\.(?P<section>[^ ]+) .* SUBALIGN\((?P<subalign>[0-9]+)\)$")
|
||||
|
||||
|
||||
def get_align(address: int):
|
||||
return (
|
||||
128 * ((address % 128) == 0)
|
||||
or 64 * ((address % 64) == 0)
|
||||
or 32 * ((address % 32) == 0)
|
||||
or 16 * ((address % 16) == 0)
|
||||
or 8 * ((address % 8) == 0)
|
||||
or 4 * ((address % 4) == 0)
|
||||
or 2 * ((address % 2) == 0)
|
||||
or 1
|
||||
)
|
||||
|
||||
|
||||
def make_align_map(config: dict[str, Any]):
|
||||
segments = cast(list[dict[str, Any] | list[Any]] | None, config["segments"])
|
||||
assert segments
|
||||
main_segment = next(
|
||||
(segment for segment in segments if isinstance(segment, dict) and segment.get("name") == "main"), None
|
||||
)
|
||||
assert main_segment, "cannot find main segment"
|
||||
|
||||
subsegments = cast(list[dict[str, Any] | list[Any]] | None, main_segment["subsegments"])
|
||||
assert subsegments, "cannot find main subsegments"
|
||||
|
||||
align_map: dict[str, int] = {}
|
||||
|
||||
for subsegment in subsegments:
|
||||
if not isinstance(subsegment, dict):
|
||||
continue
|
||||
|
||||
s_type = cast(str | None, subsegment.get("type"))
|
||||
vram = cast(int | None, subsegment.get("vram"))
|
||||
name = cast(str | None, subsegment.get("name"))
|
||||
if not s_type or not vram or not name:
|
||||
continue
|
||||
|
||||
if name.endswith("bin"):
|
||||
continue
|
||||
|
||||
align = get_align(vram)
|
||||
|
||||
if not s_type.startswith("."):
|
||||
name = f"{s_type}#{name}.{s_type}"
|
||||
else:
|
||||
name = f"{s_type[1:]}#{name}"
|
||||
|
||||
align_map[name] = align
|
||||
|
||||
return align_map
|
||||
|
||||
|
||||
def fix_linkerscript(config: dict[str, Any], linkerscript_path: Path):
|
||||
align_map = make_align_map(config)
|
||||
|
||||
section_subalign = cast(dict[str, int], config["_section_subalign"])
|
||||
|
||||
line_count = 0
|
||||
with open(linkerscript_path, mode="r") as fh:
|
||||
for line in fh:
|
||||
line_count += 1
|
||||
|
||||
patched_lines: list[str] = []
|
||||
|
||||
with open(linkerscript_path, mode="r") as fh:
|
||||
for line in tqdm.tqdm(fh, desc="Fixing linker script", total=line_count):
|
||||
if match := re_subsegment_line.match(line):
|
||||
indent = cast(str, match["indent"])
|
||||
name = cast(str, match["name"])
|
||||
section = cast(str, match["section"])
|
||||
|
||||
# force each subsegment in the following sections to have align 8
|
||||
if section == "text":
|
||||
patched_lines.append(f"{indent}. = ALIGN(., 8);\n")
|
||||
|
||||
key = f"{section}#{name}"
|
||||
if align := align_map.get(key):
|
||||
patched_lines.append(f"{indent}. = ALIGN(., {align});\n")
|
||||
|
||||
if match := re_section_line.match(line):
|
||||
indent = cast(str, match["indent"])
|
||||
section = cast(str, match["section"])
|
||||
subalign = cast(str, match["subalign"])
|
||||
|
||||
# force each section to have the subalign specified in the yaml
|
||||
if section in section_subalign:
|
||||
current_subalign = f"SUBALIGN({subalign})"
|
||||
fixed_subalign = f"SUBALIGN({section_subalign[section]})"
|
||||
line = line.replace(current_subalign, fixed_subalign)
|
||||
|
||||
patched_lines.append(line)
|
||||
|
||||
with open(linkerscript_path, mode="w") as fh:
|
||||
fh.writelines(patched_lines)
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) != 3:
|
||||
print("usage: fix_linkerscript.py CONFIG_YAML_PATH LINKERSCRIPT_PATH")
|
||||
exit(1)
|
||||
|
||||
config_path = Path(sys.argv[1])
|
||||
linkerscript_path = Path(sys.argv[2])
|
||||
|
||||
with open(config_path, mode="r") as fh:
|
||||
try:
|
||||
config = cast(dict[str, Any], yaml.safe_load(fh))
|
||||
except yaml.YAMLError as e:
|
||||
print(e)
|
||||
raise e
|
||||
|
||||
fix_linkerscript(config, linkerscript_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,153 @@
|
||||
import json
|
||||
import argparse
|
||||
|
||||
from typing import Protocol, Iterable, Any, cast
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def fix_unit(unit: dict[str, Any]):
|
||||
# name: str = unit["name"]
|
||||
fuzzy_match_percent: float | None = unit["measures"].get("fuzzy_match_percent")
|
||||
|
||||
assert fuzzy_match_percent and 0 < fuzzy_match_percent < 100
|
||||
|
||||
text_section = next(section for section in unit["sections"] if section["name"] == ".text")
|
||||
# text_size: int = int(text_section["size"])
|
||||
text_section_fuzzy_match_percent: float = text_section["fuzzy_match_percent"]
|
||||
|
||||
if 0 < text_section_fuzzy_match_percent < 100:
|
||||
# fix text section fuzzy match percent
|
||||
text_section["fuzzy_match_percent"] = 100.0
|
||||
|
||||
text_fuzzy_match_percent: float = unit["measures"]["fuzzy_match_percent"]
|
||||
|
||||
assert text_fuzzy_match_percent == fuzzy_match_percent
|
||||
|
||||
functions = unit["functions"]
|
||||
|
||||
total_code: int = int(unit["measures"]["total_code"])
|
||||
matched_code: int = int(unit["measures"]["matched_code"])
|
||||
matched_functions = unit["measures"]["matched_functions"]
|
||||
|
||||
computed_total_code: int = 0
|
||||
computed_matched_code: int = 0
|
||||
computed_matched_functions: int = 0
|
||||
|
||||
duplicate_functions = (
|
||||
"Tim2CalcBufWidth__2",
|
||||
"_ftoi0__2",
|
||||
"ItemGetMain__2",
|
||||
"setD3_CHCR__2",
|
||||
"setD4_CHCR__2",
|
||||
"setD4_CHCR__3",
|
||||
"_fpadd_parts__2",
|
||||
)
|
||||
|
||||
for function in functions:
|
||||
function_size: int = int(function["size"])
|
||||
try:
|
||||
function_fuzzy_match_percent: float = function["fuzzy_match_percent"]
|
||||
except KeyError:
|
||||
# fix known function duplicates
|
||||
if function["name"] in duplicate_functions:
|
||||
function["fuzzy_match_percent"] = 100.0
|
||||
function_fuzzy_match_percent = 100.0
|
||||
matched_code += function_size
|
||||
matched_functions += 1
|
||||
else:
|
||||
raise
|
||||
|
||||
if function_fuzzy_match_percent == 100.0:
|
||||
computed_matched_code += function_size
|
||||
computed_matched_functions += 1
|
||||
|
||||
# fix function fuzzy match percent
|
||||
function["fuzzy_match_percent"] = 100.0
|
||||
|
||||
computed_total_code += function_size
|
||||
|
||||
assert total_code == computed_total_code
|
||||
assert matched_code == computed_matched_code
|
||||
assert matched_functions == computed_matched_functions
|
||||
|
||||
# fix unit measures
|
||||
unit["measures"]["fuzzy_match_percent"] = 100.0
|
||||
unit["measures"]["matched_code"] = unit["measures"]["total_code"]
|
||||
unit["measures"]["matched_code_percent"] = 100.0
|
||||
unit["measures"]["matched_functions"] = unit["measures"]["total_functions"]
|
||||
unit["measures"]["matched_functions_percent"] = 100.0
|
||||
|
||||
|
||||
def fix_report(report_path: Path):
|
||||
report = json.loads(report_path.read_text())
|
||||
|
||||
units: Iterable[Any] = report["units"]
|
||||
|
||||
computed_total_code: int = 0
|
||||
computed_matched_code: int = 0
|
||||
|
||||
computed_total_functions: int = 0
|
||||
computed_matched_functions: int = 0
|
||||
|
||||
total_code: int = int(report["measures"]["total_code"])
|
||||
total_functions: int = report["measures"]["total_functions"]
|
||||
|
||||
for unit in units:
|
||||
# name: str = unit["name"]
|
||||
unit_total_code: int = int(unit["measures"]["total_code"])
|
||||
unit_total_functions: int = unit["measures"]["total_functions"]
|
||||
fuzzy_match_percent: float | None = unit["measures"].get("fuzzy_match_percent")
|
||||
|
||||
if fuzzy_match_percent and 0 < fuzzy_match_percent < 100:
|
||||
fix_unit(unit)
|
||||
|
||||
if fuzzy_match_percent and fuzzy_match_percent > 0:
|
||||
computed_matched_code += unit_total_code
|
||||
computed_matched_functions += unit_total_functions
|
||||
|
||||
computed_total_functions += unit_total_functions
|
||||
computed_total_code += unit_total_code
|
||||
|
||||
assert total_code == computed_total_code
|
||||
assert total_functions == computed_total_functions
|
||||
|
||||
# fix report measures
|
||||
report["measures"]["fuzzy_match_percent"] = 100.0 * computed_matched_code / computed_total_code
|
||||
report["measures"]["matched_code"] = str(computed_matched_code)
|
||||
report["measures"]["matched_code_percent"] = report["measures"]["fuzzy_match_percent"]
|
||||
report["measures"]["matched_functions"] = computed_matched_functions
|
||||
report["measures"]["matched_functions_percent"] = 100.0 * computed_matched_functions / computed_total_functions
|
||||
|
||||
categories = report["categories"]
|
||||
assert len(categories) == 1
|
||||
assert categories[0]["measures"]["total_code"] == report["measures"]["total_code"]
|
||||
assert categories[0]["measures"]["total_units"] == report["measures"]["total_units"]
|
||||
|
||||
categories[0]["measures"]["fuzzy_match_percent"] = report["measures"]["fuzzy_match_percent"]
|
||||
categories[0]["measures"]["matched_code"] = report["measures"]["matched_code"]
|
||||
categories[0]["measures"]["matched_code_percent"] = report["measures"]["matched_code_percent"]
|
||||
categories[0]["measures"]["matched_functions"] = report["measures"]["matched_functions"]
|
||||
categories[0]["measures"]["matched_functions_percent"] = report["measures"]["matched_functions_percent"]
|
||||
|
||||
# /path/to/report.json -> /path/to/report_fixed.json
|
||||
# fixed_report_path = report_path.with_name(f"{report_path.stem}_fixed{report_path.suffix}")
|
||||
# fixed_report_path.write_text(json.dumps(report))
|
||||
report_path.write_text(json.dumps(report))
|
||||
|
||||
print(f"Wrote fixed report to {report_path}")
|
||||
|
||||
|
||||
def main():
|
||||
class ArgsProtocol(Protocol):
|
||||
report_path: Path
|
||||
|
||||
parser = argparse.ArgumentParser(description="fixes objdiff report")
|
||||
parser.add_argument("report_path", metavar="report-path", type=Path, help="path to the report generated by objdiff")
|
||||
|
||||
args = cast(ArgsProtocol, parser.parse_args())
|
||||
|
||||
fix_report(args.report_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,32 @@
|
||||
import re
|
||||
import argparse
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--language", required=True, choices=["us", "eu"])
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.language == "us":
|
||||
map_path = Path("config/us/build/SLUS_203.88.map")
|
||||
else:
|
||||
map_path = Path("config/eu/build/SLES_508.21.map")
|
||||
|
||||
with open(map_path, mode="r") as fh:
|
||||
for n, line in enumerate(fh):
|
||||
line = line.rstrip("\n")
|
||||
if match := re.match(r"^\s*0x([0-9a-fA-F]+)\s+[^ ]+?([0-9a-fA-F]{6,8})\s*$", line):
|
||||
addr = match.group(1)
|
||||
label = match.group(2)
|
||||
if addr.upper() != label.upper():
|
||||
print(f"{map_path}:{n+1} {line}")
|
||||
return
|
||||
|
||||
print("no mismatches found")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,288 @@
|
||||
# pyright: reportUnknownMemberType=false
|
||||
|
||||
import re
|
||||
import argparse
|
||||
|
||||
from typing import cast
|
||||
from pathlib import Path
|
||||
from elftools.elf.elffile import ELFFile
|
||||
from elftools.elf.sections import SymbolTableSection
|
||||
|
||||
|
||||
"""
|
||||
matches variable declaration:
|
||||
/* SECTION ADDRESS */ VAR_TYPE NAME[NUMEL];
|
||||
"""
|
||||
re_glob = re.compile(
|
||||
r"^/\* (?P<section>.*?) (?P<address>.*?) \*/ (?P<var_type>[^(]*) (?P<name>.*?)(?:\[(?P<numel>.*?)\])?;"
|
||||
)
|
||||
|
||||
"""
|
||||
matches structs (or unions) with no typedef:
|
||||
struct NAME { // SIZE
|
||||
...
|
||||
};
|
||||
"""
|
||||
re_struct = re.compile(
|
||||
r"^(?:struct|union) (?P<name>.*?) \{ // (?P<size>0x[0-9a-f]+)\n.*?^\};", flags=re.MULTILINE | re.DOTALL
|
||||
)
|
||||
|
||||
"""
|
||||
matches structs (or unions) with typedef:
|
||||
typedef struct { // SIZE
|
||||
...
|
||||
} NAME;
|
||||
"""
|
||||
re_typedef_struct = re.compile(
|
||||
r"^typedef (?:struct|union) \{ // (?P<size>0x[0-9a-f]+)\n.*?^\} (?P<name>.*?);", flags=re.MULTILINE | re.DOTALL
|
||||
)
|
||||
|
||||
sizes: dict[str, int] = {
|
||||
"char": 1,
|
||||
"u_char": 1,
|
||||
"short": 2,
|
||||
"short int": 2,
|
||||
"u_short": 2,
|
||||
"int": 4,
|
||||
"u_int": 4,
|
||||
"float": 4,
|
||||
"u_long128": 8,
|
||||
"sceVu0FMATRIX": 4 * 4 * 4,
|
||||
"sceVu0FVECTOR": 4 * 4,
|
||||
"sceSifClientData": 0x2C,
|
||||
}
|
||||
|
||||
command_script_keywords = (
|
||||
"VERSION",
|
||||
"SECTIONS",
|
||||
"ABSOLUTE",
|
||||
"LOADADDR",
|
||||
"ALIGN",
|
||||
"DEFINED",
|
||||
"NEXT",
|
||||
"SIZEOF",
|
||||
"SIZEOF_HEADERS",
|
||||
"MAX",
|
||||
"MIN",
|
||||
"PHDRS",
|
||||
"CREATE_OBJECT_SYMBOLS",
|
||||
"BYTE",
|
||||
"SHORT",
|
||||
"LONG",
|
||||
"SQUAD",
|
||||
"FILL",
|
||||
"BLOCK",
|
||||
"NOLOAD",
|
||||
"AT",
|
||||
"OVERLAY",
|
||||
"NOCROSSREFS",
|
||||
"PT_NULL",
|
||||
"PT_LOAD",
|
||||
"PT_DYNAMIC",
|
||||
"PT_INTERP",
|
||||
"PT_NOTE",
|
||||
"PT_SHLIB",
|
||||
"PT_PHDR",
|
||||
"ENTRY",
|
||||
"FLOAT",
|
||||
"NOFLOAT",
|
||||
"FORCE_COMMON_ALLOCATION",
|
||||
"INCLUDE",
|
||||
"INPUT",
|
||||
"GROUP",
|
||||
"OUTPUT",
|
||||
"OUTPUT_ARCH",
|
||||
"OUTPUT_FORMAT",
|
||||
"SEARCH_DIR",
|
||||
"STARTUP",
|
||||
"TARGET",
|
||||
"NOCROSSREFS",
|
||||
)
|
||||
|
||||
|
||||
class GlobalVarLineMatch:
|
||||
max_section_len: int = 0
|
||||
max_address_len: int = 0
|
||||
max_var_type_len: int = 0
|
||||
max_name_len: int = 0
|
||||
max_numel_len: int = 0
|
||||
|
||||
_name_cache: list[str] = []
|
||||
|
||||
@classmethod
|
||||
def _get_unique_name(cls, name: str):
|
||||
i = 1
|
||||
unique_name = f"{name}__local_{i}"
|
||||
while unique_name in cls._name_cache:
|
||||
i += 1
|
||||
unique_name = f"{name}__local_{i}"
|
||||
cls._name_cache.append(unique_name)
|
||||
return unique_name
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
symtab: SymbolTableSection,
|
||||
/,
|
||||
*,
|
||||
section: str,
|
||||
address: str | int,
|
||||
var_type: str,
|
||||
name: str,
|
||||
numel: str | int | None = None,
|
||||
):
|
||||
self.symtab: SymbolTableSection = symtab
|
||||
|
||||
self.section: str = section
|
||||
self.address: int = address if isinstance(address, int) else int(address, 16)
|
||||
self.var_type: str = var_type
|
||||
self.name: str = name
|
||||
self.numel: int
|
||||
|
||||
while self.name.startswith("*"):
|
||||
self.name = self.name[1:]
|
||||
self.var_type += "*"
|
||||
|
||||
# resolve function pointers (e.g., 'void (*SpecialEventInitTbl[0])...')
|
||||
if match := re.match(r"\(\*(.*?)\[.*?\]\).*", self.name):
|
||||
self.name = match.group(1)
|
||||
|
||||
self.name = self.name.replace("*", "")
|
||||
|
||||
if numel is not None:
|
||||
if isinstance(numel, int):
|
||||
self.numel = numel
|
||||
else:
|
||||
prod_numel = 1
|
||||
numels = numel.split("][")
|
||||
for n in numels:
|
||||
prod_numel *= int(n)
|
||||
self.numel = prod_numel
|
||||
|
||||
if numel is None:
|
||||
self.numel = 1
|
||||
|
||||
# get symbol
|
||||
symbol = self.symtab.get_symbol_by_name(self.name)
|
||||
|
||||
assert isinstance(symbol, list)
|
||||
|
||||
symbol = next(
|
||||
(sym for sym in symbol if sym.name == self.name and sym.entry["st_value"] == self.address),
|
||||
None,
|
||||
)
|
||||
|
||||
assert symbol
|
||||
|
||||
self.symbol = symbol
|
||||
|
||||
if not self.is_global:
|
||||
self.name = self._get_unique_name(self.name)
|
||||
|
||||
escaped_name_len = len(str(self.name))
|
||||
if self.name in command_script_keywords:
|
||||
escaped_name_len += 2
|
||||
|
||||
GlobalVarLineMatch.max_section_len = max(GlobalVarLineMatch.max_section_len, len(str(self.section)))
|
||||
GlobalVarLineMatch.max_address_len = max(GlobalVarLineMatch.max_address_len, len(str(self.address)))
|
||||
GlobalVarLineMatch.max_var_type_len = max(GlobalVarLineMatch.max_var_type_len, len(str(self.var_type)))
|
||||
GlobalVarLineMatch.max_name_len = max(GlobalVarLineMatch.max_name_len, escaped_name_len)
|
||||
GlobalVarLineMatch.max_numel_len = max(GlobalVarLineMatch.max_numel_len, len(str(self.numel)))
|
||||
|
||||
@property
|
||||
def is_global(self):
|
||||
return cast(str, self.symbol["st_info"]["bind"]) == "STB_GLOBAL"
|
||||
|
||||
@property
|
||||
def is_hidden(self):
|
||||
return cast(str, self.symbol["st_other"]["visibility"]) == "STV_HIDDEN"
|
||||
|
||||
@property
|
||||
def size(self):
|
||||
size = cast(int, self.symbol["st_size"])
|
||||
if size == 0:
|
||||
if "*" in self.var_type:
|
||||
size = 4
|
||||
elif self.var_type in sizes:
|
||||
size = sizes[self.var_type] * self.numel
|
||||
return size
|
||||
|
||||
def to_string(self, as_linker_command_file: bool):
|
||||
name = self.name if self.name not in command_script_keywords else f'"{self.name}"'
|
||||
cls_str = f"{name:{GlobalVarLineMatch.max_name_len}s} = 0x{self.address:08x};"
|
||||
|
||||
if not as_linker_command_file:
|
||||
if self.size > 0:
|
||||
cls_str = f"{cls_str} // size:0x{self.size:x}"
|
||||
else:
|
||||
cls_str = f"{cls_str} //0 {self.var_type} * {self.numel}"
|
||||
|
||||
# cls_str += f" bind:{'global' if self.is_global else 'local'}"
|
||||
# cls_str += f" visibility:{'hidden' if self.is_hidden else 'visible'}"
|
||||
|
||||
return cls_str
|
||||
|
||||
def __str__(self):
|
||||
return self.to_string(as_linker_command_file=False)
|
||||
|
||||
|
||||
def parse_globals(elf_path: Path, globals_path: Path, types_path: Path, as_linker_command_file: bool):
|
||||
with open(elf_path, mode="rb") as fh:
|
||||
elf = ELFFile(fh)
|
||||
|
||||
# Find the symbol table.
|
||||
symtab = elf.get_section_by_name(".symtab")
|
||||
assert isinstance(symtab, SymbolTableSection)
|
||||
|
||||
with open(types_path, mode="r") as f:
|
||||
types_data = f.read()
|
||||
|
||||
for struct_type, size_hex_str in re_struct.findall(types_data):
|
||||
struct_type = cast(str, struct_type)
|
||||
struct_size = int(cast(str, size_hex_str), 16)
|
||||
# assert not (struct_type in sizes and sizes[struct_type] != struct_size)
|
||||
sizes[struct_type] = struct_size
|
||||
|
||||
for size_hex_str, struct_type in re_typedef_struct.findall(types_data):
|
||||
struct_type = cast(str, struct_type)
|
||||
struct_size = int(cast(str, size_hex_str), 16)
|
||||
# assert not (struct_type in sizes and sizes[struct_type] != struct_size), (
|
||||
# struct_type,
|
||||
# sizes[struct_type],
|
||||
# struct_size,
|
||||
# )
|
||||
sizes[struct_type] = struct_size
|
||||
|
||||
if "tagSE_WRK" in sizes:
|
||||
sizes["SE_WRK"] = sizes["tagSE_WRK"]
|
||||
|
||||
with open(globals_path, mode="r") as f:
|
||||
lines = [line.strip() for line in f.readlines() if line.startswith("/*") and line.strip().endswith(";")]
|
||||
|
||||
matches = [GlobalVarLineMatch(symtab, **match.groupdict()) for line in lines if (match := re_glob.match(line))]
|
||||
matches.sort(key=lambda match: match.address)
|
||||
|
||||
for match in matches:
|
||||
print(match.to_string(as_linker_command_file=as_linker_command_file))
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--language", required=True, choices=["us", "eu"])
|
||||
parser.add_argument("--as-linker-command-file", action="store_true")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.language == "us":
|
||||
elf_path = Path("config/us/SLUS_203.88")
|
||||
globals_path = Path("ccc/SLUS_203.88/globals.h")
|
||||
types_path = Path("ccc/SLUS_203.88/types.h")
|
||||
else:
|
||||
elf_path = Path("config/eu/SLES_508.21")
|
||||
globals_path = Path("ccc/SLES_508.21/globals.h")
|
||||
types_path = Path("ccc/SLES_508.21/types.h")
|
||||
|
||||
parse_globals(elf_path, globals_path, types_path, args.as_linker_command_file)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,378 @@
|
||||
# pyright: reportUnknownMemberType=false
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import json
|
||||
import argparse
|
||||
import itertools
|
||||
|
||||
from typing import Protocol, TextIO, cast
|
||||
from pathlib import Path
|
||||
from elftools.elf.elffile import ELFFile
|
||||
from elftools.elf.sections import SymbolTableSection, Symbol
|
||||
|
||||
import ccc_json_v7
|
||||
|
||||
Range = tuple[int, int]
|
||||
|
||||
command_script_keywords = (
|
||||
"VERSION",
|
||||
"SECTIONS",
|
||||
"ABSOLUTE",
|
||||
"LOADADDR",
|
||||
"ALIGN",
|
||||
"DEFINED",
|
||||
"NEXT",
|
||||
"SIZEOF",
|
||||
"SIZEOF_HEADERS",
|
||||
"MAX",
|
||||
"MIN",
|
||||
"PHDRS",
|
||||
"CREATE_OBJECT_SYMBOLS",
|
||||
"BYTE",
|
||||
"SHORT",
|
||||
"LONG",
|
||||
"SQUAD",
|
||||
"FILL",
|
||||
"BLOCK",
|
||||
"NOLOAD",
|
||||
"AT",
|
||||
"OVERLAY",
|
||||
"NOCROSSREFS",
|
||||
"PT_NULL",
|
||||
"PT_LOAD",
|
||||
"PT_DYNAMIC",
|
||||
"PT_INTERP",
|
||||
"PT_NOTE",
|
||||
"PT_SHLIB",
|
||||
"PT_PHDR",
|
||||
"ENTRY",
|
||||
"FLOAT",
|
||||
"NOFLOAT",
|
||||
"FORCE_COMMON_ALLOCATION",
|
||||
"INCLUDE",
|
||||
"INPUT",
|
||||
"GROUP",
|
||||
"OUTPUT",
|
||||
"OUTPUT_ARCH",
|
||||
"OUTPUT_FORMAT",
|
||||
"SEARCH_DIR",
|
||||
"STARTUP",
|
||||
"TARGET",
|
||||
"NOCROSSREFS",
|
||||
)
|
||||
|
||||
skip_symbols = (
|
||||
"_fbss",
|
||||
"_gp",
|
||||
)
|
||||
|
||||
size_exceptions = {
|
||||
"dorcon": 0x1C,
|
||||
}
|
||||
|
||||
|
||||
def parse_types(ccc_model_v7: ccc_json_v7.CCCJSONv7Model):
|
||||
type_map: dict[str, int] = {}
|
||||
|
||||
for n, dt in enumerate(ccc_model_v7.deduplicated_types):
|
||||
if dt.descriptor == "builtin":
|
||||
assert dt.name and dt.name not in type_map
|
||||
assert dt.class_ is not None
|
||||
size_bits = int(dt.class_.split("-", maxsplit=1)[0])
|
||||
assert size_bits % 8 == 0
|
||||
type_map[dt.name] = size_bits // 8
|
||||
|
||||
elif dt.descriptor == "type_name":
|
||||
assert dt.name is not None
|
||||
assert dt.type_name is not None
|
||||
if dt.name in type_map:
|
||||
if dt.size_bits:
|
||||
assert type_map[dt.name] == dt.size_bits // 8
|
||||
if dt.name == dt.type_name == "void":
|
||||
type_map["void"] = type_map["int"]
|
||||
continue
|
||||
assert dt.type_name in type_map, (n, dt.type_name)
|
||||
type_map[dt.name] = type_map[dt.type_name]
|
||||
|
||||
elif dt.descriptor == "pointer":
|
||||
assert dt.name is not None
|
||||
assert dt.value_type is not None
|
||||
if dt.value_type.descriptor == "function_type":
|
||||
assert dt.name not in type_map
|
||||
type_map[dt.name] = type_map["int"]
|
||||
elif dt.value_type.descriptor == "type_name":
|
||||
assert dt.value_type.type_name
|
||||
assert dt.value_type.type_name in type_map
|
||||
type_map[dt.name] = type_map[dt.value_type.type_name]
|
||||
|
||||
elif dt.descriptor == "struct":
|
||||
assert dt.name is not None
|
||||
if not dt.conflict:
|
||||
assert dt.name not in type_map, n
|
||||
else:
|
||||
if dt.name in type_map:
|
||||
continue
|
||||
assert dt.size_bits is not None
|
||||
assert dt.size_bits % 8 == 0
|
||||
type_map[dt.name] = dt.size_bits // 8
|
||||
|
||||
elif dt.descriptor == "array":
|
||||
assert dt.name is not None
|
||||
element_count = 1
|
||||
element_type = dt.element_type
|
||||
type_name = None
|
||||
while element_type:
|
||||
if element_type.element_count is not None:
|
||||
element_count *= element_type.element_count
|
||||
if element_type.type_name is not None:
|
||||
type_name = element_type.type_name
|
||||
assert type_name in type_map
|
||||
element_count *= type_map[type_name]
|
||||
element_type = element_type.element_type
|
||||
assert type_name and type_name in type_map, n
|
||||
type_map[dt.name] = type_map[type_name] * element_count
|
||||
|
||||
elif dt.descriptor == "enum":
|
||||
if dt.name:
|
||||
if not dt.conflict:
|
||||
assert dt.name not in type_map, n
|
||||
type_map[dt.name] = type_map["int"]
|
||||
|
||||
elif dt.descriptor == "union":
|
||||
assert dt.name
|
||||
assert dt.size_bits
|
||||
assert dt.size_bits % 8 == 0
|
||||
type_map[dt.name] = dt.size_bits // 8
|
||||
|
||||
else:
|
||||
assert False, f"unknown {n}"
|
||||
|
||||
return type_map
|
||||
|
||||
|
||||
def in_range(ranges: list[Range], address: int):
|
||||
if not ranges:
|
||||
return True
|
||||
|
||||
in_range = any((ra[0] <= address <= ra[1]) for ra in ranges)
|
||||
|
||||
return in_range
|
||||
|
||||
|
||||
def parse_ccc_model_v7(ccc_model_v7: ccc_json_v7.CCCJSONv7Model, ranges: list[Range]):
|
||||
type_sizes = parse_types(ccc_model_v7)
|
||||
|
||||
if "pointer" not in type_sizes:
|
||||
type_sizes["pointer"] = type_sizes["int"]
|
||||
|
||||
static_locals: list[ccc_json_v7.Local] = []
|
||||
global_vars: list[ccc_json_v7.Global] = []
|
||||
|
||||
for file in ccc_model_v7.files:
|
||||
for global_var in file.globals:
|
||||
assert global_var.storage.global_address
|
||||
if in_range(ranges, global_var.storage.global_address):
|
||||
global_vars.append(global_var)
|
||||
|
||||
for function in file.functions:
|
||||
for local in function.locals:
|
||||
if local.storage_class == "static":
|
||||
assert local.storage.global_address
|
||||
if in_range(ranges, local.storage.global_address):
|
||||
static_locals.append(local)
|
||||
|
||||
return type_sizes, global_vars, static_locals
|
||||
|
||||
|
||||
class SymbolWithNoNameException(Exception): ...
|
||||
|
||||
|
||||
class ParsedSymbol:
|
||||
_name_map: dict[str, list[ParsedSymbol]] = {}
|
||||
_max_name_len: int = 0
|
||||
|
||||
def __init__(self, symtab: SymbolTableSection, symbol: Symbol) -> None:
|
||||
self.symtab = symtab
|
||||
self.symbol = symbol
|
||||
|
||||
if not self.name:
|
||||
raise SymbolWithNoNameException
|
||||
|
||||
if self.name not in ParsedSymbol._name_map:
|
||||
ParsedSymbol._name_map[self.name] = []
|
||||
|
||||
ParsedSymbol._name_map[self.name].append(self)
|
||||
ParsedSymbol._max_name_len = max(ParsedSymbol._max_name_len, len(self.name))
|
||||
|
||||
@property
|
||||
def name(self) -> str:
|
||||
return re.sub(r"^(.*?)(\.\d+)?$", r"\1", self.symbol.name)
|
||||
|
||||
@property
|
||||
def size(self) -> int:
|
||||
return cast(int, self.symbol["st_size"])
|
||||
|
||||
@property
|
||||
def address(self) -> int:
|
||||
return cast(int, self.symbol.entry["st_value"])
|
||||
|
||||
def in_range(self, range: Range):
|
||||
return range[0] <= self.address <= range[1]
|
||||
|
||||
def to_undefined_syms(self) -> str:
|
||||
homonyms = ParsedSymbol._name_map[self.name]
|
||||
assert len(homonyms) > 0
|
||||
if len(homonyms) == 1:
|
||||
assert homonyms[0] == self
|
||||
name = self.name
|
||||
else:
|
||||
idx = homonyms.index(self)
|
||||
name = f"{self.name}__local_{idx + 1}"
|
||||
|
||||
name = name if name not in command_script_keywords else f'"{name}"'
|
||||
|
||||
# name_len_fmt = ParsedSymbol._max_name_len + len("__local_9999")
|
||||
name_len_fmt = 40
|
||||
|
||||
return f"{name:{name_len_fmt}s} = 0x{self.address:08x};"
|
||||
|
||||
def to_symbol_addrs(
|
||||
self,
|
||||
type_sizes: dict[str, int],
|
||||
global_vars: list[ccc_json_v7.Global],
|
||||
static_locals: list[ccc_json_v7.Local],
|
||||
) -> str:
|
||||
homonyms = ParsedSymbol._name_map[self.name]
|
||||
assert len(homonyms) > 0
|
||||
if len(homonyms) == 1:
|
||||
assert homonyms[0] == self
|
||||
name = self.name
|
||||
else:
|
||||
idx = homonyms.index(self)
|
||||
name = f"{self.name}__local_{idx + 1}"
|
||||
|
||||
# name_len_fmt = ParsedSymbol._max_name_len + len("__local_9999")
|
||||
name_len_fmt = 40
|
||||
|
||||
size = size_exceptions.get(self.name, self.size)
|
||||
if size == 0:
|
||||
var = next(
|
||||
(global_ for global_ in global_vars if global_.storage.global_address == self.address), None
|
||||
) or next((local for local in static_locals if local.storage.global_address == self.address), None)
|
||||
|
||||
if not var:
|
||||
print(f"cannot find {self.name} with size 0 in json")
|
||||
else:
|
||||
element_count, type_name = var.parsed_size
|
||||
type_size = type_sizes.get(type_name)
|
||||
assert type_size is not None, type_name
|
||||
size = element_count * type_size
|
||||
|
||||
if not size:
|
||||
print(f"size of {self.name} is also 0 using json")
|
||||
|
||||
size_str = f"size:0x{size:x}" if size else ""
|
||||
|
||||
return f"{name:{name_len_fmt}s} = 0x{self.address:08x}; // {size_str}"
|
||||
|
||||
|
||||
def parse_symbols_safe(elf_path: Path, dest_path: Path, json_path: Path, ranges: list[Range]):
|
||||
if not dest_path.is_dir():
|
||||
raise RuntimeError(f"{dest_path} is not a directory")
|
||||
|
||||
symbol_addrs_path = dest_path / "symbols_addrs.txt"
|
||||
undefined_syms_path = dest_path / "undefined_syms.txt"
|
||||
|
||||
if symbol_addrs_path.exists() or undefined_syms_path.exists():
|
||||
raise RuntimeError("symbols_addrs.txt or undefined_syms.txt already exist in dest folder")
|
||||
|
||||
with open(elf_path, mode="rb") as elf_fh, open(json_path, mode="r") as json_fh:
|
||||
elf = ELFFile(elf_fh)
|
||||
|
||||
json_data = json.load(json_fh)
|
||||
ccc_model_v7 = ccc_json_v7.CCCJSONv7Model.model_validate(json_data)
|
||||
type_sizes, global_vars, static_locals = parse_ccc_model_v7(ccc_model_v7, ranges)
|
||||
|
||||
with open(symbol_addrs_path, mode="w") as symbol_addrs, open(undefined_syms_path, mode="w") as undefined_syms:
|
||||
parse_symbols(elf, symbol_addrs, undefined_syms, ranges, type_sizes, global_vars, static_locals)
|
||||
|
||||
|
||||
def parse_symbols(
|
||||
elf: ELFFile,
|
||||
symbol_addrs: TextIO,
|
||||
undefined_syms: TextIO,
|
||||
ranges: list[Range],
|
||||
type_sizes: dict[str, int],
|
||||
global_vars: list[ccc_json_v7.Global],
|
||||
static_locals: list[ccc_json_v7.Local],
|
||||
):
|
||||
# find symbol table
|
||||
symtab = elf.get_section_by_name(".symtab")
|
||||
assert isinstance(symtab, SymbolTableSection)
|
||||
|
||||
parsed_symbols: list[ParsedSymbol] = []
|
||||
|
||||
for symbol in symtab.iter_symbols():
|
||||
try:
|
||||
parsed_symbol = ParsedSymbol(symtab, symbol)
|
||||
if parsed_symbol.name in skip_symbols:
|
||||
continue
|
||||
parsed_symbols.append(parsed_symbol)
|
||||
except SymbolWithNoNameException:
|
||||
pass
|
||||
|
||||
parsed_symbols.sort(key=lambda x: x.address)
|
||||
|
||||
for parsed_symbol in parsed_symbols:
|
||||
if ranges:
|
||||
in_range = any(parsed_symbol.in_range(ra) for ra in ranges)
|
||||
if not in_range:
|
||||
continue
|
||||
|
||||
parsed_symbol.address
|
||||
symbol_addrs.write(parsed_symbol.to_symbol_addrs(type_sizes, global_vars, static_locals))
|
||||
symbol_addrs.write("\n")
|
||||
|
||||
undefined_syms.write(parsed_symbol.to_undefined_syms())
|
||||
undefined_syms.write("\n")
|
||||
|
||||
|
||||
def main():
|
||||
class ArgsProtocol(Protocol):
|
||||
elf_path: Path
|
||||
dest_path: Path
|
||||
json_path: Path
|
||||
ranges: list[Range]
|
||||
|
||||
def range_type(arg: str) -> Range:
|
||||
if m := re.match(r"^([0-9a-f]+)-([0-9a-f]+)$", arg):
|
||||
range_start, range_end = int(m.group(1), 16), int(m.group(2), 16)
|
||||
if not (range_end > range_start):
|
||||
raise argparse.ArgumentTypeError("non monotonic range")
|
||||
return range_start, range_end
|
||||
raise argparse.ArgumentTypeError("invalid range")
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("elf_path", metavar="ELF", type=Path, help="path to ELF file")
|
||||
parser.add_argument("--json", dest="json_path", type=Path, help="path to CCC json (v7)")
|
||||
parser.add_argument(
|
||||
"--dest",
|
||||
dest="dest_path",
|
||||
type=Path,
|
||||
required=True,
|
||||
help="folder where symbol_addrs.txt and undefined_syms.txt will be created "
|
||||
"(existing files will not be overwritten)",
|
||||
)
|
||||
parser.add_argument("--range", action="append", dest="ranges", nargs="+", type=range_type)
|
||||
|
||||
args = parser.parse_args()
|
||||
args.ranges = list(itertools.chain(*args.ranges))
|
||||
args = cast(ArgsProtocol, args)
|
||||
|
||||
parse_symbols_safe(args.elf_path, args.dest_path, args.json_path, args.ranges)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,414 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Literal, Optional
|
||||
from pydantic import RootModel, BaseModel, Field
|
||||
|
||||
|
||||
class Compiler(BaseModel):
|
||||
name: str
|
||||
asm_function_macro: str = Field("glabel", description="")
|
||||
asm_function_alt_macro: str = Field("glabel", description="")
|
||||
asm_jtbl_label_macro: str = Field("glabel", description="")
|
||||
asm_data_macro: str = Field("glabel", description="")
|
||||
asm_end_label: str = Field("", description="")
|
||||
c_newline: str = Field("\n", description="")
|
||||
asm_inc_header: str = Field("", description="")
|
||||
include_macro_inc: bool = Field(True, description="")
|
||||
asm_emit_size_directive: Optional[bool] = Field(None, description="")
|
||||
|
||||
|
||||
class SplatOpts(BaseModel):
|
||||
# Debug / logging
|
||||
verbose: Optional[bool] = Field(None, description="Verbose")
|
||||
dump_symbols: Optional[bool] = Field(None, description="Dump symbols")
|
||||
modes: Optional[list[str]] = Field(None, description="modes")
|
||||
|
||||
# Project configuration
|
||||
base_path: Path = Field(
|
||||
None, description="Determines the base path of the project. Everything is relative to this path"
|
||||
)
|
||||
target_path: Optional[Path] = Field(None, description="Determines the path to the target binary")
|
||||
elf_path: Optional[Path] = Field(None, description="Path to the final elf target")
|
||||
platform: Optional[str] = Field(None, description="Determines the platform of the target binary")
|
||||
compiler: Optional[Compiler | str] = Field(
|
||||
None, description="Determines the compiler used to compile the target binary"
|
||||
)
|
||||
endianness: Optional[Literal["big", "little"]] = Field(
|
||||
None, description="Determines the endianness of the target binary"
|
||||
)
|
||||
section_order: Optional[list[str]] = Field(
|
||||
None,
|
||||
description="Determines the default section order of the target binary. This can be overridden per-segment",
|
||||
)
|
||||
generated_c_preamble: Optional[str] = Field(
|
||||
None, description="Determines the code that is inserted by default in generated .c files"
|
||||
)
|
||||
generated_s_preamble: Optional[str] = Field(
|
||||
None, description="Determines the code that is inserted by default in generated .s files"
|
||||
)
|
||||
use_o_as_suffix: Optional[bool] = Field(
|
||||
None, description="Determines whether to use .o as the suffix for all binary files?... TODO document"
|
||||
)
|
||||
gp_value: Optional[int] = Field(
|
||||
None, description="the value of the $gp register to correctly calculate offset to %gp_rel relocs"
|
||||
)
|
||||
check_consecutive_segment_types: Optional[bool] = Field(
|
||||
None, description="Checks and errors if there are any non consecutive segment types"
|
||||
)
|
||||
|
||||
# Paths
|
||||
asset_path: Optional[Path] = Field(None, description="")
|
||||
symbol_addrs_paths: Optional[list[Path]] = Field(
|
||||
None,
|
||||
description="""Determines the path to the symbol addresses file(s)
|
||||
A symbol_addrs file is to be updated/curated manually and contains addresses of symbols
|
||||
as well as optional metadata such as rom address, type, and more
|
||||
|
||||
It's possible to use more than one file by supplying a list instead of a string""",
|
||||
)
|
||||
reloc_addrs_paths: Optional[list[Path]] = Field(None, description="")
|
||||
build_path: Optional[Path] = Field(None, description="Determines the path to the project build directory")
|
||||
src_path: Optional[Path] = Field(None, description="Determines the path to the source code directory")
|
||||
asm_path: Optional[Path] = Field(None, description="Determines the path to the asm code directory")
|
||||
data_path: Optional[Path] = Field(None, description="Determines the path to the asm data directory")
|
||||
nonmatchings_path: Optional[Path] = Field(None, description="Determines the path to the asm nonmatchings directory")
|
||||
cache_path: Optional[Path] = Field(
|
||||
None, description="Determines the path to the cache file (used when supplied --use-cache via the CLI)"
|
||||
)
|
||||
hasm_in_src_path: Optional[bool] = Field(
|
||||
None, description="Tells splat to consider `hasm` files to be relative to `src_path` instead of `asm_path`."
|
||||
)
|
||||
create_undefined_funcs_auto: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to create an automatically-generated undefined functions fil. "
|
||||
"This file stores all functions that are referenced in the code but are not defined as seen by splat",
|
||||
)
|
||||
undefined_funcs_auto_path: Optional[Path] = Field(
|
||||
None, description="Determines the path to the undefined_funcs_auto file"
|
||||
)
|
||||
|
||||
create_undefined_syms_auto: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to create an automatically-generated undefined symbols file. "
|
||||
"This file stores all symbols that are referenced in the code but are not defined as seen by splat",
|
||||
)
|
||||
undefined_syms_auto_path: Optional[Path] = Field(
|
||||
None, description="Determines the path to the undefined_symbols_auto file"
|
||||
)
|
||||
|
||||
extensions_path: Optional[Path] = Field(
|
||||
None, description="Determines the path in which to search for custom splat extensions"
|
||||
)
|
||||
|
||||
lib_path: Optional[Path] = Field(
|
||||
None, description="Determines the path to library files that are to be linked into the target binary"
|
||||
)
|
||||
|
||||
# TODO document
|
||||
elf_section_list_path: Optional[Path] = Field(None, description="")
|
||||
|
||||
# Linker script
|
||||
subalign: Optional[int] = Field(
|
||||
None, description="Determines the default subalign value to be specified in the generated linker script"
|
||||
)
|
||||
|
||||
auto_all_sections: Optional[list[str]] = Field(
|
||||
None,
|
||||
description="The following option determines whether to automatically configure the linker script to link "
|
||||
'against specified sections for all "base" (asm/c) files when the yaml doesn\'t have manual configurations '
|
||||
"for these sections.",
|
||||
)
|
||||
ld_script_path: Optional[Path] = Field(
|
||||
None, description="Determines the desired path to the linker script that splat will generate"
|
||||
)
|
||||
ld_symbol_header_path: Optional[Path] = Field(
|
||||
None,
|
||||
description="Determines the desired path to the linker symbol header, which exposes externed definitions "
|
||||
"for all segment ram/rom start/end locations",
|
||||
)
|
||||
ld_discard_section: Optional[bool] = Field(
|
||||
None, description="Determines whether to add a discard section with a wildcard to the linker script"
|
||||
)
|
||||
ld_sections_allowlist: Optional[list[str]] = Field(
|
||||
None,
|
||||
description="A list of sections to preserve during link time. It can be useful to preserve debugging sections",
|
||||
)
|
||||
ld_sections_denylist: Optional[list[str]] = Field(
|
||||
None,
|
||||
description="A list of sections to discard during link time. It can be useful to avoid using the wildcard "
|
||||
"discard. Note that this option does not turn off `ld_discard_section`",
|
||||
)
|
||||
ld_wildcard_sections: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to add wildcards for section linking in the linker script "
|
||||
"(.rodata* for example)",
|
||||
)
|
||||
ld_use_symbolic_vram_addresses: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to use `follows_vram` (segment option) and `vram_symbol` / `follows_classes` "
|
||||
"(vram_class options) to calculate vram addresses in the linker script. If disabled, this uses the plain "
|
||||
"integer values for vram addresses defined in the yaml.",
|
||||
)
|
||||
ld_partial_linking: Optional[bool] = Field(
|
||||
None,
|
||||
description="Change linker script generation to allow partially linking segments. Requires both "
|
||||
"`ld_partial_scripts_path` and `ld_partial_build_segments_path` to be set.",
|
||||
)
|
||||
ld_partial_scripts_path: Optional[Path] = Field(
|
||||
None, description="Folder were each intermediary linker script will be written to."
|
||||
)
|
||||
ld_partial_build_segments_path: Optional[Path] = Field(
|
||||
None, description="Folder where the built partially linked segments will be placed by the build system."
|
||||
)
|
||||
ld_dependencies: Optional[bool] = Field(
|
||||
None,
|
||||
description="Generate a dependency file for every linker script generated. Dependency files will have the "
|
||||
"same path and name as the corresponding linker script, but changing the extension to `.d`. Requires "
|
||||
"`elf_path` to be set.",
|
||||
)
|
||||
ld_legacy_generation: Optional[bool] = Field(
|
||||
None,
|
||||
description="Legacy linker script generation does not impose the section_order specified in the yaml "
|
||||
"options or per-segment options.",
|
||||
)
|
||||
segment_end_before_align: Optional[bool] = Field(
|
||||
None,
|
||||
description="If enabled, the end symbol for each segment will be placed before the alignment directive "
|
||||
"for the segment",
|
||||
)
|
||||
segment_symbols_style: Optional[str] = Field(
|
||||
None,
|
||||
description="Controls the style of the auto-generated segment symbols in the linker script. "
|
||||
"Possible values: Optional[splat, makerom",
|
||||
)
|
||||
ld_rom_start: Optional[int] = Field(
|
||||
None, description="Specifies the starting offset for rom address symbols in the linker script."
|
||||
)
|
||||
ld_fill_value: Optional[int] = Field(
|
||||
None,
|
||||
description="The value passed to the FILL statement on each segment. `None` disables using FILL "
|
||||
"statements on the linker script. Defaults to a fill value of 0.",
|
||||
)
|
||||
ld_bss_is_noload: Optional[bool] = Field(
|
||||
None,
|
||||
description="Allows to control if `bss` sections (and derivative sections) will be put on a `NOLOAD` "
|
||||
"segment on the generated linker script or not.",
|
||||
)
|
||||
ld_align_segment_vram_end: Optional[bool] = Field(
|
||||
None, description="Allows to toggle aligning the `*_VRAM_END` linker symbol for each segment."
|
||||
)
|
||||
ld_align_section_vram_end: Optional[bool] = Field(
|
||||
None, description="Allows to toggle aligning the `*_END` linker symbol for each section of each section."
|
||||
)
|
||||
ld_generate_symbol_per_data_segment: Optional[bool] = Field(
|
||||
None, description="If enabled, the generated linker script will have a linker symbol for each data file"
|
||||
)
|
||||
ld_bss_contains_common: Optional[bool] = Field(
|
||||
None, description="Sets the default option for the `bss_contains_common` attribute of all segments."
|
||||
)
|
||||
|
||||
################################################################################
|
||||
# C file options
|
||||
################################################################################
|
||||
create_c_files: Optional[bool] = Field(
|
||||
None, description="Determines whether to create new c files if they don't exist"
|
||||
)
|
||||
auto_decompile_empty_functions: Optional[bool] = Field(
|
||||
None, description='Determines whether to "auto-decompile" empty functions'
|
||||
)
|
||||
do_c_func_detection: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to detect matched/unmatched functions in existing c files so we can avoid "
|
||||
"creating .s files for already-decompiled functions",
|
||||
)
|
||||
c_newline: Optional[str] = Field(None, description="Determines the newline char(s) to be used in c files")
|
||||
|
||||
################################################################################
|
||||
# (Dis)assembly-related options
|
||||
################################################################################
|
||||
symbol_name_format: Optional[str] = Field(
|
||||
None, description="The following options determine the format that symbols should be named by default"
|
||||
)
|
||||
symbol_name_format_no_rom: Optional[str] = Field(
|
||||
None, description="Same as above but for symbols with no rom address"
|
||||
)
|
||||
find_file_boundaries: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to detect and hint to the user about likely file splits " "when disassembling",
|
||||
)
|
||||
pair_rodata_to_text: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to detect and hint to the user about possible rodata sections "
|
||||
"corresponding to a text section",
|
||||
)
|
||||
migrate_rodata_to_functions: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to attempt to automatically migrate rodata into functions "
|
||||
"(only works in certain circumstances)",
|
||||
)
|
||||
asm_inc_header: Optional[str] = Field(
|
||||
None, description="Determines the header to be used in every asm file that's included from c files"
|
||||
)
|
||||
asm_function_macro: Optional[str] = Field(
|
||||
None, description="Determines the macro used to declare functions in asm files"
|
||||
)
|
||||
asm_function_alt_macro: Optional[str] = Field(
|
||||
None,
|
||||
description="Determines the macro used to declare symbols in the middle of functions in asm files "
|
||||
"(which may be alternative entries)",
|
||||
)
|
||||
asm_jtbl_label_macro: Optional[str] = Field(
|
||||
None, description="Determines the macro used to declare jumptable labels in asm files"
|
||||
)
|
||||
asm_data_macro: Optional[str] = Field(
|
||||
None, description="Determines the macro used to declare data symbols in asm files"
|
||||
)
|
||||
asm_end_label: Optional[str] = Field(
|
||||
None, description="Determines the macro used at the end of a function, such as endlabel or .end"
|
||||
)
|
||||
asm_emit_size_directive: Optional[bool] = Field(
|
||||
None, description="Toggles the .size directive emitted by the disassembler"
|
||||
)
|
||||
include_macro_inc: Optional[bool] = Field(
|
||||
None, description="Determines including the macro.inc file on non-migrated rodata variables"
|
||||
)
|
||||
mnemonic_ljust: Optional[int] = Field(
|
||||
None, description="Determines the number of characters to left align before the TODO finish documenting"
|
||||
)
|
||||
rom_address_padding: Optional[bool] = Field(None, description="Determines whether to pad the rom address")
|
||||
mips_abi_gpr: Optional[str] = Field(
|
||||
None, description="Determines which ABI names to use for general purpose registers"
|
||||
)
|
||||
mips_abi_float_regs: Optional[str] = Field(
|
||||
None,
|
||||
description="""Determines which ABI names to use for floating point registers
|
||||
Valid values: 'numeric', 'o32', 'n32', 'n64'
|
||||
o32 is highly recommended, as it provides logically named registers for floating point instructions
|
||||
For more info, see https://gist.github.com/EllipticEllipsis/27eef11205c7a59d8ea85632bc49224d""",
|
||||
)
|
||||
named_regs_for_c_funcs: Optional[bool] = Field(
|
||||
None, description="Determines whether functions inside c files should have named registers"
|
||||
)
|
||||
add_set_gp_64: Optional[bool] = Field(None, description='Determines whether to add ".set gp=64" to asm/hasm files')
|
||||
create_asm_dependencies: Optional[bool] = Field(
|
||||
None,
|
||||
description="Generate .asmproc.d dependency files for each C file which still reference functions "
|
||||
"in assembly files",
|
||||
)
|
||||
string_encoding: Optional[str] = Field(
|
||||
None, description="Global option for rodata string encoding. This can be overridden per segment"
|
||||
)
|
||||
data_string_encoding: Optional[str] = Field(
|
||||
None, description="Global option for data string encoding. This can be overridden per segment"
|
||||
)
|
||||
rodata_string_guesser_level: Optional[int] = Field(
|
||||
None, description="Global option for the rodata string guesser. 0 disables the guesser completely."
|
||||
)
|
||||
data_string_guesser_level: Optional[int] = Field(
|
||||
None, description="Global option for the data string guesser. 0 disables the guesser completely."
|
||||
)
|
||||
allow_data_addends: Optional[bool] = Field(
|
||||
None,
|
||||
description="Global option for allowing data symbols using addends on symbol references. "
|
||||
"It can be overridden per symbol",
|
||||
)
|
||||
disasm_unknown: Optional[bool] = Field(
|
||||
None,
|
||||
description="Tells the disassembler to try disassembling functions with unknown instructions instead of "
|
||||
"falling back to disassembling as raw data",
|
||||
)
|
||||
detect_redundant_function_end: Optional[bool] = Field(
|
||||
None,
|
||||
description="Tries to detect redundant and unreferenced functions ends and merge them together. "
|
||||
"This option is ignored if the compiler is not set to IDO.",
|
||||
)
|
||||
disassemble_all: Optional[bool] = Field(
|
||||
None, description="Don't skip disassembling already matched functions and migrated sections"
|
||||
)
|
||||
|
||||
################################################################################
|
||||
# N64-specific options
|
||||
################################################################################
|
||||
header_encoding: Optional[str] = Field(None, description="Determines the encoding of the header")
|
||||
gfx_ucode: Optional[str] = Field(
|
||||
None,
|
||||
description="""Determines the type gfx ucode (used by gfx segments)
|
||||
Valid options are ['f3d', 'f3db', 'f3dex', 'f3dexb', 'f3dex2']""",
|
||||
)
|
||||
libultra_symbols: Optional[bool] = Field(
|
||||
None,
|
||||
description="Use named libultra symbols by default. Those will need to be added to a linker script "
|
||||
"manually by the user",
|
||||
)
|
||||
ique_symbols: Optional[bool] = Field(
|
||||
None,
|
||||
description="Use named libultra symbols by default. Those will need to be added to a linker script "
|
||||
"manually by the user",
|
||||
)
|
||||
hardware_regs: Optional[bool] = Field(
|
||||
None,
|
||||
description="Use named hardware register symbols by default. Those will need to be added to a linker script "
|
||||
"manually by the user",
|
||||
)
|
||||
image_type_in_extension: Optional[bool] = Field(
|
||||
None, description="Append the image type to the output file extension"
|
||||
)
|
||||
|
||||
################################################################################
|
||||
# Compiler-specific options
|
||||
################################################################################
|
||||
use_legacy_include_asm: Optional[bool] = Field(
|
||||
None,
|
||||
description="Determines whether to use a legacy INCLUDE_ASM macro "
|
||||
"format in c files only applies to GCC/SN64",
|
||||
)
|
||||
|
||||
|
||||
class VramClass(BaseModel):
|
||||
name: str = Field(..., description="")
|
||||
vram: Optional[int] = Field(None, description="")
|
||||
|
||||
|
||||
class DictSegment(BaseModel):
|
||||
start: int = Field(None, description="")
|
||||
rom_start: Optional[int] = Field(None, description="")
|
||||
rom_start: Optional[int] = Field(None, description="")
|
||||
rom_end: Optional[int] = Field(None, description="")
|
||||
type: str = Field(..., description="")
|
||||
name: str = Field(..., description="")
|
||||
vram: Optional[int] = Field(None, description="")
|
||||
vram_start: Optional[int] = Field(None, description="")
|
||||
vram_symbol: Optional[str] = Field(None, description="")
|
||||
vram_class: Optional[VramClass] = Field(None, description="")
|
||||
follows_vram: Optional[str] = Field(None, description="")
|
||||
align: Optional[int] = Field(None, description="")
|
||||
subalign: Optional[int] = Field(None, description="")
|
||||
section_order: Optional[list[str]] = Field(None, description="")
|
||||
subsegments: Optional[list[Segment]] = Field(None, description="")
|
||||
bss_size: Optional[int] = Field(None, description="")
|
||||
args: Optional[list[str]] = Field(None, description="")
|
||||
|
||||
|
||||
class ListSegment(RootModel[tuple[int, str, str] | tuple[int, str] | tuple[int]]): ...
|
||||
|
||||
|
||||
Segment = DictSegment | ListSegment
|
||||
|
||||
|
||||
class Config(BaseModel):
|
||||
name: str
|
||||
sha1: str
|
||||
options: SplatOpts
|
||||
segments: list[Segment]
|
||||
|
||||
|
||||
def main():
|
||||
import json
|
||||
|
||||
with open("../../.vscode/schema/splat_config.schema.json", "w") as fh:
|
||||
fh.write(json.dumps(Config.model_json_schema(), indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user