Files
hashbro 91196b8832 init
2026-08-08 05:00:54 +08:00

648 lines
24 KiB
Python

"""Safely retarget the native initial daily URL to a channel-scoped path.
type-0x01 thin dylibs and the fat core (``tmp.dylib`` / erupt_flee) represent
the path as an immutable CFString. The replacement is longer than the
original, so it is stored in validated, file-backed padding between the Mach-O
load commands and ``__text``. The CFString data pointer and length are then
updated while preserving the architecture's dyld relocation encoding.
Fat binaries are patched slice-by-slice in place (arm64 + arm64e).
"""
from __future__ import annotations
import struct
from dataclasses import dataclass, field
from _channel_patch import validate_channel_id
MH_MAGIC_64 = 0xFEEDFACF
FAT_MAGIC = 0xCAFEBABE
FAT_CIGAM = 0xBEBAFECA
CPU_TYPE_ARM64 = 0x0100000C
LC_SEGMENT_64 = 0x19
LC_UUID = 0x1B
LC_DYLD_INFO_ONLY = 0x80000022
LC_DYLD_CHAINED_FIXUPS = 0x80000034
OLD_DAILY_PATH = b"/sync/daily.html"
PATH_TEMPLATE = "/channel/{channel}/sync/daily.html"
_POINTER_SIZE = 8
@dataclass(frozen=True)
class _Profile:
architecture: str
uuid: str
path_offset: int
cfstring_offset: int
cave_offset: int
raw_reference: int
# These offsets are not used blindly: _inspect_source derives every location
# from Mach-O sections/relocations and requires it to equal the matching UUID's
# known layout before any bytes are changed.
_PROFILES = {
# type-0x01 secondary packs (thin)
"73827b4262ba3a5989ada4fe6f67e266": _Profile(
architecture="arm64",
uuid="73827b4262ba3a5989ada4fe6f67e266",
path_offset=0x3C9F8,
cfstring_offset=0x45488,
cave_offset=0x10B0,
raw_reference=0x3C9F8,
),
"ec21ac8ad85333d5a4f9723ff3f31a2c": _Profile(
architecture="arm64e",
uuid="ec21ac8ad85333d5a4f9723ff3f31a2c",
path_offset=0x409F0,
cfstring_offset=0x49A20,
cave_offset=0x10F0,
raw_reference=0x00100000000409F0,
),
# sync core tmp.dylib slices (fat: arm64 + arm64e)
"4262fd020de4337e9a690626a28fa4e3": _Profile(
architecture="arm64",
uuid="4262fd020de4337e9a690626a28fa4e3",
path_offset=0xE1AB8,
cfstring_offset=0x1134D8,
cave_offset=0x1180,
raw_reference=0xE1AB8,
),
"8a796924959b310481a4ace808db8521": _Profile(
architecture="arm64e",
uuid="8a796924959b310481a4ace808db8521",
path_offset=0xF1FE4,
cfstring_offset=0x123D80,
cave_offset=0x11C0,
raw_reference=0x00100000000F1FE4,
),
}
@dataclass(frozen=True)
class _FatSlice:
offset: int
size: int
@dataclass
class _Section:
segment: str
name: str
addr: int
size: int
offset: int
@dataclass
class _Segment:
name: str
vmaddr: int
vmsize: int
fileoff: int
filesize: int
initprot: int
sections: list[_Section] = field(default_factory=list)
@dataclass
class _MachO:
data: bytes
cpu_subtype: int
load_end: int
uuid: str
segments: list[_Segment]
dyld_rebase: tuple[int, int] | None
chained_fixups: tuple[int, int] | None
def section(self, segment: str, name: str) -> _Section:
hits = [
section
for item in self.segments
for section in item.sections
if section.segment == segment and section.name == name
]
if len(hits) != 1:
raise ValueError(f"expected one {segment},{name} section; found {len(hits)}")
return hits[0]
def segment(self, name: str) -> _Segment:
hits = [segment for segment in self.segments if segment.name == name]
if len(hits) != 1:
raise ValueError(f"expected one {name} segment; found {len(hits)}")
return hits[0]
@dataclass(frozen=True)
class _Inspection:
macho: _MachO
profile: _Profile
path_vmaddr: int
cfstring_offset: int
reference_offset: int
cave_vmaddr: int
def _fail(label: str, message: str) -> SystemExit:
return SystemExit(f"{label}: {message}. Refusing unsafe initial daily path patch.")
def _name(raw: bytes) -> str:
return raw.split(b"\x00", 1)[0].decode("ascii")
def _parse_macho(data: bytes, *, label: str) -> _MachO:
if len(data) < 32:
raise _fail(label, "truncated Mach-O header")
magic, cputype, cpusubtype, _filetype, ncmds, sizeofcmds = struct.unpack_from(
"<IIIIII", data, 0
)
if magic != MH_MAGIC_64 or cputype != CPU_TYPE_ARM64:
raise _fail(label, "expected a thin little-endian arm64 Mach-O")
load_end = 32 + sizeofcmds
if load_end > len(data):
raise _fail(label, "load commands exceed file size")
segments: list[_Segment] = []
uuid = ""
dyld_rebase: tuple[int, int] | None = None
chained_fixups: tuple[int, int] | None = None
offset = 32
for _ in range(ncmds):
if offset + 8 > load_end:
raise _fail(label, "truncated load command")
cmd, cmdsize = struct.unpack_from("<II", data, offset)
if cmdsize < 8 or offset + cmdsize > load_end:
raise _fail(label, "invalid load command size")
if cmd == LC_SEGMENT_64:
if cmdsize < 72:
raise _fail(label, "truncated LC_SEGMENT_64")
segname = _name(data[offset + 8 : offset + 24])
vmaddr, vmsize, fileoff, filesize = struct.unpack_from(
"<QQQQ", data, offset + 24
)
initprot = struct.unpack_from("<i", data, offset + 60)[0]
nsects = struct.unpack_from("<I", data, offset + 64)[0]
if 72 + nsects * 80 > cmdsize:
raise _fail(label, f"{segname} section table exceeds load command")
segment = _Segment(
segname, vmaddr, vmsize, fileoff, filesize, initprot
)
section_offset = offset + 72
for _section_index in range(nsects):
sectname = _name(data[section_offset : section_offset + 16])
section_segment = _name(
data[section_offset + 16 : section_offset + 32]
)
addr, size = struct.unpack_from("<QQ", data, section_offset + 32)
file_offset = struct.unpack_from("<I", data, section_offset + 48)[0]
if file_offset and file_offset + size > len(data):
raise _fail(label, f"{section_segment},{sectname} exceeds file size")
segment.sections.append(
_Section(section_segment, sectname, addr, size, file_offset)
)
section_offset += 80
segments.append(segment)
elif cmd == LC_UUID:
if cmdsize != 24 or uuid:
raise _fail(label, "invalid or duplicate LC_UUID")
uuid = data[offset + 8 : offset + 24].hex()
elif cmd == LC_DYLD_INFO_ONLY:
if cmdsize != 48:
raise _fail(label, "invalid LC_DYLD_INFO_ONLY")
rebase_off, rebase_size = struct.unpack_from("<II", data, offset + 8)
dyld_rebase = (rebase_off, rebase_size)
elif cmd == LC_DYLD_CHAINED_FIXUPS:
if cmdsize != 16:
raise _fail(label, "invalid LC_DYLD_CHAINED_FIXUPS")
chained_fixups = struct.unpack_from("<II", data, offset + 8)
offset += cmdsize
if offset != load_end or not uuid:
raise _fail(label, "load command extent or UUID differs")
return _MachO(
data,
cpusubtype,
load_end,
uuid,
segments,
dyld_rebase,
chained_fixups,
)
def _vmaddr_for_file_offset(macho: _MachO, offset: int) -> int:
hits = [
segment.vmaddr + offset - segment.fileoff
for segment in macho.segments
if segment.fileoff <= offset < segment.fileoff + segment.filesize
]
if len(hits) != 1:
raise ValueError(f"file offset {offset:#x} is not in one file-backed segment")
return hits[0]
def _file_offset_for_vmaddr(macho: _MachO, address: int) -> int:
hits = [
segment.fileoff + address - segment.vmaddr
for segment in macho.segments
if (
segment.vmaddr <= address < segment.vmaddr + segment.filesize
and segment.filesize
)
]
if len(hits) != 1:
raise ValueError(f"vmaddr {address:#x} is not in one file-backed segment")
return hits[0]
def _read_uleb(data: bytes, offset: int, end: int) -> tuple[int, int]:
value = 0
shift = 0
while offset < end and shift < 64:
byte = data[offset]
offset += 1
value |= (byte & 0x7F) << shift
if not byte & 0x80:
return value, offset
shift += 7
raise ValueError("invalid ULEB128")
def _classic_rebase_locations(macho: _MachO) -> set[int]:
if macho.dyld_rebase is None:
raise ValueError("missing LC_DYLD_INFO_ONLY")
start, size = macho.dyld_rebase
end = start + size
if start < macho.load_end or end > len(macho.data):
raise ValueError("invalid dyld rebase stream")
segment_index = -1
address = 0
locations: set[int] = set()
offset = start
def record(count: int, skip: int = 0) -> None:
nonlocal address
if segment_index < 0 or segment_index >= len(macho.segments):
raise ValueError("rebase before segment selection")
segment = macho.segments[segment_index]
for _ in range(count):
if not (segment.vmaddr <= address < segment.vmaddr + segment.vmsize):
raise ValueError("rebase address exceeds segment")
locations.add(address)
address += _POINTER_SIZE + skip
while offset < end:
byte = macho.data[offset]
offset += 1
opcode, immediate = byte & 0xF0, byte & 0x0F
if opcode == 0x00:
break
if opcode == 0x10: # SET_TYPE_IMM
if immediate != 1:
raise ValueError("unsupported non-pointer rebase type")
elif opcode == 0x20: # SET_SEGMENT_AND_OFFSET_ULEB
segment_index = immediate
delta, offset = _read_uleb(macho.data, offset, end)
if segment_index >= len(macho.segments):
raise ValueError("invalid rebase segment index")
address = macho.segments[segment_index].vmaddr + delta
elif opcode == 0x30: # ADD_ADDR_ULEB
delta, offset = _read_uleb(macho.data, offset, end)
address += delta
elif opcode == 0x40: # ADD_ADDR_IMM_SCALED
address += immediate * _POINTER_SIZE
elif opcode == 0x50: # DO_REBASE_IMM_TIMES
record(immediate)
elif opcode == 0x60: # DO_REBASE_ULEB_TIMES
count, offset = _read_uleb(macho.data, offset, end)
record(count)
elif opcode == 0x70: # DO_REBASE_ADD_ADDR_ULEB
skip, offset = _read_uleb(macho.data, offset, end)
record(1, skip)
elif opcode == 0x80: # DO_REBASE_ULEB_TIMES_SKIPPING_ULEB
count, offset = _read_uleb(macho.data, offset, end)
skip, offset = _read_uleb(macho.data, offset, end)
record(count, skip)
else:
raise ValueError(f"unsupported rebase opcode {opcode:#x}")
return locations
def _arm64e_target(raw: int) -> int | None:
# DYLD_CHAINED_PTR_ARM64E non-auth rebase:
# target:43, high8:8, next:11, bind:1, auth:1.
if (raw >> 63) & 1 or (raw >> 62) & 1:
return None
target = raw & ((1 << 43) - 1)
high8 = (raw >> 43) & 0xFF
return target | (high8 << 56)
def _chained_arm64e_locations(macho: _MachO) -> set[int]:
if macho.chained_fixups is None:
raise ValueError("missing LC_DYLD_CHAINED_FIXUPS")
dataoff, datasize = macho.chained_fixups
end = dataoff + datasize
if dataoff < macho.load_end or end > len(macho.data) or datasize < 28:
raise ValueError("invalid chained-fixups payload")
(
version,
starts_offset,
_imports_offset,
_symbols_offset,
_imports_count,
_imports_format,
_symbols_format,
) = struct.unpack_from("<7I", macho.data, dataoff)
if version != 0:
raise ValueError(f"unsupported chained-fixups version {version}")
starts = dataoff + starts_offset
if starts + 4 > end:
raise ValueError("invalid chained starts offset")
segment_count = struct.unpack_from("<I", macho.data, starts)[0]
if segment_count != len(macho.segments):
raise ValueError("chained segment count differs from Mach-O segments")
table_end = starts + 4 + segment_count * 4
if table_end > end:
raise ValueError("truncated chained segment table")
locations: set[int] = set()
for segment_index in range(segment_count):
relative = struct.unpack_from("<I", macho.data, starts + 4 + 4 * segment_index)[0]
if not relative:
continue
info = starts + relative
if info + 22 > end:
raise ValueError("truncated chained segment info")
size, page_size, pointer_format, segment_offset, _max_ptr, page_count = (
struct.unpack_from("<IHHQIH", macho.data, info)
)
if pointer_format != 1: # DYLD_CHAINED_PTR_ARM64E
raise ValueError(f"unsupported chained pointer format {pointer_format}")
if size < 22 + page_count * 2 or info + size > end:
raise ValueError("invalid chained segment info size")
segment = macho.segments[segment_index]
if segment_offset != segment.vmaddr:
raise ValueError("chained segment offset differs from vmaddr")
for page_index in range(page_count):
page_start = struct.unpack_from(
"<H", macho.data, info + 22 + page_index * 2
)[0]
if page_start == 0xFFFF:
continue
if page_start & 0x8000:
raise ValueError("unsupported multi-start chained page")
address = segment_offset + page_index * page_size + page_start
while True:
file_offset = segment.fileoff + address - segment.vmaddr
if file_offset + 8 > segment.fileoff + segment.filesize:
raise ValueError("chained pointer exceeds file-backed segment")
locations.add(address)
raw = struct.unpack_from("<Q", macho.data, file_offset)[0]
next_delta = (raw >> 51) & 0x7FF
if not next_delta:
break
address += next_delta * 8
return locations
def _inspect_source(data: bytes, *, replacement_size: int, label: str) -> _Inspection:
try:
macho = _parse_macho(data, label=label)
profile = _PROFILES.get(macho.uuid)
if profile is None:
raise ValueError(f"unsupported Mach-O UUID {macho.uuid}")
base_subtype = macho.cpu_subtype & 0x00FFFFFF
architecture = "arm64e" if base_subtype == 2 else "arm64" if base_subtype == 0 else ""
if architecture != profile.architecture:
raise ValueError(
f"CPU subtype resolves to {architecture or base_subtype}, expected "
f"{profile.architecture}"
)
cstring = macho.section("__TEXT", "__cstring")
needle = OLD_DAILY_PATH + b"\x00"
region = data[cstring.offset : cstring.offset + cstring.size]
relative_hits = []
start = 0
while True:
hit = region.find(needle, start)
if hit < 0:
break
relative_hits.append(hit)
start = hit + 1
if len(relative_hits) != 1:
raise ValueError(f"source daily path hits={len(relative_hits)}, expected 1")
path_offset = cstring.offset + relative_hits[0]
path_vmaddr = cstring.addr + relative_hits[0]
cfstring = macho.section("__DATA_CONST", "__cfstring")
if cfstring.size % 32:
raise ValueError("__cfstring size is not record-aligned")
candidates: list[tuple[int, int]] = []
for record_offset in range(
cfstring.offset, cfstring.offset + cfstring.size, 32
):
raw = struct.unpack_from("<Q", data, record_offset + 16)[0]
length = struct.unpack_from("<Q", data, record_offset + 24)[0]
target = raw if architecture == "arm64" else _arm64e_target(raw)
if target == path_vmaddr and length == len(OLD_DAILY_PATH):
candidates.append((record_offset, raw))
if len(candidates) != 1:
raise ValueError(f"daily path CFString refs={len(candidates)}, expected 1")
cfstring_offset, raw_reference = candidates[0]
reference_offset = cfstring_offset + 16
reference_vmaddr = _vmaddr_for_file_offset(macho, reference_offset)
if architecture == "arm64":
locations = _classic_rebase_locations(macho)
path_references = [
address
for address in locations
if struct.unpack_from(
"<Q", data, _file_offset_for_vmaddr(macho, address)
)[0]
== path_vmaddr
]
else:
locations = _chained_arm64e_locations(macho)
path_references = [
address
for address in locations
if _arm64e_target(
struct.unpack_from(
"<Q", data, _file_offset_for_vmaddr(macho, address)
)[0]
)
== path_vmaddr
]
if path_references != [reference_vmaddr]:
raise ValueError(
"relocated source path references differ "
f"(found={tuple(hex(address) for address in path_references)})"
)
text_segment = macho.segment("__TEXT")
text_section = macho.section("__TEXT", "__text")
cave_offset = (macho.load_end + 15) & ~15
if not (
text_segment.fileoff <= cave_offset
and cave_offset + replacement_size <= text_section.offset
and text_section.offset <= text_segment.fileoff + text_segment.filesize
and text_segment.initprot & 1
):
raise ValueError("load-command padding is not safe file-backed __TEXT data")
if data[cave_offset : cave_offset + replacement_size] != b"\x00" * replacement_size:
raise ValueError("expected zero-filled __TEXT cave differs")
cave_vmaddr = _vmaddr_for_file_offset(macho, cave_offset)
observed = (
path_offset,
cfstring_offset,
cave_offset,
raw_reference,
)
expected = (
profile.path_offset,
profile.cfstring_offset,
profile.cave_offset,
profile.raw_reference,
)
if observed != expected:
raise ValueError(
"derived source layout differs from UUID profile "
f"(observed={tuple(hex(x) for x in observed)})"
)
return _Inspection(
macho,
profile,
path_vmaddr,
cfstring_offset,
reference_offset,
cave_vmaddr,
)
except (IndexError, struct.error, UnicodeDecodeError, ValueError) as exc:
raise _fail(label, str(exc)) from exc
def _fat_slices(data: bytes, *, label: str) -> list[_FatSlice] | None:
if len(data) < 8:
return None
magic = struct.unpack_from(">I", data, 0)[0]
if magic not in (FAT_MAGIC, FAT_CIGAM):
return None
nfat = struct.unpack_from(">I", data, 4)[0]
if nfat < 1 or 8 + nfat * 20 > len(data):
raise _fail(label, "invalid fat header")
slices: list[_FatSlice] = []
for index in range(nfat):
_cputype, _cpusubtype, offset, size, _align = struct.unpack_from(
">IIIII", data, 8 + index * 20
)
if offset < 0 or size < 1 or offset + size > len(data):
raise _fail(label, f"fat slice {index} exceeds file size")
slices.append(_FatSlice(offset, size))
return slices
def _patch_thin_daily_path(
data: bytes, channel: str, *, label: str
) -> tuple[bytes, dict[str, object]]:
new_path = PATH_TEMPLATE.format(channel=channel).encode("ascii")
encoded = new_path + b"\x00"
inspection = _inspect_source(data, replacement_size=len(encoded), label=label)
buf = bytearray(data)
cave_offset = inspection.profile.cave_offset
buf[cave_offset : cave_offset + len(encoded)] = encoded
raw = struct.unpack_from("<Q", data, inspection.reference_offset)[0]
if inspection.profile.architecture == "arm64":
new_raw = inspection.cave_vmaddr
else:
if _arm64e_target(raw) != inspection.path_vmaddr:
raise _fail(label, "arm64e source reference changed before write")
if inspection.cave_vmaddr >= 1 << 43:
raise _fail(label, "arm64e cave target exceeds chained pointer range")
new_raw = (raw & ~((1 << 43) - 1)) | inspection.cave_vmaddr
struct.pack_into("<Q", buf, inspection.reference_offset, new_raw)
struct.pack_into("<Q", buf, inspection.cfstring_offset + 24, len(new_path))
patched = bytes(buf)
if (
patched[cave_offset : cave_offset + len(encoded)] != encoded
or struct.unpack_from("<Q", patched, inspection.cfstring_offset + 24)[0]
!= len(new_path)
):
raise _fail(label, "post-patch path or CFString length validation failed")
target = (
struct.unpack_from("<Q", patched, inspection.reference_offset)[0]
if inspection.profile.architecture == "arm64"
else _arm64e_target(
struct.unpack_from("<Q", patched, inspection.reference_offset)[0]
)
)
if target != inspection.cave_vmaddr:
raise _fail(label, "post-patch CFString reference validation failed")
metadata: dict[str, object] = {
"architecture": inspection.profile.architecture,
"macho_uuid": inspection.profile.uuid,
"strategy": "cfstring_retarget_to_text_padding",
"source_path": OLD_DAILY_PATH.decode("ascii"),
"path": new_path.decode("ascii"),
"path_file_offset": cave_offset,
"path_vmaddr": f"0x{inspection.cave_vmaddr:x}",
"cfstring_file_offset": inspection.cfstring_offset,
"reference_file_offset": inspection.reference_offset,
}
return patched, metadata
def patch_initial_daily_path(
data: bytes, channel: str, *, label: str = "dylib"
) -> tuple[bytes, dict[str, object]]:
"""Patch the initial daily CFString and return bytes plus manifest metadata.
Accepts thin type-0x01 dylibs or the fat core ``tmp.dylib``.
"""
channel = validate_channel_id(channel, name="channel")
slices = _fat_slices(data, label=label)
if slices is None:
return _patch_thin_daily_path(data, channel, label=label)
if len(slices) != 2:
raise _fail(label, f"expected fat arm64+arm64e (2 slices), found {len(slices)}")
buf = bytearray(data)
slice_meta: list[dict[str, object]] = []
for index, slice_info in enumerate(slices):
thin = bytes(buf[slice_info.offset : slice_info.offset + slice_info.size])
patched_thin, meta = _patch_thin_daily_path(
thin, channel, label=f"{label}[slice{index}]"
)
if len(patched_thin) != slice_info.size:
raise _fail(label, "fat slice size changed after path patch")
buf[slice_info.offset : slice_info.offset + slice_info.size] = patched_thin
meta = {
**meta,
"fat_slice_offset": slice_info.offset,
"fat_slice_size": slice_info.size,
}
slice_meta.append(meta)
architectures = {item["architecture"] for item in slice_meta}
if architectures != {"arm64", "arm64e"}:
raise _fail(label, f"unexpected fat architectures {sorted(architectures)}")
new_path = PATH_TEMPLATE.format(channel=channel)
metadata: dict[str, object] = {
"container": "fat",
"strategy": "cfstring_retarget_to_text_padding",
"source_path": OLD_DAILY_PATH.decode("ascii"),
"path": new_path,
"slices": slice_meta,
}
return bytes(buf), metadata