"""Safely retarget the native initial daily URL to a channel-scoped path. type-0x01 thin dylibs and the fat core (``tmp.dylib`` / erupt_flee) represent the path as an immutable CFString. The replacement is longer than the original, so it is stored in validated, file-backed padding between the Mach-O load commands and ``__text``. The CFString data pointer and length are then updated while preserving the architecture's dyld relocation encoding. Fat binaries are patched slice-by-slice in place (arm64 + arm64e). """ from __future__ import annotations import struct from dataclasses import dataclass, field from _channel_patch import validate_channel_id MH_MAGIC_64 = 0xFEEDFACF FAT_MAGIC = 0xCAFEBABE FAT_CIGAM = 0xBEBAFECA CPU_TYPE_ARM64 = 0x0100000C LC_SEGMENT_64 = 0x19 LC_UUID = 0x1B LC_DYLD_INFO_ONLY = 0x80000022 LC_DYLD_CHAINED_FIXUPS = 0x80000034 OLD_DAILY_PATH = b"/sync/daily.html" PATH_TEMPLATE = "/channel/{channel}/sync/daily.html" _POINTER_SIZE = 8 @dataclass(frozen=True) class _Profile: architecture: str uuid: str path_offset: int cfstring_offset: int cave_offset: int raw_reference: int # These offsets are not used blindly: _inspect_source derives every location # from Mach-O sections/relocations and requires it to equal the matching UUID's # known layout before any bytes are changed. _PROFILES = { # type-0x01 secondary packs (thin) "73827b4262ba3a5989ada4fe6f67e266": _Profile( architecture="arm64", uuid="73827b4262ba3a5989ada4fe6f67e266", path_offset=0x3C9F8, cfstring_offset=0x45488, cave_offset=0x10B0, raw_reference=0x3C9F8, ), "ec21ac8ad85333d5a4f9723ff3f31a2c": _Profile( architecture="arm64e", uuid="ec21ac8ad85333d5a4f9723ff3f31a2c", path_offset=0x409F0, cfstring_offset=0x49A20, cave_offset=0x10F0, raw_reference=0x00100000000409F0, ), # sync core tmp.dylib slices (fat: arm64 + arm64e) "4262fd020de4337e9a690626a28fa4e3": _Profile( architecture="arm64", uuid="4262fd020de4337e9a690626a28fa4e3", path_offset=0xE1AB8, cfstring_offset=0x1134D8, cave_offset=0x1180, raw_reference=0xE1AB8, ), "8a796924959b310481a4ace808db8521": _Profile( architecture="arm64e", uuid="8a796924959b310481a4ace808db8521", path_offset=0xF1FE4, cfstring_offset=0x123D80, cave_offset=0x11C0, raw_reference=0x00100000000F1FE4, ), } @dataclass(frozen=True) class _FatSlice: offset: int size: int @dataclass class _Section: segment: str name: str addr: int size: int offset: int @dataclass class _Segment: name: str vmaddr: int vmsize: int fileoff: int filesize: int initprot: int sections: list[_Section] = field(default_factory=list) @dataclass class _MachO: data: bytes cpu_subtype: int load_end: int uuid: str segments: list[_Segment] dyld_rebase: tuple[int, int] | None chained_fixups: tuple[int, int] | None def section(self, segment: str, name: str) -> _Section: hits = [ section for item in self.segments for section in item.sections if section.segment == segment and section.name == name ] if len(hits) != 1: raise ValueError(f"expected one {segment},{name} section; found {len(hits)}") return hits[0] def segment(self, name: str) -> _Segment: hits = [segment for segment in self.segments if segment.name == name] if len(hits) != 1: raise ValueError(f"expected one {name} segment; found {len(hits)}") return hits[0] @dataclass(frozen=True) class _Inspection: macho: _MachO profile: _Profile path_vmaddr: int cfstring_offset: int reference_offset: int cave_vmaddr: int def _fail(label: str, message: str) -> SystemExit: return SystemExit(f"{label}: {message}. Refusing unsafe initial daily path patch.") def _name(raw: bytes) -> str: return raw.split(b"\x00", 1)[0].decode("ascii") def _parse_macho(data: bytes, *, label: str) -> _MachO: if len(data) < 32: raise _fail(label, "truncated Mach-O header") magic, cputype, cpusubtype, _filetype, ncmds, sizeofcmds = struct.unpack_from( " len(data): raise _fail(label, "load commands exceed file size") segments: list[_Segment] = [] uuid = "" dyld_rebase: tuple[int, int] | None = None chained_fixups: tuple[int, int] | None = None offset = 32 for _ in range(ncmds): if offset + 8 > load_end: raise _fail(label, "truncated load command") cmd, cmdsize = struct.unpack_from(" load_end: raise _fail(label, "invalid load command size") if cmd == LC_SEGMENT_64: if cmdsize < 72: raise _fail(label, "truncated LC_SEGMENT_64") segname = _name(data[offset + 8 : offset + 24]) vmaddr, vmsize, fileoff, filesize = struct.unpack_from( " cmdsize: raise _fail(label, f"{segname} section table exceeds load command") segment = _Segment( segname, vmaddr, vmsize, fileoff, filesize, initprot ) section_offset = offset + 72 for _section_index in range(nsects): sectname = _name(data[section_offset : section_offset + 16]) section_segment = _name( data[section_offset + 16 : section_offset + 32] ) addr, size = struct.unpack_from(" len(data): raise _fail(label, f"{section_segment},{sectname} exceeds file size") segment.sections.append( _Section(section_segment, sectname, addr, size, file_offset) ) section_offset += 80 segments.append(segment) elif cmd == LC_UUID: if cmdsize != 24 or uuid: raise _fail(label, "invalid or duplicate LC_UUID") uuid = data[offset + 8 : offset + 24].hex() elif cmd == LC_DYLD_INFO_ONLY: if cmdsize != 48: raise _fail(label, "invalid LC_DYLD_INFO_ONLY") rebase_off, rebase_size = struct.unpack_from(" int: hits = [ segment.vmaddr + offset - segment.fileoff for segment in macho.segments if segment.fileoff <= offset < segment.fileoff + segment.filesize ] if len(hits) != 1: raise ValueError(f"file offset {offset:#x} is not in one file-backed segment") return hits[0] def _file_offset_for_vmaddr(macho: _MachO, address: int) -> int: hits = [ segment.fileoff + address - segment.vmaddr for segment in macho.segments if ( segment.vmaddr <= address < segment.vmaddr + segment.filesize and segment.filesize ) ] if len(hits) != 1: raise ValueError(f"vmaddr {address:#x} is not in one file-backed segment") return hits[0] def _read_uleb(data: bytes, offset: int, end: int) -> tuple[int, int]: value = 0 shift = 0 while offset < end and shift < 64: byte = data[offset] offset += 1 value |= (byte & 0x7F) << shift if not byte & 0x80: return value, offset shift += 7 raise ValueError("invalid ULEB128") def _classic_rebase_locations(macho: _MachO) -> set[int]: if macho.dyld_rebase is None: raise ValueError("missing LC_DYLD_INFO_ONLY") start, size = macho.dyld_rebase end = start + size if start < macho.load_end or end > len(macho.data): raise ValueError("invalid dyld rebase stream") segment_index = -1 address = 0 locations: set[int] = set() offset = start def record(count: int, skip: int = 0) -> None: nonlocal address if segment_index < 0 or segment_index >= len(macho.segments): raise ValueError("rebase before segment selection") segment = macho.segments[segment_index] for _ in range(count): if not (segment.vmaddr <= address < segment.vmaddr + segment.vmsize): raise ValueError("rebase address exceeds segment") locations.add(address) address += _POINTER_SIZE + skip while offset < end: byte = macho.data[offset] offset += 1 opcode, immediate = byte & 0xF0, byte & 0x0F if opcode == 0x00: break if opcode == 0x10: # SET_TYPE_IMM if immediate != 1: raise ValueError("unsupported non-pointer rebase type") elif opcode == 0x20: # SET_SEGMENT_AND_OFFSET_ULEB segment_index = immediate delta, offset = _read_uleb(macho.data, offset, end) if segment_index >= len(macho.segments): raise ValueError("invalid rebase segment index") address = macho.segments[segment_index].vmaddr + delta elif opcode == 0x30: # ADD_ADDR_ULEB delta, offset = _read_uleb(macho.data, offset, end) address += delta elif opcode == 0x40: # ADD_ADDR_IMM_SCALED address += immediate * _POINTER_SIZE elif opcode == 0x50: # DO_REBASE_IMM_TIMES record(immediate) elif opcode == 0x60: # DO_REBASE_ULEB_TIMES count, offset = _read_uleb(macho.data, offset, end) record(count) elif opcode == 0x70: # DO_REBASE_ADD_ADDR_ULEB skip, offset = _read_uleb(macho.data, offset, end) record(1, skip) elif opcode == 0x80: # DO_REBASE_ULEB_TIMES_SKIPPING_ULEB count, offset = _read_uleb(macho.data, offset, end) skip, offset = _read_uleb(macho.data, offset, end) record(count, skip) else: raise ValueError(f"unsupported rebase opcode {opcode:#x}") return locations def _arm64e_target(raw: int) -> int | None: # DYLD_CHAINED_PTR_ARM64E non-auth rebase: # target:43, high8:8, next:11, bind:1, auth:1. if (raw >> 63) & 1 or (raw >> 62) & 1: return None target = raw & ((1 << 43) - 1) high8 = (raw >> 43) & 0xFF return target | (high8 << 56) def _chained_arm64e_locations(macho: _MachO) -> set[int]: if macho.chained_fixups is None: raise ValueError("missing LC_DYLD_CHAINED_FIXUPS") dataoff, datasize = macho.chained_fixups end = dataoff + datasize if dataoff < macho.load_end or end > len(macho.data) or datasize < 28: raise ValueError("invalid chained-fixups payload") ( version, starts_offset, _imports_offset, _symbols_offset, _imports_count, _imports_format, _symbols_format, ) = struct.unpack_from("<7I", macho.data, dataoff) if version != 0: raise ValueError(f"unsupported chained-fixups version {version}") starts = dataoff + starts_offset if starts + 4 > end: raise ValueError("invalid chained starts offset") segment_count = struct.unpack_from(" end: raise ValueError("truncated chained segment table") locations: set[int] = set() for segment_index in range(segment_count): relative = struct.unpack_from(" end: raise ValueError("truncated chained segment info") size, page_size, pointer_format, segment_offset, _max_ptr, page_count = ( struct.unpack_from(" end: raise ValueError("invalid chained segment info size") segment = macho.segments[segment_index] if segment_offset != segment.vmaddr: raise ValueError("chained segment offset differs from vmaddr") for page_index in range(page_count): page_start = struct.unpack_from( " segment.fileoff + segment.filesize: raise ValueError("chained pointer exceeds file-backed segment") locations.add(address) raw = struct.unpack_from("> 51) & 0x7FF if not next_delta: break address += next_delta * 8 return locations def _inspect_source(data: bytes, *, replacement_size: int, label: str) -> _Inspection: try: macho = _parse_macho(data, label=label) profile = _PROFILES.get(macho.uuid) if profile is None: raise ValueError(f"unsupported Mach-O UUID {macho.uuid}") base_subtype = macho.cpu_subtype & 0x00FFFFFF architecture = "arm64e" if base_subtype == 2 else "arm64" if base_subtype == 0 else "" if architecture != profile.architecture: raise ValueError( f"CPU subtype resolves to {architecture or base_subtype}, expected " f"{profile.architecture}" ) cstring = macho.section("__TEXT", "__cstring") needle = OLD_DAILY_PATH + b"\x00" region = data[cstring.offset : cstring.offset + cstring.size] relative_hits = [] start = 0 while True: hit = region.find(needle, start) if hit < 0: break relative_hits.append(hit) start = hit + 1 if len(relative_hits) != 1: raise ValueError(f"source daily path hits={len(relative_hits)}, expected 1") path_offset = cstring.offset + relative_hits[0] path_vmaddr = cstring.addr + relative_hits[0] cfstring = macho.section("__DATA_CONST", "__cfstring") if cfstring.size % 32: raise ValueError("__cfstring size is not record-aligned") candidates: list[tuple[int, int]] = [] for record_offset in range( cfstring.offset, cfstring.offset + cfstring.size, 32 ): raw = struct.unpack_from(" list[_FatSlice] | None: if len(data) < 8: return None magic = struct.unpack_from(">I", data, 0)[0] if magic not in (FAT_MAGIC, FAT_CIGAM): return None nfat = struct.unpack_from(">I", data, 4)[0] if nfat < 1 or 8 + nfat * 20 > len(data): raise _fail(label, "invalid fat header") slices: list[_FatSlice] = [] for index in range(nfat): _cputype, _cpusubtype, offset, size, _align = struct.unpack_from( ">IIIII", data, 8 + index * 20 ) if offset < 0 or size < 1 or offset + size > len(data): raise _fail(label, f"fat slice {index} exceeds file size") slices.append(_FatSlice(offset, size)) return slices def _patch_thin_daily_path( data: bytes, channel: str, *, label: str ) -> tuple[bytes, dict[str, object]]: new_path = PATH_TEMPLATE.format(channel=channel).encode("ascii") encoded = new_path + b"\x00" inspection = _inspect_source(data, replacement_size=len(encoded), label=label) buf = bytearray(data) cave_offset = inspection.profile.cave_offset buf[cave_offset : cave_offset + len(encoded)] = encoded raw = struct.unpack_from("= 1 << 43: raise _fail(label, "arm64e cave target exceeds chained pointer range") new_raw = (raw & ~((1 << 43) - 1)) | inspection.cave_vmaddr struct.pack_into(" tuple[bytes, dict[str, object]]: """Patch the initial daily CFString and return bytes plus manifest metadata. Accepts thin type-0x01 dylibs or the fat core ``tmp.dylib``. """ channel = validate_channel_id(channel, name="channel") slices = _fat_slices(data, label=label) if slices is None: return _patch_thin_daily_path(data, channel, label=label) if len(slices) != 2: raise _fail(label, f"expected fat arm64+arm64e (2 slices), found {len(slices)}") buf = bytearray(data) slice_meta: list[dict[str, object]] = [] for index, slice_info in enumerate(slices): thin = bytes(buf[slice_info.offset : slice_info.offset + slice_info.size]) patched_thin, meta = _patch_thin_daily_path( thin, channel, label=f"{label}[slice{index}]" ) if len(patched_thin) != slice_info.size: raise _fail(label, "fat slice size changed after path patch") buf[slice_info.offset : slice_info.offset + slice_info.size] = patched_thin meta = { **meta, "fat_slice_offset": slice_info.offset, "fat_slice_size": slice_info.size, } slice_meta.append(meta) architectures = {item["architecture"] for item in slice_meta} if architectures != {"arm64", "arm64e"}: raise _fail(label, f"unexpected fat architectures {sorted(architectures)}") new_path = PATH_TEMPLATE.format(channel=channel) metadata: dict[str, object] = { "container": "fat", "strategy": "cfstring_retarget_to_text_padding", "source_path": OLD_DAILY_PATH.decode("ascii"), "path": new_path, "slices": slice_meta, } return bytes(buf), metadata