#!/usr/bin/env python3 # SPDX-License-Identifier: GPL-3.0-or-later """Read-only, host-only parser for bounded PS5 SIECAF structural metadata. This module never decrypts, extracts content, invokes ps5-bar-tool, opens a network connection, or writes an output file. The byte layout is based on: * https://www.psdevwiki.com/ps5/Archive.dat * c0w-ar/ps5-bar-tool include/bar_file.h at 36d014672bc87577a6e0d750c2cccadc3fae0854 """ from __future__ import annotations import argparse from contextlib import contextmanager import hashlib import json from pathlib import Path import struct from typing import Any, BinaryIO, Iterator import zipfile MAGIC = b"SIECAF\x00\x00" HEADER = struct.Struct("<8sQiIiI16s12sIQQQ") SEGMENT_META = struct.Struct(" str: return hashlib.sha256(value).hexdigest() def _checked_add(left: int, right: int, label: str) -> int: result = left + right if left < 0 or right < 0 or result > UINT64_MAX: raise SiecafError(f"{label}: uint64 addition overflow") return result def _checked_mul(left: int, right: int, label: str) -> int: result = left * right if left < 0 or right < 0 or result > UINT64_MAX: raise SiecafError(f"{label}: uint64 multiplication overflow") return result def _read_exact(stream: BinaryIO, size: int, label: str) -> bytes: value = stream.read(size) if len(value) != size: raise SiecafError(f"{label}: truncated input") return value def _canonical_hash(records: list[dict[str, Any]]) -> str: encoded = json.dumps( records, sort_keys=True, separators=(",", ":"), ensure_ascii=True ).encode("ascii") return _sha256(encoded) def inspect_siecaf( stream: BinaryIO, source_size: int, *, source_label: str = "" ) -> dict[str, Any]: """Parse structural metadata from an already-open, read-only stream.""" if source_size < HEADER.size: return { "source": source_label, "source_size": source_size, "classification": "SIECAF_MALFORMED", "errors": ["header: truncated input"], "warnings": [], } stream.seek(0) header_raw = _read_exact(stream, HEADER.size, "header") ( magic, unknown_u64, mode, pad1, version, pad2, key, iv12, pad3, segment_count, file_offset, file_size, ) = HEADER.unpack(header_raw) header = { "magic_hex": magic.hex(), "magic_ascii": magic.rstrip(b"\x00").decode("ascii", errors="replace"), "unknown_u64": unknown_u64, "mode_i32": mode, "padding_1_u32": pad1, "version_i32": version, "padding_2_u32": pad2, "key_or_unknown_16_redacted": True, "key_or_unknown_16_sha256": _sha256(key), "iv_12_hex": iv12.hex(), "padding_3_u32": pad3, "segment_count": segment_count, "file_offset": file_offset, "file_size": file_size, "raw_sha256": _sha256(header_raw), } if magic != MAGIC: return { "source": source_label, "source_size": source_size, "classification": "SIECAF_MALFORMED", "header": header, "errors": ["magic: expected SIECAF\\0\\0"], "warnings": [], } if version not in SUPPORTED_VERSIONS: return { "source": source_label, "source_size": source_size, "classification": "SIECAF_UNSUPPORTED_VERSION", "header": header, "errors": [f"unsupported SIECAF header version: {version}"], "warnings": [], } errors: list[str] = [] warnings: list[str] = [] if segment_count == 0: errors.append("segment_count: zero") try: metadata_bytes = _checked_mul( segment_count, SEGMENT_META.size, "segment metadata table" ) hash_bytes = _checked_mul( segment_count, SECTION_HASH.size, "section hash table" ) metadata_end = _checked_add(HEADER.size, metadata_bytes, "metadata end") tables_end = _checked_add(metadata_end, hash_bytes, "tables end") declared_end = _checked_add(file_offset, file_size, "declared file end") except SiecafError as error: return { "source": source_label, "source_size": source_size, "classification": "SIECAF_MALFORMED", "header": header, "errors": [str(error)], "warnings": [], } if tables_end > source_size: errors.append("tables: outside source file") if tables_end > file_offset: errors.append("tables: overlap declared data region") if file_offset % ALIGNMENT != 0: errors.append(f"file_offset: not aligned to {ALIGNMENT}") if declared_end > source_size: errors.append("declared file range: outside source file") elif declared_end < source_size: warnings.append( f"trailing data after declared file range: {source_size - declared_end}" ) if errors: return { "source": source_label, "source_size": source_size, "classification": "SIECAF_MALFORMED", "header": header, "tables_end": tables_end, "errors": errors, "warnings": warnings, } metadata: list[dict[str, Any]] = [] for index in range(segment_count): raw = _read_exact(stream, SEGMENT_META.size, f"segment metadata {index}") ( section_id, padding_1, part_number, data_offset, aligned_length, hash_key_id, encryption_key_id, segment_iv12, segment_iv_padding, unaligned_length, ) = SEGMENT_META.unpack(raw) iv16 = segment_iv12 + struct.pack(" 1 } ) duplicate_metadata_keys = sorted( { (section_id, part_number) for section_id, part_number in metadata_keys if metadata_keys.count((section_id, part_number)) > 1 } ) repeated_metadata_ids = sorted( {item for item in metadata_ids if metadata_ids.count(item) > 1} ) duplicate_hash_ids = sorted({item for item in hash_ids if hash_ids.count(item) > 1}) if duplicate_metadata_ids: errors.append( f"duplicate metadata section ID/part keys: {duplicate_metadata_keys}" ) if duplicate_hash_ids: errors.append(f"duplicate hash section IDs: {duplicate_hash_ids}") if set(hash_ids) != set(range(segment_count)): errors.append("hash section IDs do not cover metadata table indexes") hash_by_id = { item["section_id"]: item for item in hashes if item["section_id"] not in duplicate_hash_ids } ranges: list[tuple[int, int, int]] = [] for item in metadata: section_id = item["section_id"] start = item["data_offset"] length = item["aligned_length"] unaligned = item["unaligned_length"] try: end = _checked_add(start, length, f"section {section_id} end") except SiecafError as error: errors.append(str(error)) continue if section_id < 0: errors.append(f"section {section_id}: negative ID") if start % ALIGNMENT != 0: errors.append(f"section {section_id}: unaligned data offset") if length == 0 and unaligned != 0: errors.append(f"section {section_id}: incoherent unaligned length") if length > 0 and length % ALIGNMENT != 0: errors.append(f"section {section_id}: incoherent aligned length") if length > 0 and (unaligned > length or length - unaligned >= ALIGNMENT): errors.append(f"section {section_id}: incoherent unaligned length") if start < file_offset or end > declared_end or end > source_size: errors.append(f"section {section_id}: range outside declared file data") if length > 0: ranges.append((start, end, section_id)) hash_record = hash_by_id.get(item["table_index"]) item["section_type"] = ( hash_record["section_type"] if hash_record is not None else None ) item["section_hash_128_hex"] = ( hash_record["section_hash_128_hex"] if hash_record is not None else None ) ranges.sort() overlaps: list[dict[str, int]] = [] gaps: list[dict[str, int]] = [] previous_end = file_offset previous_id = -1 for start, end, section_id in ranges: if start < previous_end: overlaps.append( { "left_section_id": previous_id, "right_section_id": section_id, "overlap_bytes": previous_end - start, } ) elif start > previous_end: gaps.append( { "after_section_id": previous_id, "before_section_id": section_id, "gap_bytes": start - previous_end, } ) if end > previous_end: previous_end = end previous_id = section_id if overlaps: errors.append(f"overlapping section ranges: {len(overlaps)}") if gaps: warnings.append(f"gaps between section ranges: {len(gaps)}") trailing_data = max(0, source_size - previous_end) if trailing_data: warnings.append(f"trailing data after final section: {trailing_data}") if previous_end < declared_end: warnings.append( f"uncovered bytes inside declared data range: {declared_end - previous_end}" ) normalized_segments = [ { key: item[key] for key in ( "section_id", "part_number", "data_offset", "aligned_length", "unaligned_length", "hash_key_id_or_algorithm_type", "encryption_key_id_or_algorithm_version", "iv_hex", ) } for item in sorted( metadata, key=lambda value: ( value["section_id"], value["part_number"], value["data_offset"], ), ) ] normalized_hashes = [ { key: item[key] for key in ("section_id", "section_type", "section_hash_128_hex") } for item in sorted( hashes, key=lambda value: (value["section_id"], value["section_type"]) ) ] normalized_layout = [ { key: item[key] for key in ( "section_id", "part_number", "data_offset", "aligned_length", "unaligned_length", "hash_key_id_or_algorithm_type", "encryption_key_id_or_algorithm_version", ) } for item in normalized_segments ] normalized_segment_table_sha256 = _canonical_hash(normalized_segments) normalized_hash_blocks_sha256 = _canonical_hash(normalized_hashes) normalized_layout_sha256 = _canonical_hash( [ { "unknown_u64": unknown_u64, "version_i32": version, "segment_count": segment_count, "file_offset": file_offset, "file_size": file_size, }, *normalized_layout, ] ) structural_fingerprint_sha256 = _canonical_hash( [ { "header_raw_sha256": header["raw_sha256"], "normalized_segment_table_sha256": (normalized_segment_table_sha256), "normalized_hash_blocks_sha256": normalized_hash_blocks_sha256, } ] ) return { "source": source_label, "source_size": source_size, "classification": ("SIECAF_MALFORMED" if errors else "SIECAF_VALID_STRUCTURE"), "header": header, "tables_end": tables_end, "table_padding_bytes": file_offset - tables_end, "segments": sorted(metadata, key=lambda item: item["table_index"]), "section_hashes": sorted(hashes, key=lambda item: item["table_index"]), "normalized_segment_table_sha256": normalized_segment_table_sha256, "normalized_hash_blocks_sha256": normalized_hash_blocks_sha256, "normalized_layout_sha256": normalized_layout_sha256, "structural_fingerprint_sha256": structural_fingerprint_sha256, "duplicate_metadata_section_ids": duplicate_metadata_ids, "duplicate_metadata_section_keys": [ {"section_id": section_id, "part_number": part_number} for section_id, part_number in duplicate_metadata_keys ], "repeated_metadata_section_ids": repeated_metadata_ids, "duplicate_hash_section_ids": duplicate_hash_ids, "overlaps": overlaps, "gaps": gaps, "trailing_data_bytes": trailing_data, "errors": errors, "warnings": warnings, } def compare_structures(left: dict[str, Any], right: dict[str, Any]) -> dict[str, Any]: """Classify two already-parsed structural fingerprints.""" if ( left.get("classification") == "SIECAF_UNSUPPORTED_VERSION" or right.get("classification") == "SIECAF_UNSUPPORTED_VERSION" ): classification = "SIECAF_UNSUPPORTED_VERSION" elif ( left.get("classification") != "SIECAF_VALID_STRUCTURE" or right.get("classification") != "SIECAF_VALID_STRUCTURE" ): classification = "SIECAF_MALFORMED" elif left.get("structural_fingerprint_sha256") == right.get( "structural_fingerprint_sha256" ): classification = "SIECAF_STRUCTURAL_EXACT" elif left.get("normalized_layout_sha256") == right.get("normalized_layout_sha256"): classification = "SIECAF_LAYOUT_MATCH_HASHES_DIFFER" else: classification = "SIECAF_LAYOUT_DIFFERENT" return { "classification": classification, "left_structural_fingerprint_sha256": left.get("structural_fingerprint_sha256"), "right_structural_fingerprint_sha256": right.get( "structural_fingerprint_sha256" ), "left_layout_sha256": left.get("normalized_layout_sha256"), "right_layout_sha256": right.get("normalized_layout_sha256"), "left_hash_blocks_sha256": left.get("normalized_hash_blocks_sha256"), "right_hash_blocks_sha256": right.get("normalized_hash_blocks_sha256"), } @contextmanager def open_source( path: Path, zip_entry: str | None ) -> Iterator[tuple[BinaryIO, int, str]]: if zip_entry is None: with path.open("rb") as stream: yield stream, path.stat().st_size, str(path) return with zipfile.ZipFile(path) as archive: info = archive.getinfo(zip_entry) with archive.open(info) as stream: yield stream, info.file_size, f"{path}::{zip_entry}" def main() -> int: parser = argparse.ArgumentParser() parser.add_argument("path", type=Path) parser.add_argument("--zip-entry") parser.add_argument("--label") parser.add_argument("--pretty", action="store_true") args = parser.parse_args() try: with open_source(args.path.resolve(), args.zip_entry) as ( stream, size, default_label, ): result = inspect_siecaf( stream, size, source_label=args.label or default_label ) except (OSError, KeyError, zipfile.BadZipFile, SiecafError) as error: result = { "source": args.label or str(args.path), "classification": "SIECAF_MALFORMED", "errors": [str(error)], "warnings": [], } print( json.dumps( result, sort_keys=True, indent=2 if args.pretty else None, ) ) return 0 if result["classification"] == "SIECAF_VALID_STRUCTURE" else 2 if __name__ == "__main__": raise SystemExit(main())