581 lines
24 KiB
Python
581 lines
24 KiB
Python
"""T-103 原始规格面板证据的本机确定性隐私脱敏。
|
||
|
||
此模块只处理人工采集的本地文件:不连接设备、不识别规格;仅可按已取证的固定
|
||
几何和严格格式,将跨隐私边界的价格叶节点投影到派生 XML。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from dataclasses import dataclass
|
||
from hashlib import sha256
|
||
import json
|
||
from pathlib import Path
|
||
import re
|
||
import shutil
|
||
from typing import Any
|
||
from uuid import uuid4
|
||
from xml.etree import ElementTree
|
||
|
||
from PIL import Image, ImageDraw, UnidentifiedImageError
|
||
|
||
from ..pdd.product_url import ProductUrl, ProductUrlError, parse_product_url
|
||
from ..pdd.sku_panel_state import HUMAN_DECLARED_STATES
|
||
|
||
|
||
SANITIZER_VERSION = "t103-privacy-v5"
|
||
EXPECTED_GOODS_ID = "937122477375"
|
||
EXPECTED_PDD_VERSION = "8.17.0"
|
||
EXPECTED_DEVICE_MODEL = "PKG110"
|
||
EXPECTED_ANDROID_VERSION = "16"
|
||
EXPECTED_SCREENSHOT_WIDTH = 1080
|
||
EXPECTED_SCREENSHOT_HEIGHT = 2376
|
||
EXPECTED_XML_WIDTH = 1080
|
||
EXPECTED_XML_HEIGHT = 2376
|
||
_ARTIFACT_FILES = ("screenshot.png", "hierarchy.xml")
|
||
_SHA256_RE = re.compile(r"[0-9a-f]{64}\Z")
|
||
_BOUNDS_RE = re.compile(r"\[(-?\d+),(-?\d+)\]\[(-?\d+),(-?\d+)\]\Z")
|
||
_FULL_PHONE_RE = re.compile(r"(?:\+?86)?1[3-9]\d{9}")
|
||
_MASKED_PHONE_RE = re.compile(r"1[3-9]\d\*{4}\d{4}")
|
||
_MASK_TRANSLATION = str.maketrans({"*": "*", "•": "*", "·": "*", "×": "*", "x": "*", "X": "*"})
|
||
_SEPARATOR_RE = re.compile(r"[\s\-‐‑‒–—―()()]+")
|
||
# 这两个槽位来自 T-103 当前第一态、1080×2376 XML 坐标的人工审查。它们不是通用
|
||
# 页面判据;坐标、文本或结构任何变化都停止发布,交由人重新取证。
|
||
_CROSSING_PRICE_SLOTS = {
|
||
(396, 503, 712, 570): "[396,503][712,570]",
|
||
(730, 503, 895, 570): "[730,503][895,570]",
|
||
}
|
||
_CROSSING_PRICE_BOUNDS = frozenset(_CROSSING_PRICE_SLOTS)
|
||
_PRICE_PROJECTION_ATTRIBUTES = (
|
||
"bounds",
|
||
"text",
|
||
"package",
|
||
"class",
|
||
"clickable",
|
||
"enabled",
|
||
"visible-to-user",
|
||
)
|
||
# 仅接受普通 ASCII 空格,且每个可分隔位置最多一个;禁止换行、折扣、支付/提交文案和
|
||
# 任何其它字符。前缀捕获组用于区分当前价与至多一个划线/原价候选。
|
||
# T-103 人工在 live 规格面板确认当前价槽的完整非敏感前缀仅为“快卖完”;不得兼容
|
||
# 未取证的“快卖光”或其它相近文案。
|
||
_CROSSING_PRICE_TEXT_RE = re.compile(r" {0,1}(?:(快卖完) {0,1})?[¥¥] {0,1}[1-9]\d*\.\d{2} {0,1}\Z")
|
||
_CROSSING_PRICE_PREFIX_RE = re.compile(r" {0,1}(?:快卖完 {0,1})?[¥¥] {0,1}[1-9]\d*\.\d{2} {0,1}")
|
||
_CROSSING_PRICE_ALLOWED_CHARACTERS = frozenset(" 快卖完¥¥0123456789.")
|
||
|
||
|
||
class SkuEvidenceSanitizationError(RuntimeError):
|
||
"""原始证据不能被安全地发布为派生证据。"""
|
||
|
||
|
||
@dataclass
|
||
class _CleanupStats:
|
||
"""仅记录节点数量,供派生 manifest 审计;不记录任何页面文本。"""
|
||
|
||
removed_nodes: int = 0
|
||
cleared_crossing_nodes: int = 0
|
||
preserved_crossing_price_nodes: int = 0
|
||
retained_below_nodes: int = 0
|
||
max_right: int = 0
|
||
max_bottom: int = 0
|
||
current_price_candidates: int = 0
|
||
original_price_candidates: int = 0
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class PrivacyMaskConfig:
|
||
"""仅描述已人工确认的隐私几何区域,绝不承担页面或规格判据。"""
|
||
|
||
version: str
|
||
screenshot_width: int
|
||
screenshot_height: int
|
||
xml_width: int
|
||
xml_height: int
|
||
privacy_top: int
|
||
|
||
|
||
PRIVACY_MASK_CONFIG = PrivacyMaskConfig(
|
||
version=SANITIZER_VERSION,
|
||
screenshot_width=EXPECTED_SCREENSHOT_WIDTH,
|
||
screenshot_height=EXPECTED_SCREENSHOT_HEIGHT,
|
||
xml_width=EXPECTED_XML_WIDTH,
|
||
xml_height=EXPECTED_XML_HEIGHT,
|
||
# 主审在原始截图确认 y < 540 为收货/手机号区域;整宽遮罩优先保护隐私而非保留版面。
|
||
privacy_top=540,
|
||
)
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class SkuEvidenceSanitizationResult:
|
||
"""已经原子发布的派生证据位置。"""
|
||
|
||
output_directory: Path
|
||
manifest_path: Path
|
||
screenshot_path: Path
|
||
hierarchy_path: Path
|
||
|
||
|
||
def sanitize_sku_panel_evidence(raw_directory: Path, output_directory: Path) -> SkuEvidenceSanitizationResult:
|
||
"""校验 raw 三文件,并发布同级 ``derived`` 的脱敏副本。
|
||
|
||
目标已存在时在读取任何输入前拒绝,避免混入旧派生物或覆盖人工保留文件。
|
||
"""
|
||
|
||
raw = Path(raw_directory)
|
||
target = Path(output_directory)
|
||
_validate_directories(raw, target)
|
||
if target.exists():
|
||
raise SkuEvidenceSanitizationError("派生证据目录已存在,拒绝覆盖。")
|
||
|
||
staging: Path | None = None
|
||
try:
|
||
source_manifest_path = _required_file(raw, "manifest.json")
|
||
source_screenshot_path = _required_file(raw, "screenshot.png")
|
||
source_hierarchy_path = _required_file(raw, "hierarchy.xml")
|
||
manifest = _read_source_manifest(source_manifest_path)
|
||
link, state, source_hashes = _validate_source_manifest(manifest)
|
||
_verify_source_hashes(source_screenshot_path, source_hierarchy_path, source_hashes)
|
||
|
||
staging = raw.parent / f".derived.staging-{uuid4().hex}"
|
||
staging.mkdir()
|
||
derived_screenshot_path = staging / "screenshot.png"
|
||
_sanitize_screenshot(source_screenshot_path, derived_screenshot_path)
|
||
derived_hierarchy_path = staging / "hierarchy.xml"
|
||
cleanup_stats = _sanitize_hierarchy(source_hierarchy_path, derived_hierarchy_path)
|
||
|
||
derived_manifest_path = staging / "manifest.json"
|
||
derived_manifest_path.write_text(
|
||
json.dumps(
|
||
_derived_manifest(
|
||
manifest,
|
||
link,
|
||
state,
|
||
source_manifest_path,
|
||
source_screenshot_path,
|
||
source_hierarchy_path,
|
||
derived_screenshot_path,
|
||
derived_hierarchy_path,
|
||
cleanup_stats,
|
||
),
|
||
ensure_ascii=False,
|
||
indent=2,
|
||
sort_keys=True,
|
||
)
|
||
+ "\n",
|
||
encoding="utf-8",
|
||
)
|
||
_publish_staging(staging, target)
|
||
except SkuEvidenceSanitizationError:
|
||
_clean_staging(staging)
|
||
raise
|
||
except (OSError, ValueError, ElementTree.ParseError, UnidentifiedImageError) as error:
|
||
_clean_staging(staging)
|
||
# 原始异常可能含文件路径、JSON/XML 文本或其他敏感内容,不能向 CLI/日志传播。
|
||
raise SkuEvidenceSanitizationError("原始证据无法安全脱敏,未发布任何派生产物。") from error
|
||
except Exception as error:
|
||
_clean_staging(staging)
|
||
raise SkuEvidenceSanitizationError("原始证据脱敏未完成,未发布任何派生产物。") from error
|
||
|
||
return SkuEvidenceSanitizationResult(
|
||
output_directory=target,
|
||
manifest_path=target / "manifest.json",
|
||
screenshot_path=target / "screenshot.png",
|
||
hierarchy_path=target / "hierarchy.xml",
|
||
)
|
||
|
||
|
||
def _validate_directories(raw: Path, target: Path) -> None:
|
||
if raw.name != "raw" or not raw.is_dir():
|
||
raise SkuEvidenceSanitizationError("原始证据目录必须是存在的 raw 目录。")
|
||
if target.name != "derived" or target.parent != raw.parent:
|
||
raise SkuEvidenceSanitizationError("派生证据目录必须是 raw 同级的 derived 目录。")
|
||
|
||
|
||
def _required_file(raw: Path, filename: str) -> Path:
|
||
candidate = raw / filename
|
||
if not candidate.is_file():
|
||
raise SkuEvidenceSanitizationError("原始证据文件集合不完整。")
|
||
return candidate
|
||
|
||
|
||
def _read_source_manifest(path: Path) -> dict[str, Any]:
|
||
try:
|
||
value = json.loads(path.read_text(encoding="utf-8"))
|
||
except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error:
|
||
raise SkuEvidenceSanitizationError("原始证据 manifest 无效。") from error
|
||
if not isinstance(value, dict):
|
||
raise SkuEvidenceSanitizationError("原始证据 manifest 结构无效。")
|
||
return value
|
||
|
||
|
||
def _validate_source_manifest(manifest: dict[str, Any]) -> tuple[ProductUrl, str, dict[str, str]]:
|
||
product = manifest.get("product")
|
||
device = manifest.get("device")
|
||
state = manifest.get("human_declared_state")
|
||
if manifest.get("schema_version") != 1 or not isinstance(product, dict) or not isinstance(device, dict):
|
||
raise SkuEvidenceSanitizationError("原始证据 manifest 缺少必要元数据。")
|
||
canonical_url = product.get("canonical_url")
|
||
goods_id = product.get("goods_id")
|
||
try:
|
||
link = parse_product_url(canonical_url)
|
||
except ProductUrlError as error:
|
||
raise SkuEvidenceSanitizationError("原始证据商品元数据不匹配。") from error
|
||
if (
|
||
link.goods_id != EXPECTED_GOODS_ID
|
||
or goods_id != EXPECTED_GOODS_ID
|
||
or device.get("model") != EXPECTED_DEVICE_MODEL
|
||
or device.get("android_version") != EXPECTED_ANDROID_VERSION
|
||
or device.get("pdd_version") != EXPECTED_PDD_VERSION
|
||
or device.get("pdd_package") != "com.xunmeng.pinduoduo"
|
||
or not isinstance(state, str)
|
||
or state not in HUMAN_DECLARED_STATES
|
||
):
|
||
raise SkuEvidenceSanitizationError("原始证据元数据与脱敏配置不匹配。")
|
||
return link, state, _artifact_hashes(manifest)
|
||
|
||
|
||
def _artifact_hashes(manifest: dict[str, Any]) -> dict[str, str]:
|
||
artifacts = manifest.get("artifacts")
|
||
if not isinstance(artifacts, list):
|
||
raise SkuEvidenceSanitizationError("原始证据 manifest 缺少文件校验信息。")
|
||
hashes: dict[str, str] = {}
|
||
for artifact in artifacts:
|
||
if not isinstance(artifact, dict):
|
||
raise SkuEvidenceSanitizationError("原始证据 manifest 文件校验信息无效。")
|
||
path = artifact.get("path")
|
||
digest = artifact.get("sha256")
|
||
if path not in _ARTIFACT_FILES or path in hashes or not isinstance(digest, str) or not _SHA256_RE.fullmatch(digest):
|
||
raise SkuEvidenceSanitizationError("原始证据 manifest 文件校验信息无效。")
|
||
hashes[path] = digest
|
||
if set(hashes) != set(_ARTIFACT_FILES):
|
||
raise SkuEvidenceSanitizationError("原始证据 manifest 文件校验信息不完整。")
|
||
return hashes
|
||
|
||
|
||
def _verify_source_hashes(screenshot_path: Path, hierarchy_path: Path, expected: dict[str, str]) -> None:
|
||
if (
|
||
_sha256_file(screenshot_path) != expected["screenshot.png"]
|
||
or _sha256_file(hierarchy_path) != expected["hierarchy.xml"]
|
||
):
|
||
raise SkuEvidenceSanitizationError("原始证据文件校验失败。")
|
||
|
||
|
||
def _sanitize_screenshot(source: Path, target: Path) -> None:
|
||
try:
|
||
with Image.open(source) as image:
|
||
image.load()
|
||
if image.format != "PNG" or image.size != (
|
||
PRIVACY_MASK_CONFIG.screenshot_width,
|
||
PRIVACY_MASK_CONFIG.screenshot_height,
|
||
):
|
||
raise SkuEvidenceSanitizationError("原始截图分辨率或格式与脱敏配置不匹配。")
|
||
sanitized = image.convert("RGBA")
|
||
except SkuEvidenceSanitizationError:
|
||
raise
|
||
except (OSError, UnidentifiedImageError) as error:
|
||
raise SkuEvidenceSanitizationError("原始截图无效。") from error
|
||
|
||
# 用不透明黑色覆盖 y < 540,保证截图与 XML 使用相同的隐私几何边界。
|
||
ImageDraw.Draw(sanitized).rectangle(
|
||
(0, 0, PRIVACY_MASK_CONFIG.screenshot_width - 1, PRIVACY_MASK_CONFIG.privacy_top - 1),
|
||
fill=(0, 0, 0, 255),
|
||
)
|
||
sanitized.save(target, format="PNG", optimize=False, compress_level=9)
|
||
|
||
|
||
def _sanitize_hierarchy(source: Path, target: Path) -> _CleanupStats:
|
||
try:
|
||
root = ElementTree.parse(source).getroot()
|
||
except (OSError, ElementTree.ParseError) as error:
|
||
raise SkuEvidenceSanitizationError("原始节点树无效。") from error
|
||
if root.tag != "hierarchy":
|
||
raise SkuEvidenceSanitizationError("原始节点树结构不匹配。")
|
||
if not list(root):
|
||
raise SkuEvidenceSanitizationError("原始节点树结构不匹配。")
|
||
stats = _CleanupStats()
|
||
_clear_node_text(root)
|
||
for child in list(root):
|
||
_sanitize_node(root, child, stats)
|
||
_require_expected_xml_coordinate_space(stats)
|
||
if stats.removed_nodes < 1 or stats.retained_below_nodes < 1:
|
||
raise SkuEvidenceSanitizationError("原始节点树未满足隐私几何结构。")
|
||
_require_safe_crossing_price_projection(stats)
|
||
if _contains_phone(root):
|
||
raise SkuEvidenceSanitizationError("派生节点树仍包含手机号,拒绝发布。")
|
||
ElementTree.ElementTree(root).write(target, encoding="utf-8", xml_declaration=True)
|
||
return stats
|
||
|
||
|
||
def _sanitize_node(parent: ElementTree.Element, node: ElementTree.Element, stats: _CleanupStats) -> None:
|
||
if node.tag != "node":
|
||
raise SkuEvidenceSanitizationError("原始节点树结构不匹配。")
|
||
bounds = _parse_bounds(node.get("bounds"))
|
||
_observe_bounds(stats, bounds)
|
||
position = _vertical_position(bounds)
|
||
if position == "private":
|
||
# 私有带内的父节点不可以悄然包含下方子节点,否则会把仍需审计的下方内容一起丢失。
|
||
for descendant in node.iter("node"):
|
||
descendant_bounds = _parse_bounds(descendant.get("bounds"))
|
||
_observe_bounds(stats, descendant_bounds)
|
||
if _vertical_position(descendant_bounds) != "private":
|
||
raise SkuEvidenceSanitizationError("原始节点树 bounds 结构不匹配。")
|
||
stats.removed_nodes += sum(1 for _ in node.iter("node"))
|
||
parent.remove(node)
|
||
return
|
||
if position == "crossing":
|
||
if bounds in _CROSSING_PRICE_BOUNDS and node.get("text"):
|
||
_project_crossing_price_node(node, bounds, stats)
|
||
else:
|
||
# 全屏/跨界容器可保留其下方子节点,但自身所有属性和文本都可能含地址或手机号。
|
||
_clear_node_text(node)
|
||
stats.cleared_crossing_nodes += 1
|
||
else:
|
||
stats.retained_below_nodes += 1
|
||
for child in list(node):
|
||
_sanitize_node(node, child, stats)
|
||
|
||
|
||
def _parse_bounds(value: object) -> tuple[int, int, int, int]:
|
||
if not isinstance(value, str):
|
||
raise SkuEvidenceSanitizationError("原始节点树 bounds 缺失或无效。")
|
||
match = _BOUNDS_RE.fullmatch(value)
|
||
if match is None:
|
||
raise SkuEvidenceSanitizationError("原始节点树 bounds 缺失或无效。")
|
||
left, top, right, bottom = (int(group) for group in match.groups())
|
||
if not (0 <= left < right and 0 <= top < bottom):
|
||
raise SkuEvidenceSanitizationError("原始节点树 bounds 缺失或无效。")
|
||
return left, top, right, bottom
|
||
|
||
|
||
def _observe_bounds(stats: _CleanupStats, bounds: tuple[int, int, int, int]) -> None:
|
||
_, _, right, bottom = bounds
|
||
stats.max_right = max(stats.max_right, right)
|
||
stats.max_bottom = max(stats.max_bottom, bottom)
|
||
|
||
|
||
def _require_expected_xml_coordinate_space(stats: _CleanupStats) -> None:
|
||
if (
|
||
stats.max_right != PRIVACY_MASK_CONFIG.xml_width
|
||
or stats.max_bottom != PRIVACY_MASK_CONFIG.xml_height
|
||
):
|
||
raise SkuEvidenceSanitizationError(
|
||
f"原始节点树坐标范围不匹配(observed {stats.max_right}x{stats.max_bottom})。"
|
||
)
|
||
|
||
|
||
def _vertical_position(bounds: tuple[int, int, int, int]) -> str:
|
||
_, top, _, bottom = bounds
|
||
if bottom <= PRIVACY_MASK_CONFIG.privacy_top:
|
||
return "private"
|
||
if top >= PRIVACY_MASK_CONFIG.privacy_top:
|
||
return "below"
|
||
return "crossing"
|
||
|
||
|
||
def _project_crossing_price_node(
|
||
node: ElementTree.Element,
|
||
bounds: tuple[int, int, int, int],
|
||
stats: _CleanupStats,
|
||
) -> None:
|
||
"""投影唯一允许的跨界价格叶节点;任何结构漂移一律拒绝发布。"""
|
||
|
||
if (
|
||
len(node) != 0
|
||
or node.get("package") != "com.xunmeng.pinduoduo"
|
||
or node.get("class") != "android.widget.TextView"
|
||
or node.get("clickable") != "false"
|
||
or node.get("enabled") != "true"
|
||
or node.get("visible-to-user") != "true"
|
||
):
|
||
raise SkuEvidenceSanitizationError("跨界价格节点结构不匹配,拒绝发布。")
|
||
text = node.get("text")
|
||
if text is None:
|
||
raise SkuEvidenceSanitizationError("跨界价格节点文本不匹配,拒绝发布。")
|
||
match = _CROSSING_PRICE_TEXT_RE.fullmatch(text)
|
||
if match is None:
|
||
raise _crossing_price_text_mismatch_error(bounds, text)
|
||
|
||
# 只有这七项经上述检查后可进入派生 XML;尤其不复制 content-desc、resource-id 等原始属性。
|
||
node.attrib = {attribute: node.attrib[attribute] for attribute in _PRICE_PROJECTION_ATTRIBUTES}
|
||
node.text = None
|
||
node.tail = None
|
||
stats.preserved_crossing_price_nodes += 1
|
||
if match.group(1) is not None:
|
||
stats.current_price_candidates += 1
|
||
else:
|
||
stats.original_price_candidates += 1
|
||
|
||
|
||
def _crossing_price_text_mismatch_error(
|
||
bounds: tuple[int, int, int, int],
|
||
text: str,
|
||
) -> SkuEvidenceSanitizationError:
|
||
"""仅输出固定槽位与 reason,避免将任意 raw 正文带入 CLI 或日志。"""
|
||
|
||
reason = _crossing_price_text_mismatch_reason(text)
|
||
slot = _CROSSING_PRICE_SLOTS[bounds]
|
||
return SkuEvidenceSanitizationError(f"跨界价格节点文本不匹配:slot={slot};reason={reason}。")
|
||
|
||
|
||
def _crossing_price_text_mismatch_reason(text: str) -> str:
|
||
"""将未匹配文本归类为受控枚举;返回值绝不包含原始片段。"""
|
||
|
||
if "\r" in text or "\n" in text:
|
||
return "newline"
|
||
if any(character.isspace() and character != " " for character in text):
|
||
return "non_ascii_whitespace"
|
||
if any(marker in text for marker in ("提交订单", "支付", "下单", "优惠")):
|
||
return "extra_or_order"
|
||
without_one_leading_space = text[1:] if text.startswith(" ") else text
|
||
if without_one_leading_space.startswith("快要抢光"):
|
||
return "observed_prefix_kuaiyaoqiangguang"
|
||
if without_one_leading_space.startswith("快") and not without_one_leading_space.startswith("快卖完"):
|
||
return "known_prefix_missing"
|
||
if "¥" not in text and "¥" not in text:
|
||
return "currency_missing"
|
||
if _CROSSING_PRICE_PREFIX_RE.match(text) is not None:
|
||
return "extra_or_order"
|
||
if any(character not in _CROSSING_PRICE_ALLOWED_CHARACTERS for character in text):
|
||
return "forbidden_characters"
|
||
return "amount_shape"
|
||
|
||
|
||
def _require_safe_crossing_price_projection(stats: _CleanupStats) -> None:
|
||
"""当前价必须唯一;原价仅可选且唯一,避免把任意金额释放为价格证据。"""
|
||
|
||
if (
|
||
stats.current_price_candidates != 1
|
||
or stats.original_price_candidates > 1
|
||
or stats.preserved_crossing_price_nodes != stats.current_price_candidates + stats.original_price_candidates
|
||
):
|
||
raise SkuEvidenceSanitizationError("跨界价格候选不唯一或缺失,拒绝发布。")
|
||
|
||
|
||
def _clear_node_text(node: ElementTree.Element) -> None:
|
||
node.attrib = {"bounds": node.attrib["bounds"]} if "bounds" in node.attrib else {}
|
||
node.text = None
|
||
node.tail = None
|
||
|
||
|
||
def _contains_phone(root: ElementTree.Element) -> bool:
|
||
"""逐项与跨节点复检电话,避免分隔符、遮罩字符或节点切分绕过。"""
|
||
|
||
all_values: list[str] = []
|
||
content_values: list[str] = []
|
||
for element in root.iter():
|
||
if element.text:
|
||
all_values.append(element.text)
|
||
content_values.append(element.text)
|
||
for attribute, value in element.attrib.items():
|
||
all_values.append(value)
|
||
if attribute != "bounds":
|
||
content_values.append(value)
|
||
if element.tail:
|
||
all_values.append(element.tail)
|
||
content_values.append(element.tail)
|
||
normalized_values = [_normalize_phone_value(value) for value in all_values]
|
||
normalized_all_document = "".join(normalized_values)
|
||
normalized_document = "".join(_normalize_phone_value(value) for value in content_values)
|
||
return (
|
||
any(_matches_phone(value) for value in normalized_values)
|
||
or _matches_phone(normalized_all_document)
|
||
or _matches_phone(normalized_document)
|
||
)
|
||
|
||
|
||
def _normalize_phone_value(value: str) -> str:
|
||
return _SEPARATOR_RE.sub("", value.translate(_MASK_TRANSLATION))
|
||
|
||
|
||
def _matches_phone(value: str) -> bool:
|
||
return _FULL_PHONE_RE.search(value) is not None or _MASKED_PHONE_RE.search(value) is not None
|
||
|
||
|
||
def _derived_manifest(
|
||
source_manifest: dict[str, Any],
|
||
link: ProductUrl,
|
||
state: str,
|
||
source_manifest_path: Path,
|
||
source_screenshot_path: Path,
|
||
source_hierarchy_path: Path,
|
||
derived_screenshot_path: Path,
|
||
derived_hierarchy_path: Path,
|
||
cleanup_stats: _CleanupStats,
|
||
) -> dict[str, Any]:
|
||
device = source_manifest["device"]
|
||
return {
|
||
"schema_version": 1,
|
||
"privacy_tier": "SANITIZED",
|
||
"sanitizer_version": PRIVACY_MASK_CONFIG.version,
|
||
"screenshot_space": {
|
||
"width": PRIVACY_MASK_CONFIG.screenshot_width,
|
||
"height": PRIVACY_MASK_CONFIG.screenshot_height,
|
||
"privacy_mask_rectangle": [
|
||
0,
|
||
0,
|
||
PRIVACY_MASK_CONFIG.screenshot_width,
|
||
PRIVACY_MASK_CONFIG.privacy_top,
|
||
],
|
||
},
|
||
"xml_coordinate_space": {
|
||
"width": PRIVACY_MASK_CONFIG.xml_width,
|
||
"height": PRIVACY_MASK_CONFIG.xml_height,
|
||
"privacy_mask_rectangle": [0, 0, PRIVACY_MASK_CONFIG.xml_width, PRIVACY_MASK_CONFIG.privacy_top],
|
||
"observed_max": {"right": cleanup_stats.max_right, "bottom": cleanup_stats.max_bottom},
|
||
},
|
||
"privacy_cleanup": {
|
||
"removed_nodes": cleanup_stats.removed_nodes,
|
||
"cleared_crossing_nodes": cleanup_stats.cleared_crossing_nodes,
|
||
"preserved_crossing_price_nodes": cleanup_stats.preserved_crossing_price_nodes,
|
||
"retained_below_nodes": cleanup_stats.retained_below_nodes,
|
||
"max_right": cleanup_stats.max_right,
|
||
"max_bottom": cleanup_stats.max_bottom,
|
||
},
|
||
"product": {"goods_id": link.goods_id},
|
||
"human_declared_state": state,
|
||
"device": {
|
||
"model": device["model"],
|
||
"android_version": device.get("android_version"),
|
||
"pdd_package": device["pdd_package"],
|
||
"pdd_version": device["pdd_version"],
|
||
},
|
||
# source hashes stay only in the local derived manifest; no raw path, serial or body is retained.
|
||
"source": {
|
||
"manifest_sha256": _sha256_file(source_manifest_path),
|
||
"artifacts": [
|
||
{"path": "screenshot.png", "sha256": _sha256_file(source_screenshot_path)},
|
||
{"path": "hierarchy.xml", "sha256": _sha256_file(source_hierarchy_path)},
|
||
],
|
||
},
|
||
"derived": {
|
||
"artifacts": [
|
||
{"path": "screenshot.png", "sha256": _sha256_file(derived_screenshot_path)},
|
||
{"path": "hierarchy.xml", "sha256": _sha256_file(derived_hierarchy_path)},
|
||
]
|
||
},
|
||
}
|
||
|
||
|
||
def _sha256_file(path: Path) -> str:
|
||
digest = sha256()
|
||
with path.open("rb") as source:
|
||
for chunk in iter(lambda: source.read(1024 * 1024), b""):
|
||
digest.update(chunk)
|
||
return digest.hexdigest()
|
||
|
||
|
||
def _clean_staging(staging: Path | None) -> None:
|
||
if staging is not None and staging.exists():
|
||
shutil.rmtree(staging)
|
||
|
||
|
||
def _publish_staging(staging: Path, target: Path) -> None:
|
||
"""发布前二次检查,并使用目录 rename 而不是会覆盖目标的 replace。"""
|
||
|
||
if target.exists():
|
||
raise SkuEvidenceSanitizationError("派生证据目录已存在,拒绝覆盖。")
|
||
try:
|
||
staging.rename(target)
|
||
except OSError as error:
|
||
# 竞态中新目标出现或文件系统拒绝 rename 时一律不尝试覆盖或重试。
|
||
raise SkuEvidenceSanitizationError("派生证据目录发布失败,未覆盖已有目录。") from error
|