#!/usr/bin/env python3 """把第三方 ERP 商品工作簿转换为后续商品目录导入使用的规范 CSV。 这个工具只读取 XLSX、在指定目录写 CSV;不发送 HTTP 请求,也不会连接数据库。 部分第三方工作簿把 OOXML 工作表 dimension 错写成 A1,不能用依赖 dimension 的 Excel 读取器。本工具直接流式读取工作表 XML,因此能稳定处理这类大文件。 """ from __future__ import annotations import argparse import csv import re import sys from collections import Counter from dataclasses import dataclass from pathlib import Path from tempfile import NamedTemporaryFile from typing import Iterator from urllib.parse import parse_qs, urlparse from zipfile import ZipFile from lxml import etree SPREADSHEET_NS = "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}" ROW_TAG = SPREADSHEET_NS + "row" CELL_TAG = SPREADSHEET_NS + "c" TEXT_TAG = SPREADSHEET_NS + "t" VALUE_TAG = SPREADSHEET_NS + "v" INLINE_TEXT_TAG = SPREADSHEET_NS + "is" GOODS_ID_PATTERN = re.compile(r"^\d{5,20}$") PDD_HOSTS = {"mobile.pinduoduo.com", "mobile.yangkeduo.com"} REQUIRED_HEADERS = ( "Parent SKU", "产品标题", "sku", "变种属性名称一", "变种属性名称二", "变种属性值一", "变种属性值二", "主图(URL)地址", "来源URL", "店铺名", "产品id", ) # 每个输入数据行对应一条规范 CSV 行。后续接口提交器按商品 ID 去重即可。 CSV_FIELDS = ( "source_file", "source_row", "source_updated_at", "shopee_goods_id", "shopee_title", "shopee_status", "shopee_shop_name", "shopee_image_url", "shopee_main_sku_code", "shopee_sku_id", "shopee_sku_code", "spec_raw", "color", "size", "advice", "parse_ok", "pdd_goods_id", "pdd_goods_url", "pdd_title", "pdd_shop_name", "association_shopee_goods_id", "association_pdd_goods_id", "source_pdd_url", "pdd_importable", "validation_code", ) class CatalogExportError(RuntimeError): """源工作簿不符合可转换要求。""" @dataclass(frozen=True) class ExportResult: """一个输入工作簿的导出统计。""" source: Path output: Path rows: int valid_pdd_rows: int validation_counts: tuple[tuple[str, int], ...] def text_of(element: etree._Element) -> str: """合并 OOXML 富文本的所有文本节点。""" return "".join(element.itertext()).strip() def column_index(cell_reference: str) -> int: """把 A、AA 等 Excel 列号转成从零开始的索引。""" letters = "".join(character for character in cell_reference if character.isalpha()) value = 0 for character in letters.upper(): value = value * 26 + ord(character) - ord("A") + 1 return value - 1 def read_shared_strings(archive: ZipFile) -> list[str]: """按需读取 shared strings;当前来源主要使用 inlineStr。""" try: handle = archive.open("xl/sharedStrings.xml") except KeyError: return [] with handle: return [ text_of(element) for _, element in etree.iterparse(handle, events=("end",), tag=SPREADSHEET_NS + "si") ] def cell_value(cell: etree._Element, shared_strings: list[str]) -> str: """读取一个 OOXML 单元格的文本值,不依赖错误的 dimension。""" cell_type = cell.get("t", "") if cell_type == "inlineStr": inline = cell.find(INLINE_TEXT_TAG) return text_of(inline) if inline is not None else "" value = cell.find(VALUE_TAG) if value is None: return "" raw = text_of(value) if cell_type == "s": try: return shared_strings[int(raw)] except (IndexError, ValueError) as exc: raise CatalogExportError(f"shared string 索引无效: {raw!r}") from exc return raw def worksheet_name(archive: ZipFile) -> str: """选择工作簿的第一个工作表;第三方导出只有一个数据表。""" sheets = sorted( name for name in archive.namelist() if name.startswith("xl/worksheets/sheet") and name.endswith(".xml") ) if not sheets: raise CatalogExportError("XLSX 中没有工作表") return sheets[0] def iter_rows(path: Path) -> Iterator[tuple[int, list[str]]]: """流式迭代第一个工作表的行,跳过错误的 dimension 元数据。""" try: archive = ZipFile(path) except OSError as exc: raise CatalogExportError(f"无法打开工作簿: {path}") from exc with archive: shared_strings = read_shared_strings(archive) with archive.open(worksheet_name(archive)) as worksheet: for _, row in etree.iterparse(worksheet, events=("end",), tag=ROW_TAG): cells: dict[int, str] = {} for cell in row.findall(CELL_TAG): reference = cell.get("r", "") if reference: cells[column_index(reference)] = cell_value(cell, shared_strings) row_number = int(row.get("r", "0")) if cells: values = ["" for _ in range(max(cells) + 1)] for index, value in cells.items(): values[index] = value yield row_number, values row.clear() while row.getprevious() is not None: del row.getparent()[0] def normalize_header(value: str) -> str: """统一繁简体和空白,便于严格识别来源中的规格列名。""" return ( value.strip() .lower() .replace(" ", "") .replace("顏", "颜") .replace("色", "色") .replace("分類", "分类") .replace("碼", "码") ) def dimension_kind(header: str) -> str: """只依据明确的字段名判断颜色或尺码,其他列不猜测。""" normalized = normalize_header(header) if "尺码" in normalized or "尺寸" in normalized: return "size" if "颜色" in normalized: return "color" return "" def value_at(values: list[str], index: int | None) -> str: return values[index].strip() if index is not None and index < len(values) else "" def canonical_pdd(raw_url: str) -> tuple[str, str, str]: """提取 PDD goods_id;失败时返回明确校验码而不是丢弃蝦皮数据。""" if not raw_url: return "", "", "PDD_URL_MISSING" parsed = urlparse(raw_url) if parsed.scheme not in {"http", "https"} or parsed.hostname not in PDD_HOSTS: return "", "", "PDD_URL_HOST_INVALID" goods_id = parse_qs(parsed.query).get("goods_id", [""])[0].strip() if not GOODS_ID_PATTERN.fullmatch(goods_id): return "", "", "PDD_GOODS_ID_MISSING" return goods_id, f"https://mobile.yangkeduo.com/goods.html?goods_id={goods_id}", "" def join_spec(first: str, second: str) -> str: """按源列顺序保留规格原文;值内即使有逗号也不拆分。""" return ",".join(value for value in (first, second) if value) def build_record( values: list[str], indexes: dict[str, int], source_file: str, row_number: int, ) -> dict[str, str]: """把来源一行映射成后续商品目录入库所需字段。""" first_name = value_at(values, indexes["变种属性名称一"]) second_name = value_at(values, indexes["变种属性名称二"]) first_value = value_at(values, indexes["变种属性值一"]) second_value = value_at(values, indexes["变种属性值二"]) color = "" size = "" if dimension_kind(first_name) == "color": color = first_value elif dimension_kind(first_name) == "size": size = first_value if dimension_kind(second_name) == "color": color = second_value elif dimension_kind(second_name) == "size": size = second_value raw_pdd_url = value_at(values, indexes["来源URL"]) pdd_goods_id, pdd_goods_url, validation_code = canonical_pdd(raw_pdd_url) shopee_goods_id = value_at(values, indexes["产品id"]) if not GOODS_ID_PATTERN.fullmatch(shopee_goods_id): validation_code = ";".join( value for value in (validation_code, "SHOPEE_GOODS_ID_INVALID") if value ) parse_ok = bool(color or size) return { "source_file": source_file, "source_row": str(row_number), "source_updated_at": value_at(values, indexes.get("更新时间")), "shopee_goods_id": shopee_goods_id, "shopee_title": value_at(values, indexes["产品标题"]), "shopee_status": "", "shopee_shop_name": value_at(values, indexes["店铺名"]), "shopee_image_url": value_at(values, indexes["主图(URL)地址"]), "shopee_main_sku_code": value_at(values, indexes["Parent SKU"]), "shopee_sku_id": "", "shopee_sku_code": value_at(values, indexes["sku"]), "spec_raw": join_spec(first_value, second_value), "color": color, "size": size, "advice": "", "parse_ok": "true" if parse_ok else "false", "pdd_goods_id": pdd_goods_id, "pdd_goods_url": pdd_goods_url, "pdd_title": "", "pdd_shop_name": "", "association_shopee_goods_id": shopee_goods_id if pdd_goods_id else "", "association_pdd_goods_id": pdd_goods_id, "source_pdd_url": raw_pdd_url, "pdd_importable": "true" if pdd_goods_id else "false", "validation_code": validation_code, } def export_file(source: Path, input_root: Path, output_root: Path) -> ExportResult: """转换一个 XLSX,并以原子替换方式写出对应 CSV。""" rows = iter_rows(source) try: _, headers = next(rows) except StopIteration as exc: raise CatalogExportError(f"{source.name} 没有表头") from exc indexes = {header: position for position, header in enumerate(headers) if header} missing = [header for header in REQUIRED_HEADERS if header not in indexes] if missing: raise CatalogExportError(f"{source.name} 缺少列: {', '.join(missing)}") relative = source.relative_to(input_root) output = (output_root / relative).with_suffix(".csv") output.parent.mkdir(parents=True, exist_ok=True) validation_counts: Counter[str] = Counter() row_count = 0 valid_pdd_rows = 0 temp_name = "" try: with NamedTemporaryFile( mode="w", encoding="utf-8-sig", newline="", delete=False, dir=output.parent, prefix=output.stem + ".", suffix=".tmp" ) as temp: temp_name = temp.name writer = csv.DictWriter(temp, fieldnames=CSV_FIELDS, extrasaction="raise") writer.writeheader() for row_number, values in rows: record = build_record(values, indexes, relative.as_posix(), row_number) writer.writerow(record) row_count += 1 if record["pdd_importable"] == "true": valid_pdd_rows += 1 if record["validation_code"]: for code in record["validation_code"].split(";"): validation_counts[code] += 1 Path(temp_name).replace(output) except Exception: if temp_name: Path(temp_name).unlink(missing_ok=True) raise return ExportResult( source=source, output=output, rows=row_count, valid_pdd_rows=valid_pdd_rows, validation_counts=tuple(sorted(validation_counts.items())), ) def discover_sources(input_path: Path) -> tuple[Path, Path, list[Path]]: """支持单文件或目录;目录模式保留其下原始相对路径。""" if input_path.is_file(): if input_path.suffix.lower() != ".xlsx": raise CatalogExportError("输入文件必须是 .xlsx") return input_path, input_path.parent, [input_path] if not input_path.is_dir(): raise CatalogExportError(f"找不到输入路径: {input_path}") sources = sorted(path for path in input_path.rglob("*.xlsx") if not path.name.startswith("~$")) if not sources: raise CatalogExportError(f"目录中没有 .xlsx 文件: {input_path}") return input_path, input_path, sources def run(args: argparse.Namespace) -> int: input_path = Path(args.input).resolve() output_root = Path(args.output_dir).resolve() _, input_root, sources = discover_sources(input_path) if output_root == input_root or output_root.is_relative_to(input_root): raise CatalogExportError("输出目录不能位于输入目录内,避免把 CSV 当成后续输入") results = [export_file(source, input_root, output_root) for source in sources] total_rows = sum(result.rows for result in results) total_valid_pdd = sum(result.valid_pdd_rows for result in results) validation_counts: Counter[str] = Counter() for result in results: validation_counts.update(dict(result.validation_counts)) print( f"{result.source.name}: {result.rows} 行,PDD 可入库 {result.valid_pdd_rows} 行," f"输出 {result.output}" ) print(f"转换完成:{len(results)} 个 XLSX,{total_rows} 行,PDD 可入库 {total_valid_pdd} 行。") if validation_counts: print("校验提示:" + ",".join(f"{code} {count}" for code, count in sorted(validation_counts.items()))) print("本工具未调用接口,也未写入任何数据库。") return 0 def build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser(description="把第三方 ERP XLSX 导出为规范商品目录 CSV") parser.add_argument("input", help="一个 XLSX 文件或包含 XLSX 的目录") parser.add_argument("--output-dir", required=True, help="CSV 输出根目录(必须在输入目录之外)") return parser def main() -> int: try: return run(build_parser().parse_args()) except CatalogExportError as exc: print(f"转换失败:{exc}", file=sys.stderr) return 1 if __name__ == "__main__": raise SystemExit(main())