397 lines
14 KiB
Python
397 lines
14 KiB
Python
#!/usr/bin/env python3
|
||||
|
|
"""把第三方 ERP 商品工作簿转换为后续商品目录导入使用的规范 CSV。
|
|||
|
|
|
|||
|
|
这个工具只读取 XLSX、在指定目录写 CSV;不发送 HTTP 请求,也不会连接数据库。
|
|||
|
|
部分第三方工作簿把 OOXML 工作表 dimension 错写成 A1,不能用依赖 dimension 的
|
|||
|
|
Excel 读取器。本工具直接流式读取工作表 XML,因此能稳定处理这类大文件。
|
|||
|
|
"""
|
|||
|
|
|
|||
|
|
from __future__ import annotations
|
|||
|
|
|
|||
|
|
import argparse
|
|||
|
|
import csv
|
|||
|
|
import re
|
|||
|
|
import sys
|
|||
|
|
from collections import Counter
|
|||
|
|
from dataclasses import dataclass
|
|||
|
|
from pathlib import Path
|
|||
|
|
from tempfile import NamedTemporaryFile
|
|||
|
|
from typing import Iterator
|
|||
|
|
from urllib.parse import parse_qs, urlparse
|
|||
|
|
from zipfile import ZipFile
|
|||
|
|
|
|||
|
|
from lxml import etree
|
|||
|
|
|
|||
|
|
|
|||
|
|
SPREADSHEET_NS = "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}"
|
|||
|
|
ROW_TAG = SPREADSHEET_NS + "row"
|
|||
|
|
CELL_TAG = SPREADSHEET_NS + "c"
|
|||
|
|
TEXT_TAG = SPREADSHEET_NS + "t"
|
|||
|
|
VALUE_TAG = SPREADSHEET_NS + "v"
|
|||
|
|
INLINE_TEXT_TAG = SPREADSHEET_NS + "is"
|
|||
|
|
GOODS_ID_PATTERN = re.compile(r"^\d{5,20}$")
|
|||
|
|
PDD_HOSTS = {"mobile.pinduoduo.com", "mobile.yangkeduo.com"}
|
|||
|
|
|
|||
|
|
REQUIRED_HEADERS = (
|
|||
|
|
"Parent SKU",
|
|||
|
|
"产品标题",
|
|||
|
|
"sku",
|
|||
|
|
"变种属性名称一",
|
|||
|
|
"变种属性名称二",
|
|||
|
|
"变种属性值一",
|
|||
|
|
"变种属性值二",
|
|||
|
|
"主图(URL)地址",
|
|||
|
|
"来源URL",
|
|||
|
|
"店铺名",
|
|||
|
|
"产品id",
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
# 每个输入数据行对应一条规范 CSV 行。后续接口提交器按商品 ID 去重即可。
|
|||
|
|
CSV_FIELDS = (
|
|||
|
|
"source_file",
|
|||
|
|
"source_row",
|
|||
|
|
"source_updated_at",
|
|||
|
|
"shopee_goods_id",
|
|||
|
|
"shopee_title",
|
|||
|
|
"shopee_status",
|
|||
|
|
"shopee_shop_name",
|
|||
|
|
"shopee_image_url",
|
|||
|
|
"shopee_main_sku_code",
|
|||
|
|
"shopee_sku_id",
|
|||
|
|
"shopee_sku_code",
|
|||
|
|
"spec_raw",
|
|||
|
|
"color",
|
|||
|
|
"size",
|
|||
|
|
"advice",
|
|||
|
|
"parse_ok",
|
|||
|
|
"pdd_goods_id",
|
|||
|
|
"pdd_goods_url",
|
|||
|
|
"pdd_title",
|
|||
|
|
"pdd_shop_name",
|
|||
|
|
"association_shopee_goods_id",
|
|||
|
|
"association_pdd_goods_id",
|
|||
|
|
"source_pdd_url",
|
|||
|
|
"pdd_importable",
|
|||
|
|
"validation_code",
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
class CatalogExportError(RuntimeError):
|
|||
|
|
"""源工作簿不符合可转换要求。"""
|
|||
|
|
|
|||
|
|
|
|||
|
|
@dataclass(frozen=True)
|
|||
|
|
class ExportResult:
|
|||
|
|
"""一个输入工作簿的导出统计。"""
|
|||
|
|
|
|||
|
|
source: Path
|
|||
|
|
output: Path
|
|||
|
|
rows: int
|
|||
|
|
valid_pdd_rows: int
|
|||
|
|
validation_counts: tuple[tuple[str, int], ...]
|
|||
|
|
|
|||
|
|
|
|||
|
|
def text_of(element: etree._Element) -> str:
|
|||
|
|
"""合并 OOXML 富文本的所有文本节点。"""
|
|||
|
|
|
|||
|
|
return "".join(element.itertext()).strip()
|
|||
|
|
|
|||
|
|
|
|||
|
|
def column_index(cell_reference: str) -> int:
|
|||
|
|
"""把 A、AA 等 Excel 列号转成从零开始的索引。"""
|
|||
|
|
|
|||
|
|
letters = "".join(character for character in cell_reference if character.isalpha())
|
|||
|
|
value = 0
|
|||
|
|
for character in letters.upper():
|
|||
|
|
value = value * 26 + ord(character) - ord("A") + 1
|
|||
|
|
return value - 1
|
|||
|
|
|
|||
|
|
|
|||
|
|
def read_shared_strings(archive: ZipFile) -> list[str]:
|
|||
|
|
"""按需读取 shared strings;当前来源主要使用 inlineStr。"""
|
|||
|
|
|
|||
|
|
try:
|
|||
|
|
handle = archive.open("xl/sharedStrings.xml")
|
|||
|
|
except KeyError:
|
|||
|
|
return []
|
|||
|
|
with handle:
|
|||
|
|
return [
|
|||
|
|
text_of(element)
|
|||
|
|
for _, element in etree.iterparse(handle, events=("end",), tag=SPREADSHEET_NS + "si")
|
|||
|
|
]
|
|||
|
|
|
|||
|
|
|
|||
|
|
def cell_value(cell: etree._Element, shared_strings: list[str]) -> str:
|
|||
|
|
"""读取一个 OOXML 单元格的文本值,不依赖错误的 dimension。"""
|
|||
|
|
|
|||
|
|
cell_type = cell.get("t", "")
|
|||
|
|
if cell_type == "inlineStr":
|
|||
|
|
inline = cell.find(INLINE_TEXT_TAG)
|
|||
|
|
return text_of(inline) if inline is not None else ""
|
|||
|
|
value = cell.find(VALUE_TAG)
|
|||
|
|
if value is None:
|
|||
|
|
return ""
|
|||
|
|
raw = text_of(value)
|
|||
|
|
if cell_type == "s":
|
|||
|
|
try:
|
|||
|
|
return shared_strings[int(raw)]
|
|||
|
|
except (IndexError, ValueError) as exc:
|
|||
|
|
raise CatalogExportError(f"shared string 索引无效: {raw!r}") from exc
|
|||
|
|
return raw
|
|||
|
|
|
|||
|
|
|
|||
|
|
def worksheet_name(archive: ZipFile) -> str:
|
|||
|
|
"""选择工作簿的第一个工作表;第三方导出只有一个数据表。"""
|
|||
|
|
|
|||
|
|
sheets = sorted(
|
|||
|
|
name
|
|||
|
|
for name in archive.namelist()
|
|||
|
|
if name.startswith("xl/worksheets/sheet") and name.endswith(".xml")
|
|||
|
|
)
|
|||
|
|
if not sheets:
|
|||
|
|
raise CatalogExportError("XLSX 中没有工作表")
|
|||
|
|
return sheets[0]
|
|||
|
|
|
|||
|
|
|
|||
|
|
def iter_rows(path: Path) -> Iterator[tuple[int, list[str]]]:
|
|||
|
|
"""流式迭代第一个工作表的行,跳过错误的 dimension 元数据。"""
|
|||
|
|
|
|||
|
|
try:
|
|||
|
|
archive = ZipFile(path)
|
|||
|
|
except OSError as exc:
|
|||
|
|
raise CatalogExportError(f"无法打开工作簿: {path}") from exc
|
|||
|
|
with archive:
|
|||
|
|
shared_strings = read_shared_strings(archive)
|
|||
|
|
with archive.open(worksheet_name(archive)) as worksheet:
|
|||
|
|
for _, row in etree.iterparse(worksheet, events=("end",), tag=ROW_TAG):
|
|||
|
|
cells: dict[int, str] = {}
|
|||
|
|
for cell in row.findall(CELL_TAG):
|
|||
|
|
reference = cell.get("r", "")
|
|||
|
|
if reference:
|
|||
|
|
cells[column_index(reference)] = cell_value(cell, shared_strings)
|
|||
|
|
row_number = int(row.get("r", "0"))
|
|||
|
|
if cells:
|
|||
|
|
values = ["" for _ in range(max(cells) + 1)]
|
|||
|
|
for index, value in cells.items():
|
|||
|
|
values[index] = value
|
|||
|
|
yield row_number, values
|
|||
|
|
row.clear()
|
|||
|
|
while row.getprevious() is not None:
|
|||
|
|
del row.getparent()[0]
|
|||
|
|
|
|||
|
|
|
|||
|
|
def normalize_header(value: str) -> str:
|
|||
|
|
"""统一繁简体和空白,便于严格识别来源中的规格列名。"""
|
|||
|
|
|
|||
|
|
return (
|
|||
|
|
value.strip()
|
|||
|
|
.lower()
|
|||
|
|
.replace(" ", "")
|
|||
|
|
.replace("顏", "颜")
|
|||
|
|
.replace("色", "色")
|
|||
|
|
.replace("分類", "分类")
|
|||
|
|
.replace("碼", "码")
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def dimension_kind(header: str) -> str:
|
|||
|
|
"""只依据明确的字段名判断颜色或尺码,其他列不猜测。"""
|
|||
|
|
|
|||
|
|
normalized = normalize_header(header)
|
|||
|
|
if "尺码" in normalized or "尺寸" in normalized:
|
|||
|
|
return "size"
|
|||
|
|
if "颜色" in normalized:
|
|||
|
|
return "color"
|
|||
|
|
return ""
|
|||
|
|
|
|||
|
|
|
|||
|
|
def value_at(values: list[str], index: int | None) -> str:
|
|||
|
|
return values[index].strip() if index is not None and index < len(values) else ""
|
|||
|
|
|
|||
|
|
|
|||
|
|
def canonical_pdd(raw_url: str) -> tuple[str, str, str]:
|
|||
|
|
"""提取 PDD goods_id;失败时返回明确校验码而不是丢弃蝦皮数据。"""
|
|||
|
|
|
|||
|
|
if not raw_url:
|
|||
|
|
return "", "", "PDD_URL_MISSING"
|
|||
|
|
parsed = urlparse(raw_url)
|
|||
|
|
if parsed.scheme not in {"http", "https"} or parsed.hostname not in PDD_HOSTS:
|
|||
|
|
return "", "", "PDD_URL_HOST_INVALID"
|
|||
|
|
goods_id = parse_qs(parsed.query).get("goods_id", [""])[0].strip()
|
|||
|
|
if not GOODS_ID_PATTERN.fullmatch(goods_id):
|
|||
|
|
return "", "", "PDD_GOODS_ID_MISSING"
|
|||
|
|
return goods_id, f"https://mobile.yangkeduo.com/goods.html?goods_id={goods_id}", ""
|
|||
|
|
|
|||
|
|
|
|||
|
|
def join_spec(first: str, second: str) -> str:
|
|||
|
|
"""按源列顺序保留规格原文;值内即使有逗号也不拆分。"""
|
|||
|
|
|
|||
|
|
return ",".join(value for value in (first, second) if value)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def build_record(
|
|||
|
|
values: list[str],
|
|||
|
|
indexes: dict[str, int],
|
|||
|
|
source_file: str,
|
|||
|
|
row_number: int,
|
|||
|
|
) -> dict[str, str]:
|
|||
|
|
"""把来源一行映射成后续商品目录入库所需字段。"""
|
|||
|
|
|
|||
|
|
first_name = value_at(values, indexes["变种属性名称一"])
|
|||
|
|
second_name = value_at(values, indexes["变种属性名称二"])
|
|||
|
|
first_value = value_at(values, indexes["变种属性值一"])
|
|||
|
|
second_value = value_at(values, indexes["变种属性值二"])
|
|||
|
|
color = ""
|
|||
|
|
size = ""
|
|||
|
|
if dimension_kind(first_name) == "color":
|
|||
|
|
color = first_value
|
|||
|
|
elif dimension_kind(first_name) == "size":
|
|||
|
|
size = first_value
|
|||
|
|
if dimension_kind(second_name) == "color":
|
|||
|
|
color = second_value
|
|||
|
|
elif dimension_kind(second_name) == "size":
|
|||
|
|
size = second_value
|
|||
|
|
raw_pdd_url = value_at(values, indexes["来源URL"])
|
|||
|
|
pdd_goods_id, pdd_goods_url, validation_code = canonical_pdd(raw_pdd_url)
|
|||
|
|
shopee_goods_id = value_at(values, indexes["产品id"])
|
|||
|
|
if not GOODS_ID_PATTERN.fullmatch(shopee_goods_id):
|
|||
|
|
validation_code = ";".join(
|
|||
|
|
value for value in (validation_code, "SHOPEE_GOODS_ID_INVALID") if value
|
|||
|
|
)
|
|||
|
|
parse_ok = bool(color or size)
|
|||
|
|
return {
|
|||
|
|
"source_file": source_file,
|
|||
|
|
"source_row": str(row_number),
|
|||
|
|
"source_updated_at": value_at(values, indexes.get("更新时间")),
|
|||
|
|
"shopee_goods_id": shopee_goods_id,
|
|||
|
|
"shopee_title": value_at(values, indexes["产品标题"]),
|
|||
|
|
"shopee_status": "",
|
|||
|
|
"shopee_shop_name": value_at(values, indexes["店铺名"]),
|
|||
|
|
"shopee_image_url": value_at(values, indexes["主图(URL)地址"]),
|
|||
|
|
"shopee_main_sku_code": value_at(values, indexes["Parent SKU"]),
|
|||
|
|
"shopee_sku_id": "",
|
|||
|
|
"shopee_sku_code": value_at(values, indexes["sku"]),
|
|||
|
|
"spec_raw": join_spec(first_value, second_value),
|
|||
|
|
"color": color,
|
|||
|
|
"size": size,
|
|||
|
|
"advice": "",
|
|||
|
|
"parse_ok": "true" if parse_ok else "false",
|
|||
|
|
"pdd_goods_id": pdd_goods_id,
|
|||
|
|
"pdd_goods_url": pdd_goods_url,
|
|||
|
|
"pdd_title": "",
|
|||
|
|
"pdd_shop_name": "",
|
|||
|
|
"association_shopee_goods_id": shopee_goods_id if pdd_goods_id else "",
|
|||
|
|
"association_pdd_goods_id": pdd_goods_id,
|
|||
|
|
"source_pdd_url": raw_pdd_url,
|
|||
|
|
"pdd_importable": "true" if pdd_goods_id else "false",
|
|||
|
|
"validation_code": validation_code,
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
|
|||
|
|
def export_file(source: Path, input_root: Path, output_root: Path) -> ExportResult:
|
|||
|
|
"""转换一个 XLSX,并以原子替换方式写出对应 CSV。"""
|
|||
|
|
|
|||
|
|
rows = iter_rows(source)
|
|||
|
|
try:
|
|||
|
|
_, headers = next(rows)
|
|||
|
|
except StopIteration as exc:
|
|||
|
|
raise CatalogExportError(f"{source.name} 没有表头") from exc
|
|||
|
|
indexes = {header: position for position, header in enumerate(headers) if header}
|
|||
|
|
missing = [header for header in REQUIRED_HEADERS if header not in indexes]
|
|||
|
|
if missing:
|
|||
|
|
raise CatalogExportError(f"{source.name} 缺少列: {', '.join(missing)}")
|
|||
|
|
|
|||
|
|
relative = source.relative_to(input_root)
|
|||
|
|
output = (output_root / relative).with_suffix(".csv")
|
|||
|
|
output.parent.mkdir(parents=True, exist_ok=True)
|
|||
|
|
validation_counts: Counter[str] = Counter()
|
|||
|
|
row_count = 0
|
|||
|
|
valid_pdd_rows = 0
|
|||
|
|
temp_name = ""
|
|||
|
|
try:
|
|||
|
|
with NamedTemporaryFile(
|
|||
|
|
mode="w", encoding="utf-8-sig", newline="", delete=False, dir=output.parent,
|
|||
|
|
prefix=output.stem + ".", suffix=".tmp"
|
|||
|
|
) as temp:
|
|||
|
|
temp_name = temp.name
|
|||
|
|
writer = csv.DictWriter(temp, fieldnames=CSV_FIELDS, extrasaction="raise")
|
|||
|
|
writer.writeheader()
|
|||
|
|
for row_number, values in rows:
|
|||
|
|
record = build_record(values, indexes, relative.as_posix(), row_number)
|
|||
|
|
writer.writerow(record)
|
|||
|
|
row_count += 1
|
|||
|
|
if record["pdd_importable"] == "true":
|
|||
|
|
valid_pdd_rows += 1
|
|||
|
|
if record["validation_code"]:
|
|||
|
|
for code in record["validation_code"].split(";"):
|
|||
|
|
validation_counts[code] += 1
|
|||
|
|
Path(temp_name).replace(output)
|
|||
|
|
except Exception:
|
|||
|
|
if temp_name:
|
|||
|
|
Path(temp_name).unlink(missing_ok=True)
|
|||
|
|
raise
|
|||
|
|
return ExportResult(
|
|||
|
|
source=source,
|
|||
|
|
output=output,
|
|||
|
|
rows=row_count,
|
|||
|
|
valid_pdd_rows=valid_pdd_rows,
|
|||
|
|
validation_counts=tuple(sorted(validation_counts.items())),
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def discover_sources(input_path: Path) -> tuple[Path, Path, list[Path]]:
|
|||
|
|
"""支持单文件或目录;目录模式保留其下原始相对路径。"""
|
|||
|
|
|
|||
|
|
if input_path.is_file():
|
|||
|
|
if input_path.suffix.lower() != ".xlsx":
|
|||
|
|
raise CatalogExportError("输入文件必须是 .xlsx")
|
|||
|
|
return input_path, input_path.parent, [input_path]
|
|||
|
|
if not input_path.is_dir():
|
|||
|
|
raise CatalogExportError(f"找不到输入路径: {input_path}")
|
|||
|
|
sources = sorted(path for path in input_path.rglob("*.xlsx") if not path.name.startswith("~$"))
|
|||
|
|
if not sources:
|
|||
|
|
raise CatalogExportError(f"目录中没有 .xlsx 文件: {input_path}")
|
|||
|
|
return input_path, input_path, sources
|
|||
|
|
|
|||
|
|
|
|||
|
|
def run(args: argparse.Namespace) -> int:
|
|||
|
|
input_path = Path(args.input).resolve()
|
|||
|
|
output_root = Path(args.output_dir).resolve()
|
|||
|
|
_, input_root, sources = discover_sources(input_path)
|
|||
|
|
if output_root == input_root or output_root.is_relative_to(input_root):
|
|||
|
|
raise CatalogExportError("输出目录不能位于输入目录内,避免把 CSV 当成后续输入")
|
|||
|
|
results = [export_file(source, input_root, output_root) for source in sources]
|
|||
|
|
total_rows = sum(result.rows for result in results)
|
|||
|
|
total_valid_pdd = sum(result.valid_pdd_rows for result in results)
|
|||
|
|
validation_counts: Counter[str] = Counter()
|
|||
|
|
for result in results:
|
|||
|
|
validation_counts.update(dict(result.validation_counts))
|
|||
|
|
print(
|
|||
|
|
f"{result.source.name}: {result.rows} 行,PDD 可入库 {result.valid_pdd_rows} 行,"
|
|||
|
|
f"输出 {result.output}"
|
|||
|
|
)
|
|||
|
|
print(f"转换完成:{len(results)} 个 XLSX,{total_rows} 行,PDD 可入库 {total_valid_pdd} 行。")
|
|||
|
|
if validation_counts:
|
|||
|
|
print("校验提示:" + ",".join(f"{code} {count}" for code, count in sorted(validation_counts.items())))
|
|||
|
|
print("本工具未调用接口,也未写入任何数据库。")
|
|||
|
|
return 0
|
|||
|
|
|
|||
|
|
|
|||
|
|
def build_parser() -> argparse.ArgumentParser:
|
|||
|
|
parser = argparse.ArgumentParser(description="把第三方 ERP XLSX 导出为规范商品目录 CSV")
|
|||
|
|
parser.add_argument("input", help="一个 XLSX 文件或包含 XLSX 的目录")
|
|||
|
|
parser.add_argument("--output-dir", required=True, help="CSV 输出根目录(必须在输入目录之外)")
|
|||
|
|
return parser
|
|||
|
|
|
|||
|
|
|
|||
|
|
def main() -> int:
|
|||
|
|
try:
|
|||
|
|
return run(build_parser().parse_args())
|
|||
|
|
except CatalogExportError as exc:
|
|||
|
|
print(f"转换失败:{exc}", file=sys.stderr)
|
|||
|
|
return 1
|
|||
|
|
|
|||
|
|
|
|||
|
|
if __name__ == "__main__":
|
|||
|
|
raise SystemExit(main())
|