feat: 导出第三方商品规范 CSV (#280)

This commit is contained in:
chengma
2026-08-20 14:17:37 +08:00
parent c4f367d8ab
commit 71fabb1db8
3 changed files with 488 additions and 0 deletions
+396
View File
@@ -0,0 +1,396 @@
#!/usr/bin/env python3
"""把第三方 ERP 商品工作簿转换为后续商品目录导入使用的规范 CSV。
这个工具只读取 XLSX、在指定目录写 CSV;不发送 HTTP 请求,也不会连接数据库。
部分第三方工作簿把 OOXML 工作表 dimension 错写成 A1,不能用依赖 dimension 的
Excel 读取器。本工具直接流式读取工作表 XML,因此能稳定处理这类大文件。
"""
from __future__ import annotations
import argparse
import csv
import re
import sys
from collections import Counter
from dataclasses import dataclass
from pathlib import Path
from tempfile import NamedTemporaryFile
from typing import Iterator
from urllib.parse import parse_qs, urlparse
from zipfile import ZipFile
from lxml import etree
SPREADSHEET_NS = "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}"
ROW_TAG = SPREADSHEET_NS + "row"
CELL_TAG = SPREADSHEET_NS + "c"
TEXT_TAG = SPREADSHEET_NS + "t"
VALUE_TAG = SPREADSHEET_NS + "v"
INLINE_TEXT_TAG = SPREADSHEET_NS + "is"
GOODS_ID_PATTERN = re.compile(r"^\d{5,20}$")
PDD_HOSTS = {"mobile.pinduoduo.com", "mobile.yangkeduo.com"}
REQUIRED_HEADERS = (
"Parent SKU",
"产品标题",
"sku",
"变种属性名称一",
"变种属性名称二",
"变种属性值一",
"变种属性值二",
"主图(URL)地址",
"来源URL",
"店铺名",
"产品id",
)
# 每个输入数据行对应一条规范 CSV 行。后续接口提交器按商品 ID 去重即可。
CSV_FIELDS = (
"source_file",
"source_row",
"source_updated_at",
"shopee_goods_id",
"shopee_title",
"shopee_status",
"shopee_shop_name",
"shopee_image_url",
"shopee_main_sku_code",
"shopee_sku_id",
"shopee_sku_code",
"spec_raw",
"color",
"size",
"advice",
"parse_ok",
"pdd_goods_id",
"pdd_goods_url",
"pdd_title",
"pdd_shop_name",
"association_shopee_goods_id",
"association_pdd_goods_id",
"source_pdd_url",
"pdd_importable",
"validation_code",
)
class CatalogExportError(RuntimeError):
"""源工作簿不符合可转换要求。"""
@dataclass(frozen=True)
class ExportResult:
"""一个输入工作簿的导出统计。"""
source: Path
output: Path
rows: int
valid_pdd_rows: int
validation_counts: tuple[tuple[str, int], ...]
def text_of(element: etree._Element) -> str:
"""合并 OOXML 富文本的所有文本节点。"""
return "".join(element.itertext()).strip()
def column_index(cell_reference: str) -> int:
"""把 A、AA 等 Excel 列号转成从零开始的索引。"""
letters = "".join(character for character in cell_reference if character.isalpha())
value = 0
for character in letters.upper():
value = value * 26 + ord(character) - ord("A") + 1
return value - 1
def read_shared_strings(archive: ZipFile) -> list[str]:
"""按需读取 shared strings;当前来源主要使用 inlineStr。"""
try:
handle = archive.open("xl/sharedStrings.xml")
except KeyError:
return []
with handle:
return [
text_of(element)
for _, element in etree.iterparse(handle, events=("end",), tag=SPREADSHEET_NS + "si")
]
def cell_value(cell: etree._Element, shared_strings: list[str]) -> str:
"""读取一个 OOXML 单元格的文本值,不依赖错误的 dimension。"""
cell_type = cell.get("t", "")
if cell_type == "inlineStr":
inline = cell.find(INLINE_TEXT_TAG)
return text_of(inline) if inline is not None else ""
value = cell.find(VALUE_TAG)
if value is None:
return ""
raw = text_of(value)
if cell_type == "s":
try:
return shared_strings[int(raw)]
except (IndexError, ValueError) as exc:
raise CatalogExportError(f"shared string 索引无效: {raw!r}") from exc
return raw
def worksheet_name(archive: ZipFile) -> str:
"""选择工作簿的第一个工作表;第三方导出只有一个数据表。"""
sheets = sorted(
name
for name in archive.namelist()
if name.startswith("xl/worksheets/sheet") and name.endswith(".xml")
)
if not sheets:
raise CatalogExportError("XLSX 中没有工作表")
return sheets[0]
def iter_rows(path: Path) -> Iterator[tuple[int, list[str]]]:
"""流式迭代第一个工作表的行,跳过错误的 dimension 元数据。"""
try:
archive = ZipFile(path)
except OSError as exc:
raise CatalogExportError(f"无法打开工作簿: {path}") from exc
with archive:
shared_strings = read_shared_strings(archive)
with archive.open(worksheet_name(archive)) as worksheet:
for _, row in etree.iterparse(worksheet, events=("end",), tag=ROW_TAG):
cells: dict[int, str] = {}
for cell in row.findall(CELL_TAG):
reference = cell.get("r", "")
if reference:
cells[column_index(reference)] = cell_value(cell, shared_strings)
row_number = int(row.get("r", "0"))
if cells:
values = ["" for _ in range(max(cells) + 1)]
for index, value in cells.items():
values[index] = value
yield row_number, values
row.clear()
while row.getprevious() is not None:
del row.getparent()[0]
def normalize_header(value: str) -> str:
"""统一繁简体和空白,便于严格识别来源中的规格列名。"""
return (
value.strip()
.lower()
.replace(" ", "")
.replace("顏", "颜")
.replace("色", "色")
.replace("分類", "分类")
.replace("碼", "码")
)
def dimension_kind(header: str) -> str:
"""只依据明确的字段名判断颜色或尺码,其他列不猜测。"""
normalized = normalize_header(header)
if "尺码" in normalized or "尺寸" in normalized:
return "size"
if "颜色" in normalized:
return "color"
return ""
def value_at(values: list[str], index: int | None) -> str:
return values[index].strip() if index is not None and index < len(values) else ""
def canonical_pdd(raw_url: str) -> tuple[str, str, str]:
"""提取 PDD goods_id;失败时返回明确校验码而不是丢弃蝦皮数据。"""
if not raw_url:
return "", "", "PDD_URL_MISSING"
parsed = urlparse(raw_url)
if parsed.scheme not in {"http", "https"} or parsed.hostname not in PDD_HOSTS:
return "", "", "PDD_URL_HOST_INVALID"
goods_id = parse_qs(parsed.query).get("goods_id", [""])[0].strip()
if not GOODS_ID_PATTERN.fullmatch(goods_id):
return "", "", "PDD_GOODS_ID_MISSING"
return goods_id, f"https://mobile.yangkeduo.com/goods.html?goods_id={goods_id}", ""
def join_spec(first: str, second: str) -> str:
"""按源列顺序保留规格原文;值内即使有逗号也不拆分。"""
return ",".join(value for value in (first, second) if value)
def build_record(
values: list[str],
indexes: dict[str, int],
source_file: str,
row_number: int,
) -> dict[str, str]:
"""把来源一行映射成后续商品目录入库所需字段。"""
first_name = value_at(values, indexes["变种属性名称一"])
second_name = value_at(values, indexes["变种属性名称二"])
first_value = value_at(values, indexes["变种属性值一"])
second_value = value_at(values, indexes["变种属性值二"])
color = ""
size = ""
if dimension_kind(first_name) == "color":
color = first_value
elif dimension_kind(first_name) == "size":
size = first_value
if dimension_kind(second_name) == "color":
color = second_value
elif dimension_kind(second_name) == "size":
size = second_value
raw_pdd_url = value_at(values, indexes["来源URL"])
pdd_goods_id, pdd_goods_url, validation_code = canonical_pdd(raw_pdd_url)
shopee_goods_id = value_at(values, indexes["产品id"])
if not GOODS_ID_PATTERN.fullmatch(shopee_goods_id):
validation_code = ";".join(
value for value in (validation_code, "SHOPEE_GOODS_ID_INVALID") if value
)
parse_ok = bool(color or size)
return {
"source_file": source_file,
"source_row": str(row_number),
"source_updated_at": value_at(values, indexes.get("更新时间")),
"shopee_goods_id": shopee_goods_id,
"shopee_title": value_at(values, indexes["产品标题"]),
"shopee_status": "",
"shopee_shop_name": value_at(values, indexes["店铺名"]),
"shopee_image_url": value_at(values, indexes["主图(URL)地址"]),
"shopee_main_sku_code": value_at(values, indexes["Parent SKU"]),
"shopee_sku_id": "",
"shopee_sku_code": value_at(values, indexes["sku"]),
"spec_raw": join_spec(first_value, second_value),
"color": color,
"size": size,
"advice": "",
"parse_ok": "true" if parse_ok else "false",
"pdd_goods_id": pdd_goods_id,
"pdd_goods_url": pdd_goods_url,
"pdd_title": "",
"pdd_shop_name": "",
"association_shopee_goods_id": shopee_goods_id if pdd_goods_id else "",
"association_pdd_goods_id": pdd_goods_id,
"source_pdd_url": raw_pdd_url,
"pdd_importable": "true" if pdd_goods_id else "false",
"validation_code": validation_code,
}
def export_file(source: Path, input_root: Path, output_root: Path) -> ExportResult:
"""转换一个 XLSX,并以原子替换方式写出对应 CSV。"""
rows = iter_rows(source)
try:
_, headers = next(rows)
except StopIteration as exc:
raise CatalogExportError(f"{source.name} 没有表头") from exc
indexes = {header: position for position, header in enumerate(headers) if header}
missing = [header for header in REQUIRED_HEADERS if header not in indexes]
if missing:
raise CatalogExportError(f"{source.name} 缺少列: {', '.join(missing)}")
relative = source.relative_to(input_root)
output = (output_root / relative).with_suffix(".csv")
output.parent.mkdir(parents=True, exist_ok=True)
validation_counts: Counter[str] = Counter()
row_count = 0
valid_pdd_rows = 0
temp_name = ""
try:
with NamedTemporaryFile(
mode="w", encoding="utf-8-sig", newline="", delete=False, dir=output.parent,
prefix=output.stem + ".", suffix=".tmp"
) as temp:
temp_name = temp.name
writer = csv.DictWriter(temp, fieldnames=CSV_FIELDS, extrasaction="raise")
writer.writeheader()
for row_number, values in rows:
record = build_record(values, indexes, relative.as_posix(), row_number)
writer.writerow(record)
row_count += 1
if record["pdd_importable"] == "true":
valid_pdd_rows += 1
if record["validation_code"]:
for code in record["validation_code"].split(";"):
validation_counts[code] += 1
Path(temp_name).replace(output)
except Exception:
if temp_name:
Path(temp_name).unlink(missing_ok=True)
raise
return ExportResult(
source=source,
output=output,
rows=row_count,
valid_pdd_rows=valid_pdd_rows,
validation_counts=tuple(sorted(validation_counts.items())),
)
def discover_sources(input_path: Path) -> tuple[Path, Path, list[Path]]:
"""支持单文件或目录;目录模式保留其下原始相对路径。"""
if input_path.is_file():
if input_path.suffix.lower() != ".xlsx":
raise CatalogExportError("输入文件必须是 .xlsx")
return input_path, input_path.parent, [input_path]
if not input_path.is_dir():
raise CatalogExportError(f"找不到输入路径: {input_path}")
sources = sorted(path for path in input_path.rglob("*.xlsx") if not path.name.startswith("~$"))
if not sources:
raise CatalogExportError(f"目录中没有 .xlsx 文件: {input_path}")
return input_path, input_path, sources
def run(args: argparse.Namespace) -> int:
input_path = Path(args.input).resolve()
output_root = Path(args.output_dir).resolve()
_, input_root, sources = discover_sources(input_path)
if output_root == input_root or output_root.is_relative_to(input_root):
raise CatalogExportError("输出目录不能位于输入目录内,避免把 CSV 当成后续输入")
results = [export_file(source, input_root, output_root) for source in sources]
total_rows = sum(result.rows for result in results)
total_valid_pdd = sum(result.valid_pdd_rows for result in results)
validation_counts: Counter[str] = Counter()
for result in results:
validation_counts.update(dict(result.validation_counts))
print(
f"{result.source.name}: {result.rows} 行,PDD 可入库 {result.valid_pdd_rows} 行,"
f"输出 {result.output}"
)
print(f"转换完成:{len(results)} 个 XLSX,{total_rows} 行,PDD 可入库 {total_valid_pdd} 行。")
if validation_counts:
print("校验提示:" + ",".join(f"{code} {count}" for code, count in sorted(validation_counts.items())))
print("本工具未调用接口,也未写入任何数据库。")
return 0
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description="把第三方 ERP XLSX 导出为规范商品目录 CSV")
parser.add_argument("input", help="一个 XLSX 文件或包含 XLSX 的目录")
parser.add_argument("--output-dir", required=True, help="CSV 输出根目录(必须在输入目录之外)")
return parser
def main() -> int:
try:
return run(build_parser().parse_args())
except CatalogExportError as exc:
print(f"转换失败:{exc}", file=sys.stderr)
return 1
if __name__ == "__main__":
raise SystemExit(main())
@@ -0,0 +1 @@
lxml==6.1.1
@@ -0,0 +1,91 @@
import csv
import importlib.util
import sys
import tempfile
import unittest
from pathlib import Path
from xml.sax.saxutils import escape
from zipfile import ZIP_DEFLATED, ZipFile
MODULE_PATH = Path(__file__).with_name("export_thirdparty_catalog_xlsx.py")
SPEC = importlib.util.spec_from_file_location("thirdparty_export", MODULE_PATH)
assert SPEC and SPEC.loader
catalog = importlib.util.module_from_spec(SPEC)
sys.modules[SPEC.name] = catalog
SPEC.loader.exec_module(catalog)
HEADERS = [
"Parent SKU", "产品标题", "sku", "变种属性名称一", "变种属性名称二",
"变种属性值一", "变种属性值二", "主图(URL)地址", "来源URL", "店铺名",
"产品id", "更新时间",
]
def inline_cell(column, row, value):
return f'<c r="{column}{row}" t="inlineStr"><is><t>{escape(value)}</t></is></c>'
def make_xlsx(path, data_rows):
rows = []
for row_number, values in enumerate([HEADERS, *data_rows], start=1):
cells = "".join(inline_cell(chr(65 + index), row_number, value) for index, value in enumerate(values))
rows.append(f'<row r="{row_number}">{cells}</row>')
worksheet = (
'<?xml version="1.0" encoding="UTF-8"?>'
'<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">'
'<dimension ref="A1"/><sheetData>' + "".join(rows) + '</sheetData></worksheet>'
)
with ZipFile(path, "w", ZIP_DEFLATED) as archive:
archive.writestr("xl/worksheets/sheet1.xml", worksheet)
class ThirdPartyCatalogExportTest(unittest.TestCase):
def make_source(self, rows):
temp = tempfile.TemporaryDirectory()
root = Path(temp.name) / "source"
source = root / "pinduoduo 域名" / "sample.xlsx"
source.parent.mkdir(parents=True)
make_xlsx(source, rows)
self.addCleanup(temp.cleanup)
return root, source, Path(temp.name) / "output"
def test_错误_dimension_仍导出完整数据和规范_pdd_url(self):
root, source, output = self.make_source([
["MAIN", "标题", "SKU-A", "颜色分类", "尺码", "黑色", "M", "https://img.example/a.jpg", "https://mobile.pinduoduo.com/goods.html?goods_id=123456789&track=1", "店铺 A", "24948397488", "2026-08-19"],
["MAIN", "标题", "SKU-B", "尺碼", "顏色", "L", "白色", "https://img.example/a.jpg", "https://mobile.yangkeduo.com/goods.html?goods_id=987654321", "店铺 A", "24948397488", "2026-08-19"],
])
result = catalog.export_file(source, root, output)
self.assertEqual((result.rows, result.valid_pdd_rows), (2, 2))
with result.output.open(encoding="utf-8-sig", newline="") as handle:
exported = list(csv.DictReader(handle))
self.assertEqual(len(exported), 2)
self.assertEqual(exported[0]["pdd_goods_url"], "https://mobile.yangkeduo.com/goods.html?goods_id=123456789")
self.assertEqual((exported[0]["color"], exported[0]["size"], exported[0]["parse_ok"]), ("黑色", "M", "true"))
self.assertEqual((exported[1]["color"], exported[1]["size"]), ("白色", "L"))
self.assertEqual(exported[0]["source_file"], "pinduoduo 域名/sample.xlsx")
def test_无效_pdd_url_保留蝦皮字段但不创建关联(self):
root, source, output = self.make_source([
["MAIN", "标题", "SKU-A", "款式", "套餐", "甲", "乙", "https://img.example/a.jpg", "https://example.com/item", "店铺 A", "24948397488", ""],
])
result = catalog.export_file(source, root, output)
self.assertEqual((result.rows, result.valid_pdd_rows), (1, 0))
with result.output.open(encoding="utf-8-sig", newline="") as handle:
exported = next(csv.DictReader(handle))
self.assertEqual(exported["shopee_goods_id"], "24948397488")
self.assertEqual(exported["spec_raw"], "甲,乙")
self.assertEqual(exported["parse_ok"], "false")
self.assertEqual(exported["association_pdd_goods_id"], "")
self.assertEqual(exported["validation_code"], "PDD_URL_HOST_INVALID")
def test_输出目录不能位于输入目录内(self):
root, _, _ = self.make_source([])
args = catalog.build_parser().parse_args([str(root), "--output-dir", str(root / "csv")])
with self.assertRaisesRegex(catalog.CatalogExportError, "输出目录"):
catalog.run(args)
if __name__ == "__main__":
unittest.main()