feat: 导出第三方商品规范 CSV (#280)
This commit is contained in:
@@ -0,0 +1,396 @@
|
||||
#!/usr/bin/env python3
|
||||
"""把第三方 ERP 商品工作簿转换为后续商品目录导入使用的规范 CSV。
|
||||
|
||||
这个工具只读取 XLSX、在指定目录写 CSV;不发送 HTTP 请求,也不会连接数据库。
|
||||
部分第三方工作簿把 OOXML 工作表 dimension 错写成 A1,不能用依赖 dimension 的
|
||||
Excel 读取器。本工具直接流式读取工作表 XML,因此能稳定处理这类大文件。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
import re
|
||||
import sys
|
||||
from collections import Counter
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from tempfile import NamedTemporaryFile
|
||||
from typing import Iterator
|
||||
from urllib.parse import parse_qs, urlparse
|
||||
from zipfile import ZipFile
|
||||
|
||||
from lxml import etree
|
||||
|
||||
|
||||
SPREADSHEET_NS = "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}"
|
||||
ROW_TAG = SPREADSHEET_NS + "row"
|
||||
CELL_TAG = SPREADSHEET_NS + "c"
|
||||
TEXT_TAG = SPREADSHEET_NS + "t"
|
||||
VALUE_TAG = SPREADSHEET_NS + "v"
|
||||
INLINE_TEXT_TAG = SPREADSHEET_NS + "is"
|
||||
GOODS_ID_PATTERN = re.compile(r"^\d{5,20}$")
|
||||
PDD_HOSTS = {"mobile.pinduoduo.com", "mobile.yangkeduo.com"}
|
||||
|
||||
REQUIRED_HEADERS = (
|
||||
"Parent SKU",
|
||||
"产品标题",
|
||||
"sku",
|
||||
"变种属性名称一",
|
||||
"变种属性名称二",
|
||||
"变种属性值一",
|
||||
"变种属性值二",
|
||||
"主图(URL)地址",
|
||||
"来源URL",
|
||||
"店铺名",
|
||||
"产品id",
|
||||
)
|
||||
|
||||
# 每个输入数据行对应一条规范 CSV 行。后续接口提交器按商品 ID 去重即可。
|
||||
CSV_FIELDS = (
|
||||
"source_file",
|
||||
"source_row",
|
||||
"source_updated_at",
|
||||
"shopee_goods_id",
|
||||
"shopee_title",
|
||||
"shopee_status",
|
||||
"shopee_shop_name",
|
||||
"shopee_image_url",
|
||||
"shopee_main_sku_code",
|
||||
"shopee_sku_id",
|
||||
"shopee_sku_code",
|
||||
"spec_raw",
|
||||
"color",
|
||||
"size",
|
||||
"advice",
|
||||
"parse_ok",
|
||||
"pdd_goods_id",
|
||||
"pdd_goods_url",
|
||||
"pdd_title",
|
||||
"pdd_shop_name",
|
||||
"association_shopee_goods_id",
|
||||
"association_pdd_goods_id",
|
||||
"source_pdd_url",
|
||||
"pdd_importable",
|
||||
"validation_code",
|
||||
)
|
||||
|
||||
|
||||
class CatalogExportError(RuntimeError):
|
||||
"""源工作簿不符合可转换要求。"""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ExportResult:
|
||||
"""一个输入工作簿的导出统计。"""
|
||||
|
||||
source: Path
|
||||
output: Path
|
||||
rows: int
|
||||
valid_pdd_rows: int
|
||||
validation_counts: tuple[tuple[str, int], ...]
|
||||
|
||||
|
||||
def text_of(element: etree._Element) -> str:
|
||||
"""合并 OOXML 富文本的所有文本节点。"""
|
||||
|
||||
return "".join(element.itertext()).strip()
|
||||
|
||||
|
||||
def column_index(cell_reference: str) -> int:
|
||||
"""把 A、AA 等 Excel 列号转成从零开始的索引。"""
|
||||
|
||||
letters = "".join(character for character in cell_reference if character.isalpha())
|
||||
value = 0
|
||||
for character in letters.upper():
|
||||
value = value * 26 + ord(character) - ord("A") + 1
|
||||
return value - 1
|
||||
|
||||
|
||||
def read_shared_strings(archive: ZipFile) -> list[str]:
|
||||
"""按需读取 shared strings;当前来源主要使用 inlineStr。"""
|
||||
|
||||
try:
|
||||
handle = archive.open("xl/sharedStrings.xml")
|
||||
except KeyError:
|
||||
return []
|
||||
with handle:
|
||||
return [
|
||||
text_of(element)
|
||||
for _, element in etree.iterparse(handle, events=("end",), tag=SPREADSHEET_NS + "si")
|
||||
]
|
||||
|
||||
|
||||
def cell_value(cell: etree._Element, shared_strings: list[str]) -> str:
|
||||
"""读取一个 OOXML 单元格的文本值,不依赖错误的 dimension。"""
|
||||
|
||||
cell_type = cell.get("t", "")
|
||||
if cell_type == "inlineStr":
|
||||
inline = cell.find(INLINE_TEXT_TAG)
|
||||
return text_of(inline) if inline is not None else ""
|
||||
value = cell.find(VALUE_TAG)
|
||||
if value is None:
|
||||
return ""
|
||||
raw = text_of(value)
|
||||
if cell_type == "s":
|
||||
try:
|
||||
return shared_strings[int(raw)]
|
||||
except (IndexError, ValueError) as exc:
|
||||
raise CatalogExportError(f"shared string 索引无效: {raw!r}") from exc
|
||||
return raw
|
||||
|
||||
|
||||
def worksheet_name(archive: ZipFile) -> str:
|
||||
"""选择工作簿的第一个工作表;第三方导出只有一个数据表。"""
|
||||
|
||||
sheets = sorted(
|
||||
name
|
||||
for name in archive.namelist()
|
||||
if name.startswith("xl/worksheets/sheet") and name.endswith(".xml")
|
||||
)
|
||||
if not sheets:
|
||||
raise CatalogExportError("XLSX 中没有工作表")
|
||||
return sheets[0]
|
||||
|
||||
|
||||
def iter_rows(path: Path) -> Iterator[tuple[int, list[str]]]:
|
||||
"""流式迭代第一个工作表的行,跳过错误的 dimension 元数据。"""
|
||||
|
||||
try:
|
||||
archive = ZipFile(path)
|
||||
except OSError as exc:
|
||||
raise CatalogExportError(f"无法打开工作簿: {path}") from exc
|
||||
with archive:
|
||||
shared_strings = read_shared_strings(archive)
|
||||
with archive.open(worksheet_name(archive)) as worksheet:
|
||||
for _, row in etree.iterparse(worksheet, events=("end",), tag=ROW_TAG):
|
||||
cells: dict[int, str] = {}
|
||||
for cell in row.findall(CELL_TAG):
|
||||
reference = cell.get("r", "")
|
||||
if reference:
|
||||
cells[column_index(reference)] = cell_value(cell, shared_strings)
|
||||
row_number = int(row.get("r", "0"))
|
||||
if cells:
|
||||
values = ["" for _ in range(max(cells) + 1)]
|
||||
for index, value in cells.items():
|
||||
values[index] = value
|
||||
yield row_number, values
|
||||
row.clear()
|
||||
while row.getprevious() is not None:
|
||||
del row.getparent()[0]
|
||||
|
||||
|
||||
def normalize_header(value: str) -> str:
|
||||
"""统一繁简体和空白,便于严格识别来源中的规格列名。"""
|
||||
|
||||
return (
|
||||
value.strip()
|
||||
.lower()
|
||||
.replace(" ", "")
|
||||
.replace("顏", "颜")
|
||||
.replace("色", "色")
|
||||
.replace("分類", "分类")
|
||||
.replace("碼", "码")
|
||||
)
|
||||
|
||||
|
||||
def dimension_kind(header: str) -> str:
|
||||
"""只依据明确的字段名判断颜色或尺码,其他列不猜测。"""
|
||||
|
||||
normalized = normalize_header(header)
|
||||
if "尺码" in normalized or "尺寸" in normalized:
|
||||
return "size"
|
||||
if "颜色" in normalized:
|
||||
return "color"
|
||||
return ""
|
||||
|
||||
|
||||
def value_at(values: list[str], index: int | None) -> str:
|
||||
return values[index].strip() if index is not None and index < len(values) else ""
|
||||
|
||||
|
||||
def canonical_pdd(raw_url: str) -> tuple[str, str, str]:
|
||||
"""提取 PDD goods_id;失败时返回明确校验码而不是丢弃蝦皮数据。"""
|
||||
|
||||
if not raw_url:
|
||||
return "", "", "PDD_URL_MISSING"
|
||||
parsed = urlparse(raw_url)
|
||||
if parsed.scheme not in {"http", "https"} or parsed.hostname not in PDD_HOSTS:
|
||||
return "", "", "PDD_URL_HOST_INVALID"
|
||||
goods_id = parse_qs(parsed.query).get("goods_id", [""])[0].strip()
|
||||
if not GOODS_ID_PATTERN.fullmatch(goods_id):
|
||||
return "", "", "PDD_GOODS_ID_MISSING"
|
||||
return goods_id, f"https://mobile.yangkeduo.com/goods.html?goods_id={goods_id}", ""
|
||||
|
||||
|
||||
def join_spec(first: str, second: str) -> str:
|
||||
"""按源列顺序保留规格原文;值内即使有逗号也不拆分。"""
|
||||
|
||||
return ",".join(value for value in (first, second) if value)
|
||||
|
||||
|
||||
def build_record(
|
||||
values: list[str],
|
||||
indexes: dict[str, int],
|
||||
source_file: str,
|
||||
row_number: int,
|
||||
) -> dict[str, str]:
|
||||
"""把来源一行映射成后续商品目录入库所需字段。"""
|
||||
|
||||
first_name = value_at(values, indexes["变种属性名称一"])
|
||||
second_name = value_at(values, indexes["变种属性名称二"])
|
||||
first_value = value_at(values, indexes["变种属性值一"])
|
||||
second_value = value_at(values, indexes["变种属性值二"])
|
||||
color = ""
|
||||
size = ""
|
||||
if dimension_kind(first_name) == "color":
|
||||
color = first_value
|
||||
elif dimension_kind(first_name) == "size":
|
||||
size = first_value
|
||||
if dimension_kind(second_name) == "color":
|
||||
color = second_value
|
||||
elif dimension_kind(second_name) == "size":
|
||||
size = second_value
|
||||
raw_pdd_url = value_at(values, indexes["来源URL"])
|
||||
pdd_goods_id, pdd_goods_url, validation_code = canonical_pdd(raw_pdd_url)
|
||||
shopee_goods_id = value_at(values, indexes["产品id"])
|
||||
if not GOODS_ID_PATTERN.fullmatch(shopee_goods_id):
|
||||
validation_code = ";".join(
|
||||
value for value in (validation_code, "SHOPEE_GOODS_ID_INVALID") if value
|
||||
)
|
||||
parse_ok = bool(color or size)
|
||||
return {
|
||||
"source_file": source_file,
|
||||
"source_row": str(row_number),
|
||||
"source_updated_at": value_at(values, indexes.get("更新时间")),
|
||||
"shopee_goods_id": shopee_goods_id,
|
||||
"shopee_title": value_at(values, indexes["产品标题"]),
|
||||
"shopee_status": "",
|
||||
"shopee_shop_name": value_at(values, indexes["店铺名"]),
|
||||
"shopee_image_url": value_at(values, indexes["主图(URL)地址"]),
|
||||
"shopee_main_sku_code": value_at(values, indexes["Parent SKU"]),
|
||||
"shopee_sku_id": "",
|
||||
"shopee_sku_code": value_at(values, indexes["sku"]),
|
||||
"spec_raw": join_spec(first_value, second_value),
|
||||
"color": color,
|
||||
"size": size,
|
||||
"advice": "",
|
||||
"parse_ok": "true" if parse_ok else "false",
|
||||
"pdd_goods_id": pdd_goods_id,
|
||||
"pdd_goods_url": pdd_goods_url,
|
||||
"pdd_title": "",
|
||||
"pdd_shop_name": "",
|
||||
"association_shopee_goods_id": shopee_goods_id if pdd_goods_id else "",
|
||||
"association_pdd_goods_id": pdd_goods_id,
|
||||
"source_pdd_url": raw_pdd_url,
|
||||
"pdd_importable": "true" if pdd_goods_id else "false",
|
||||
"validation_code": validation_code,
|
||||
}
|
||||
|
||||
|
||||
def export_file(source: Path, input_root: Path, output_root: Path) -> ExportResult:
|
||||
"""转换一个 XLSX,并以原子替换方式写出对应 CSV。"""
|
||||
|
||||
rows = iter_rows(source)
|
||||
try:
|
||||
_, headers = next(rows)
|
||||
except StopIteration as exc:
|
||||
raise CatalogExportError(f"{source.name} 没有表头") from exc
|
||||
indexes = {header: position for position, header in enumerate(headers) if header}
|
||||
missing = [header for header in REQUIRED_HEADERS if header not in indexes]
|
||||
if missing:
|
||||
raise CatalogExportError(f"{source.name} 缺少列: {', '.join(missing)}")
|
||||
|
||||
relative = source.relative_to(input_root)
|
||||
output = (output_root / relative).with_suffix(".csv")
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
validation_counts: Counter[str] = Counter()
|
||||
row_count = 0
|
||||
valid_pdd_rows = 0
|
||||
temp_name = ""
|
||||
try:
|
||||
with NamedTemporaryFile(
|
||||
mode="w", encoding="utf-8-sig", newline="", delete=False, dir=output.parent,
|
||||
prefix=output.stem + ".", suffix=".tmp"
|
||||
) as temp:
|
||||
temp_name = temp.name
|
||||
writer = csv.DictWriter(temp, fieldnames=CSV_FIELDS, extrasaction="raise")
|
||||
writer.writeheader()
|
||||
for row_number, values in rows:
|
||||
record = build_record(values, indexes, relative.as_posix(), row_number)
|
||||
writer.writerow(record)
|
||||
row_count += 1
|
||||
if record["pdd_importable"] == "true":
|
||||
valid_pdd_rows += 1
|
||||
if record["validation_code"]:
|
||||
for code in record["validation_code"].split(";"):
|
||||
validation_counts[code] += 1
|
||||
Path(temp_name).replace(output)
|
||||
except Exception:
|
||||
if temp_name:
|
||||
Path(temp_name).unlink(missing_ok=True)
|
||||
raise
|
||||
return ExportResult(
|
||||
source=source,
|
||||
output=output,
|
||||
rows=row_count,
|
||||
valid_pdd_rows=valid_pdd_rows,
|
||||
validation_counts=tuple(sorted(validation_counts.items())),
|
||||
)
|
||||
|
||||
|
||||
def discover_sources(input_path: Path) -> tuple[Path, Path, list[Path]]:
|
||||
"""支持单文件或目录;目录模式保留其下原始相对路径。"""
|
||||
|
||||
if input_path.is_file():
|
||||
if input_path.suffix.lower() != ".xlsx":
|
||||
raise CatalogExportError("输入文件必须是 .xlsx")
|
||||
return input_path, input_path.parent, [input_path]
|
||||
if not input_path.is_dir():
|
||||
raise CatalogExportError(f"找不到输入路径: {input_path}")
|
||||
sources = sorted(path for path in input_path.rglob("*.xlsx") if not path.name.startswith("~$"))
|
||||
if not sources:
|
||||
raise CatalogExportError(f"目录中没有 .xlsx 文件: {input_path}")
|
||||
return input_path, input_path, sources
|
||||
|
||||
|
||||
def run(args: argparse.Namespace) -> int:
|
||||
input_path = Path(args.input).resolve()
|
||||
output_root = Path(args.output_dir).resolve()
|
||||
_, input_root, sources = discover_sources(input_path)
|
||||
if output_root == input_root or output_root.is_relative_to(input_root):
|
||||
raise CatalogExportError("输出目录不能位于输入目录内,避免把 CSV 当成后续输入")
|
||||
results = [export_file(source, input_root, output_root) for source in sources]
|
||||
total_rows = sum(result.rows for result in results)
|
||||
total_valid_pdd = sum(result.valid_pdd_rows for result in results)
|
||||
validation_counts: Counter[str] = Counter()
|
||||
for result in results:
|
||||
validation_counts.update(dict(result.validation_counts))
|
||||
print(
|
||||
f"{result.source.name}: {result.rows} 行,PDD 可入库 {result.valid_pdd_rows} 行,"
|
||||
f"输出 {result.output}"
|
||||
)
|
||||
print(f"转换完成:{len(results)} 个 XLSX,{total_rows} 行,PDD 可入库 {total_valid_pdd} 行。")
|
||||
if validation_counts:
|
||||
print("校验提示:" + ",".join(f"{code} {count}" for code, count in sorted(validation_counts.items())))
|
||||
print("本工具未调用接口,也未写入任何数据库。")
|
||||
return 0
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(description="把第三方 ERP XLSX 导出为规范商品目录 CSV")
|
||||
parser.add_argument("input", help="一个 XLSX 文件或包含 XLSX 的目录")
|
||||
parser.add_argument("--output-dir", required=True, help="CSV 输出根目录(必须在输入目录之外)")
|
||||
return parser
|
||||
|
||||
|
||||
def main() -> int:
|
||||
try:
|
||||
return run(build_parser().parse_args())
|
||||
except CatalogExportError as exc:
|
||||
print(f"转换失败:{exc}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1 @@
|
||||
lxml==6.1.1
|
||||
@@ -0,0 +1,91 @@
|
||||
import csv
|
||||
import importlib.util
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from xml.sax.saxutils import escape
|
||||
from zipfile import ZIP_DEFLATED, ZipFile
|
||||
|
||||
|
||||
MODULE_PATH = Path(__file__).with_name("export_thirdparty_catalog_xlsx.py")
|
||||
SPEC = importlib.util.spec_from_file_location("thirdparty_export", MODULE_PATH)
|
||||
assert SPEC and SPEC.loader
|
||||
catalog = importlib.util.module_from_spec(SPEC)
|
||||
sys.modules[SPEC.name] = catalog
|
||||
SPEC.loader.exec_module(catalog)
|
||||
|
||||
|
||||
HEADERS = [
|
||||
"Parent SKU", "产品标题", "sku", "变种属性名称一", "变种属性名称二",
|
||||
"变种属性值一", "变种属性值二", "主图(URL)地址", "来源URL", "店铺名",
|
||||
"产品id", "更新时间",
|
||||
]
|
||||
|
||||
|
||||
def inline_cell(column, row, value):
|
||||
return f'<c r="{column}{row}" t="inlineStr"><is><t>{escape(value)}</t></is></c>'
|
||||
|
||||
|
||||
def make_xlsx(path, data_rows):
|
||||
rows = []
|
||||
for row_number, values in enumerate([HEADERS, *data_rows], start=1):
|
||||
cells = "".join(inline_cell(chr(65 + index), row_number, value) for index, value in enumerate(values))
|
||||
rows.append(f'<row r="{row_number}">{cells}</row>')
|
||||
worksheet = (
|
||||
'<?xml version="1.0" encoding="UTF-8"?>'
|
||||
'<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">'
|
||||
'<dimension ref="A1"/><sheetData>' + "".join(rows) + '</sheetData></worksheet>'
|
||||
)
|
||||
with ZipFile(path, "w", ZIP_DEFLATED) as archive:
|
||||
archive.writestr("xl/worksheets/sheet1.xml", worksheet)
|
||||
|
||||
|
||||
class ThirdPartyCatalogExportTest(unittest.TestCase):
|
||||
def make_source(self, rows):
|
||||
temp = tempfile.TemporaryDirectory()
|
||||
root = Path(temp.name) / "source"
|
||||
source = root / "pinduoduo 域名" / "sample.xlsx"
|
||||
source.parent.mkdir(parents=True)
|
||||
make_xlsx(source, rows)
|
||||
self.addCleanup(temp.cleanup)
|
||||
return root, source, Path(temp.name) / "output"
|
||||
|
||||
def test_错误_dimension_仍导出完整数据和规范_pdd_url(self):
|
||||
root, source, output = self.make_source([
|
||||
["MAIN", "标题", "SKU-A", "颜色分类", "尺码", "黑色", "M", "https://img.example/a.jpg", "https://mobile.pinduoduo.com/goods.html?goods_id=123456789&track=1", "店铺 A", "24948397488", "2026-08-19"],
|
||||
["MAIN", "标题", "SKU-B", "尺碼", "顏色", "L", "白色", "https://img.example/a.jpg", "https://mobile.yangkeduo.com/goods.html?goods_id=987654321", "店铺 A", "24948397488", "2026-08-19"],
|
||||
])
|
||||
result = catalog.export_file(source, root, output)
|
||||
self.assertEqual((result.rows, result.valid_pdd_rows), (2, 2))
|
||||
with result.output.open(encoding="utf-8-sig", newline="") as handle:
|
||||
exported = list(csv.DictReader(handle))
|
||||
self.assertEqual(len(exported), 2)
|
||||
self.assertEqual(exported[0]["pdd_goods_url"], "https://mobile.yangkeduo.com/goods.html?goods_id=123456789")
|
||||
self.assertEqual((exported[0]["color"], exported[0]["size"], exported[0]["parse_ok"]), ("黑色", "M", "true"))
|
||||
self.assertEqual((exported[1]["color"], exported[1]["size"]), ("白色", "L"))
|
||||
self.assertEqual(exported[0]["source_file"], "pinduoduo 域名/sample.xlsx")
|
||||
|
||||
def test_无效_pdd_url_保留蝦皮字段但不创建关联(self):
|
||||
root, source, output = self.make_source([
|
||||
["MAIN", "标题", "SKU-A", "款式", "套餐", "甲", "乙", "https://img.example/a.jpg", "https://example.com/item", "店铺 A", "24948397488", ""],
|
||||
])
|
||||
result = catalog.export_file(source, root, output)
|
||||
self.assertEqual((result.rows, result.valid_pdd_rows), (1, 0))
|
||||
with result.output.open(encoding="utf-8-sig", newline="") as handle:
|
||||
exported = next(csv.DictReader(handle))
|
||||
self.assertEqual(exported["shopee_goods_id"], "24948397488")
|
||||
self.assertEqual(exported["spec_raw"], "甲,乙")
|
||||
self.assertEqual(exported["parse_ok"], "false")
|
||||
self.assertEqual(exported["association_pdd_goods_id"], "")
|
||||
self.assertEqual(exported["validation_code"], "PDD_URL_HOST_INVALID")
|
||||
|
||||
def test_输出目录不能位于输入目录内(self):
|
||||
root, _, _ = self.make_source([])
|
||||
args = catalog.build_parser().parse_args([str(root), "--output-dir", str(root / "csv")])
|
||||
with self.assertRaisesRegex(catalog.CatalogExportError, "输出目录"):
|
||||
catalog.run(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user