feat: 完成T-204旧字段回写
实现 Excel 旧标题与旧封面路径回写原文件,按 source_file_abs/source_sheet/source_row 定位行,支持输出列自动追加、xlsm 保留 VBA、文件占用中文错误提示与 export_copy 另存副本。 Tab① 增加回写旧数据按钮与 WriteBackWorker,后台调用 excel.write_back,锁文件时提示关闭后重试;补充 Excel 与 GUI 单测覆盖回写、锁文件、另存副本和 worker 调用。 同步 harness 文档:T-204 标记完成,新增 T-204b 记录采集完成自动回写缺口,扩展 T-205 覆盖 Chrome 未启动/未登录引导保护,并更新 current-state、routes、api、architecture、requirements 与 progress。
This commit is contained in:
+214
-5
@@ -24,6 +24,17 @@ HEADER_ALIASES = {
|
||||
"alias": {"别名", "账号别名", "alias"},
|
||||
"item_id": {"商品id", "商品编号", "商品id号", "itemid", "item"},
|
||||
}
|
||||
OUTPUT_HEADERS = {
|
||||
"old_title": "旧标题",
|
||||
"old_cover_path": "旧封面图片路径",
|
||||
"new_title": "新标题",
|
||||
"new_cover_path": "新封面图片路径",
|
||||
"update_status": "更新状态",
|
||||
}
|
||||
OLD_WRITE_BACK_FIELDS = ("old_title", "old_cover_path")
|
||||
WRITEABLE_STAGES = {"collected", "generated", "applied"}
|
||||
LOCK_WINERRORS = {5, 32, 33}
|
||||
LOCK_ERRNOS = {13, 16}
|
||||
|
||||
|
||||
class ExcelError(RuntimeError):
|
||||
@@ -276,13 +287,211 @@ def match_summary(rows: list[dict], accounts: list) -> dict:
|
||||
}
|
||||
|
||||
|
||||
def write_back(batch_id, excel_path=None, path=None) -> dict:
|
||||
"""Placeholder for Excel result write-back, implemented in later tasks."""
|
||||
def _batch_tasks(batch_id, path=None):
|
||||
if not batch_id:
|
||||
raise ExcelError("缺少批次 ID,无法回写 Excel")
|
||||
db.init_db(path)
|
||||
tasks = db.list_tasks(batch_id=batch_id, path=path)
|
||||
if not tasks:
|
||||
raise ExcelError(f"批次没有可回写任务: {batch_id}")
|
||||
return tasks
|
||||
|
||||
raise ExcelError("Excel 回写将在 T-204/T-403 实现")
|
||||
|
||||
def _filter_tasks_by_excel_path(tasks, excel_path):
|
||||
if excel_path is None:
|
||||
return list(tasks)
|
||||
target = os.path.abspath(str(excel_path))
|
||||
filtered = [
|
||||
task for task in tasks
|
||||
if os.path.abspath(task.source_file_abs) == target
|
||||
]
|
||||
if not filtered:
|
||||
raise ExcelError(f"批次中没有来自该 Excel 的任务: {target}")
|
||||
return filtered
|
||||
|
||||
|
||||
def _has_write_back_data(task, fields) -> bool:
|
||||
if getattr(task, "stage", None) in WRITEABLE_STAGES:
|
||||
return True
|
||||
return any(getattr(task, field, None) not in (None, "") for field in fields)
|
||||
|
||||
|
||||
def _group_by_source_file(tasks, fields):
|
||||
groups = {}
|
||||
for task in tasks:
|
||||
if not _has_write_back_data(task, fields):
|
||||
continue
|
||||
source_file_abs = os.path.abspath(task.source_file_abs)
|
||||
groups.setdefault(source_file_abs, []).append(task)
|
||||
return groups
|
||||
|
||||
|
||||
def _is_locked_error(exc) -> bool:
|
||||
return (
|
||||
isinstance(exc, PermissionError)
|
||||
or getattr(exc, "winerror", None) in LOCK_WINERRORS
|
||||
or getattr(exc, "errno", None) in LOCK_ERRNOS
|
||||
)
|
||||
|
||||
|
||||
def _excel_suffix(path_value) -> str:
|
||||
return os.path.splitext(str(path_value))[1].lower()
|
||||
|
||||
|
||||
def _load_write_workbook(source_path):
|
||||
kwargs = {}
|
||||
if _excel_suffix(source_path) == ".xlsm":
|
||||
kwargs["keep_vba"] = True
|
||||
return load_workbook(source_path, **kwargs)
|
||||
|
||||
|
||||
def _ensure_output_columns(sheet, fields):
|
||||
header_by_normalized = {}
|
||||
for column in range(1, sheet.max_column + 1):
|
||||
header = _text(sheet.cell(row=1, column=column).value)
|
||||
if header:
|
||||
header_by_normalized.setdefault(_normalize_header(header), column)
|
||||
|
||||
columns = {}
|
||||
next_column = sheet.max_column + 1
|
||||
for field in fields:
|
||||
header = OUTPUT_HEADERS[field]
|
||||
normalized = _normalize_header(header)
|
||||
column = header_by_normalized.get(normalized)
|
||||
if column is None:
|
||||
column = next_column
|
||||
sheet.cell(row=1, column=column).value = header
|
||||
header_by_normalized[normalized] = column
|
||||
next_column += 1
|
||||
columns[field] = column
|
||||
return columns
|
||||
|
||||
|
||||
def _write_tasks_to_workbook(workbook, tasks, fields) -> int:
|
||||
rows = 0
|
||||
missing_sheets = []
|
||||
by_sheet = {}
|
||||
for task in tasks:
|
||||
by_sheet.setdefault(task.source_sheet, []).append(task)
|
||||
|
||||
for sheet_name, sheet_tasks in by_sheet.items():
|
||||
if sheet_name not in workbook.sheetnames:
|
||||
missing_sheets.append(sheet_name)
|
||||
continue
|
||||
sheet = workbook[sheet_name]
|
||||
columns = _ensure_output_columns(sheet, fields)
|
||||
for task in sheet_tasks:
|
||||
if not _has_write_back_data(task, fields):
|
||||
continue
|
||||
for field in fields:
|
||||
sheet.cell(
|
||||
row=int(task.source_row),
|
||||
column=columns[field],
|
||||
).value = getattr(task, field, None) or ""
|
||||
rows += 1
|
||||
|
||||
if missing_sheets:
|
||||
names = ", ".join(sorted(set(missing_sheets)))
|
||||
raise ExcelError(f"原 Excel 缺少导入时的工作表: {names}")
|
||||
return rows
|
||||
|
||||
|
||||
def _write_source_to_path(source_path, tasks, target_path, fields):
|
||||
workbook = None
|
||||
try:
|
||||
workbook = _load_write_workbook(source_path)
|
||||
except Exception as exc:
|
||||
if _is_locked_error(exc):
|
||||
raise ExcelError(f"Excel 文件被占用,请关闭后重试: {source_path}") from exc
|
||||
raise ExcelError(f"读取 Excel 失败: {source_path}: {exc}") from exc
|
||||
|
||||
try:
|
||||
rows = _write_tasks_to_workbook(workbook, tasks, fields)
|
||||
try:
|
||||
workbook.save(target_path)
|
||||
except Exception as exc:
|
||||
if _is_locked_error(exc):
|
||||
raise ExcelError(f"Excel 文件被占用,请关闭后重试: {target_path}") from exc
|
||||
raise ExcelError(f"保存 Excel 失败: {target_path}: {exc}") from exc
|
||||
return rows
|
||||
finally:
|
||||
if workbook is not None:
|
||||
workbook.close()
|
||||
|
||||
|
||||
def _write_result(batch_id, groups, target_for_source, fields):
|
||||
rows = 0
|
||||
written_files = []
|
||||
for source_path, tasks in groups.items():
|
||||
target_path = target_for_source(source_path)
|
||||
rows += _write_source_to_path(source_path, tasks, target_path, fields)
|
||||
written_files.append(os.path.abspath(target_path))
|
||||
return {
|
||||
"ok": True,
|
||||
"batch_id": batch_id,
|
||||
"files": len(written_files),
|
||||
"rows": rows,
|
||||
"written_files": written_files,
|
||||
}
|
||||
|
||||
|
||||
def write_back(batch_id, excel_path=None, path=None) -> dict:
|
||||
"""Write collected old-title/old-cover fields back to the original Excel file."""
|
||||
|
||||
_require_openpyxl()
|
||||
fields = OLD_WRITE_BACK_FIELDS
|
||||
tasks = _filter_tasks_by_excel_path(_batch_tasks(batch_id, path=path), excel_path)
|
||||
groups = _group_by_source_file(tasks, fields)
|
||||
return _write_result(
|
||||
batch_id,
|
||||
groups,
|
||||
lambda source_path: source_path,
|
||||
fields,
|
||||
)
|
||||
|
||||
|
||||
def _is_excel_output_path(path_value) -> bool:
|
||||
return _excel_suffix(path_value) in {".xlsx", ".xlsm"}
|
||||
|
||||
|
||||
def _unique_copy_path(out_dir, source_path):
|
||||
base_name = os.path.basename(source_path)
|
||||
stem, suffix = os.path.splitext(base_name)
|
||||
candidate = os.path.join(out_dir, f"{stem}_cmshopee回写{suffix}")
|
||||
index = 1
|
||||
while os.path.exists(candidate):
|
||||
candidate = os.path.join(out_dir, f"{stem}_cmshopee回写_{index}{suffix}")
|
||||
index += 1
|
||||
return candidate
|
||||
|
||||
|
||||
def export_copy(batch_id, out_dir_or_path, path=None) -> dict:
|
||||
"""Placeholder for exporting a copy when the original Excel is locked."""
|
||||
"""Export a write-back copy without touching the original Excel file."""
|
||||
|
||||
raise ExcelError("Excel 另存副本将在 T-204/T-403 实现")
|
||||
_require_openpyxl()
|
||||
if not out_dir_or_path:
|
||||
raise ExcelError("缺少另存路径,无法导出 Excel 副本")
|
||||
|
||||
fields = OLD_WRITE_BACK_FIELDS
|
||||
tasks = _batch_tasks(batch_id, path=path)
|
||||
groups = _group_by_source_file(tasks, fields)
|
||||
output_path = os.path.abspath(str(out_dir_or_path))
|
||||
output_is_file = _is_excel_output_path(output_path)
|
||||
if output_is_file and len(groups) > 1:
|
||||
raise ExcelError("多个源 Excel 另存副本时,目标必须是目录")
|
||||
if output_is_file:
|
||||
output_dir = os.path.dirname(output_path)
|
||||
if output_dir:
|
||||
os.makedirs(output_dir, exist_ok=True)
|
||||
|
||||
def target_for_source(source_path):
|
||||
if os.path.abspath(source_path) == output_path:
|
||||
raise ExcelError("另存副本路径不能与原 Excel 相同")
|
||||
return output_path
|
||||
else:
|
||||
os.makedirs(output_path, exist_ok=True)
|
||||
|
||||
def target_for_source(source_path):
|
||||
return _unique_copy_path(output_path, source_path)
|
||||
|
||||
return _write_result(batch_id, groups, target_for_source, fields)
|
||||
|
||||
+91
@@ -206,11 +206,14 @@ if QT_IMPORT_ERROR is None:
|
||||
self.last_import_stats = None
|
||||
self.collect_worker = None
|
||||
self.collect_thread = None
|
||||
self.write_back_worker = None
|
||||
self.write_back_thread = None
|
||||
|
||||
self.import_button = QPushButton("导入 Excel...")
|
||||
self.refresh_button = QPushButton("刷新")
|
||||
self.collect_button = QPushButton("采集旧标题/旧封面")
|
||||
self.stop_collect_button = QPushButton("停止")
|
||||
self.write_back_button = QPushButton("回写旧数据到 Excel")
|
||||
self.stop_collect_button.setEnabled(False)
|
||||
|
||||
toolbar = QHBoxLayout()
|
||||
@@ -218,6 +221,7 @@ if QT_IMPORT_ERROR is None:
|
||||
toolbar.addWidget(self.refresh_button)
|
||||
toolbar.addWidget(self.collect_button)
|
||||
toolbar.addWidget(self.stop_collect_button)
|
||||
toolbar.addWidget(self.write_back_button)
|
||||
toolbar.addStretch(1)
|
||||
|
||||
self.summary_label = QLabel("未导入任务")
|
||||
@@ -254,6 +258,7 @@ if QT_IMPORT_ERROR is None:
|
||||
self.refresh_button.clicked.connect(self.refresh_tasks)
|
||||
self.collect_button.clicked.connect(self.collect_old_data)
|
||||
self.stop_collect_button.clicked.connect(self.stop_collect)
|
||||
self.write_back_button.clicked.connect(self.write_back_old_data)
|
||||
self.show_all_button.clicked.connect(self.show_all_tasks)
|
||||
self.show_unmatched_button.clicked.connect(self.show_unmatched_tasks)
|
||||
|
||||
@@ -339,17 +344,57 @@ if QT_IMPORT_ERROR is None:
|
||||
self.collect_worker.cancel()
|
||||
self._set_status("正在停止采集...")
|
||||
|
||||
def write_back_old_data(self, checked=False):
|
||||
batch_id = self._active_batch_id()
|
||||
if not batch_id:
|
||||
self._set_status("没有可回写批次")
|
||||
return
|
||||
worker = WriteBackWorker(batch_id, db_path=self.db_path)
|
||||
worker.failed.connect(self._on_write_back_failed)
|
||||
worker.finished.connect(self._on_write_back_finished)
|
||||
thread = run_worker(worker, thread_name="WriteBackWorker", start=False)
|
||||
thread.finished.connect(lambda: self._forget_write_back_thread(thread))
|
||||
self.write_back_worker = worker
|
||||
self.write_back_thread = thread
|
||||
self._set_write_back_running(True)
|
||||
self._set_status("正在回写旧数据到 Excel...")
|
||||
thread.start()
|
||||
|
||||
def _active_batch_id(self):
|
||||
if self.current_batch_id:
|
||||
return self.current_batch_id
|
||||
batch_ids = {
|
||||
task.batch_id
|
||||
for task in self.model.all_tasks
|
||||
if getattr(task, "batch_id", None)
|
||||
}
|
||||
if len(batch_ids) == 1:
|
||||
return next(iter(batch_ids))
|
||||
return None
|
||||
|
||||
def _set_collect_running(self, running):
|
||||
self.import_button.setEnabled(not running)
|
||||
self.refresh_button.setEnabled(not running)
|
||||
self.collect_button.setEnabled(not running)
|
||||
self.write_back_button.setEnabled(not running)
|
||||
self.stop_collect_button.setEnabled(running)
|
||||
|
||||
def _set_write_back_running(self, running):
|
||||
self.import_button.setEnabled(not running)
|
||||
self.refresh_button.setEnabled(not running)
|
||||
self.collect_button.setEnabled(not running)
|
||||
self.write_back_button.setEnabled(not running)
|
||||
|
||||
def _forget_collect_thread(self, thread):
|
||||
if self.collect_thread is thread:
|
||||
self.collect_thread = None
|
||||
self.collect_worker = None
|
||||
|
||||
def _forget_write_back_thread(self, thread):
|
||||
if self.write_back_thread is thread:
|
||||
self.write_back_thread = None
|
||||
self.write_back_worker = None
|
||||
|
||||
def _on_collect_progress(self, payload):
|
||||
self._set_status(
|
||||
"采集进度:{done}/{total},成功{collected},略过{skipped},失败{failed}".format(
|
||||
@@ -388,6 +433,27 @@ if QT_IMPORT_ERROR is None:
|
||||
)
|
||||
)
|
||||
|
||||
def _on_write_back_failed(self, task_id, error):
|
||||
message = f"Excel 回写失败:{error}"
|
||||
if "被占用" in str(error):
|
||||
message += "\n请关闭原 Excel 后重试;SQLite 已保留采集结果,也可另存副本。"
|
||||
QMessageBox.warning(self, "回写旧数据", message)
|
||||
self._set_status(message.replace("\n", " "))
|
||||
|
||||
def _on_write_back_finished(self, payload):
|
||||
self._set_write_back_running(False)
|
||||
if payload.get("ok") is False:
|
||||
error = payload.get("error") or "未知错误"
|
||||
self._set_status(f"Excel 回写失败:{error}")
|
||||
return
|
||||
self.refresh_tasks()
|
||||
self._set_status(
|
||||
"Excel 回写完成:文件{files},行{rows}".format(
|
||||
files=payload.get("files", 0),
|
||||
rows=payload.get("rows", 0),
|
||||
)
|
||||
)
|
||||
|
||||
def show_all_tasks(self, checked=False):
|
||||
self.model.set_filter_mode("all")
|
||||
self._update_empty_label(len(self.model.all_tasks))
|
||||
@@ -669,6 +735,31 @@ if QT_IMPORT_ERROR is None:
|
||||
)
|
||||
|
||||
|
||||
class WriteBackWorker(BaseWorker):
|
||||
"""Write collected old fields back to Excel in a background thread."""
|
||||
|
||||
def __init__(self, batch_id, db_path=None, excel_path=None):
|
||||
super().__init__()
|
||||
self.batch_id = batch_id
|
||||
self.db_path = db_path
|
||||
self.excel_path = excel_path
|
||||
|
||||
def execute(self):
|
||||
result = excel.write_back(
|
||||
self.batch_id,
|
||||
excel_path=self.excel_path,
|
||||
path=self.db_path,
|
||||
)
|
||||
self.progress.emit(
|
||||
{
|
||||
"done": result.get("rows", 0),
|
||||
"total": result.get("rows", 0),
|
||||
"files": result.get("files", 0),
|
||||
}
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
class AccountLoginCheckWorker(BaseWorker):
|
||||
def __init__(self, account, db_path=None, config=None, timeout=8):
|
||||
super().__init__()
|
||||
|
||||
Reference in New Issue
Block a user