chore: curate monthly assessment evidence and scripts

This commit is contained in:
kkfluous
2026-08-25 12:26:35 +08:00
parent 2956496fb0
commit 52dde9195a
117 changed files with 10743 additions and 179 deletions
+216
View File
@@ -0,0 +1,216 @@
from collections import defaultdict
from datetime import datetime
from pathlib import Path
from zipfile import ZipFile
from xml.etree import ElementTree as ET
import re
from openpyxl import Workbook, load_workbook
from openpyxl.styles import Alignment, Border, Font, PatternFill, Side
from openpyxl.utils import get_column_letter
BASE = Path("/Users/kkfluous/Downloads")
ASSESSMENT = BASE / "7月核算结果/租赁任务考核_2026年7月.xlsx"
PLATFORM_DIR = BASE / "广州新能源平台_纯氢纯电里程_2026年7月_按日"
GPS_FILES = [
BASE / "里程统计[天][2026-07-01至2026-07-31].xlsx",
BASE / "里程统计[天][2026-07-01至2026-07-31] (1).xlsx",
]
OUTPUT = BASE / "7月核算结果/7月考核车辆多源里程汇总.xlsx"
NS = "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}"
def normalize_plate(value):
return re.sub(r"\s+", "", str(value or "")).upper()
def parse_gps_file(path):
result = {}
with ZipFile(path) as zf:
root = ET.fromstring(zf.read("xl/worksheets/sheet1.xml"))
for row in root.findall(f".//{NS}row"):
if int(row.get("r")) < 4:
continue
values = {}
for cell in row.findall(f"{NS}c"):
col = re.match(r"[A-Z]+", cell.get("r")).group()
if cell.get("t") == "inlineStr":
values[col] = "".join(t.text or "" for t in cell.findall(f".//{NS}t"))
else:
node = cell.find(f"{NS}v")
values[col] = node.text if node is not None else None
plate = normalize_plate(values.get("H"))
date_text = str(values.get("O") or "").strip()
if not plate or plate == "合计" or not re.fullmatch(r"2026-07-\d{2}", date_text):
continue
result[(plate, datetime.strptime(date_text, "%Y-%m-%d").date())] = float(values.get("P") or 0)
return result
platform = defaultdict(lambda: [0.0, 0.0])
platform_presence = set()
for path in sorted(PLATFORM_DIR.glob("2026-07-??_纯氢纯电里程.xlsx")):
wb = load_workbook(path, read_only=True, data_only=True)
ws = wb.active
for row in ws.iter_rows(min_row=2, values_only=True):
plate = normalize_plate(row[5])
date_match = re.search(r"2026-07-\d{2}", str(row[6] or ""))
if not plate or not date_match:
continue
date = datetime.strptime(date_match.group(), "%Y-%m-%d").date()
electric = float(row[7] or 0)
hydrogen = float(row[8] or 0)
platform[(plate, date)][0] += hydrogen
platform[(plate, date)][1] += electric
platform_presence.add((plate, date))
wb.close()
gps = {}
for path in GPS_FILES:
incoming = parse_gps_file(path)
overlap = set(gps) & set(incoming)
if overlap:
raise RuntimeError(f"GPS来源存在重复车辆日期:{sorted(overlap)[:5]}")
gps.update(incoming)
source_wb = load_workbook(ASSESSMENT, read_only=True, data_only=True)
source_ws = source_wb["业务考核视图"]
headers = [c.value for c in source_ws[1]]
idx = {name: i for i, name in enumerate(headers)}
records = []
for row in source_ws.iter_rows(min_row=2, values_only=True):
plate = normalize_plate(row[idx["车牌号"]])
if not plate:
continue
start = row[idx["考核开始日期"]].date()
end = row[idx["考核结束日期"]].date()
dates = [(start.fromordinal(start.toordinal() + i)) for i in range((end - start).days + 1)]
p_dates = [d for d in dates if (plate, d) in platform_presence]
if p_dates:
hydrogen = sum(platform[(plate, d)][0] for d in p_dates)
electric = sum(platform[(plate, d)][1] for d in p_dates)
else:
hydrogen = electric = None
test_mileage = row[idx["测试里程(km)"]]
is_test = isinstance(test_mileage, (int, float)) and test_mileage > 0
g_dates = [d for d in dates if (plate, d) in gps]
gps_total = sum(gps[(plate, d)] for d in g_dates) if g_dates else None
# 7月最终考核源表的“实际行驶里程”已按原处理链路写入测试车辆GPS里程:
# 优先智能管车日报,缺失时用两份GPS日明细补充。补充明细未命中时,
# 必须从最终源表回填,不能误标为“无来源”。
if gps_total is None and is_test:
gps_total = row[idx["实际行驶里程(km)"]]
records.append({
"plate": plate,
"customer": row[idx["客户名称"]] or "",
"start": start,
"end": end,
"days": (end - start).days + 1,
"test": "" if is_test else "",
"hydrogen": hydrogen,
"electric": electric,
"tbox": row[idx["原TBOX里程(km)"]] if row[idx["原TBOX里程(km)"]] is not None else None,
"gps": gps_total,
})
source_wb.close()
wb = Workbook()
ws = wb.active
ws.title = "7月考核车辆里程汇总"
columns = [
"序号", "车牌", "客户", "考核开始日期", "考核结束日期", "考核天数", "是否存在测试",
"纯氢里程(km)", "纯电里程(km)", "广州平台总里程(km)", "TBOX里程(km)", "GPS里程(km)",
]
ws.append(columns)
for n, rec in enumerate(records, 1):
row_num = ws.max_row + 1
ws.append([
n, rec["plate"], rec["customer"], rec["start"], rec["end"], rec["days"], rec["test"],
rec["hydrogen"] if rec["hydrogen"] is not None else "无来源",
rec["electric"] if rec["electric"] is not None else "无来源",
round(rec["hydrogen"] + rec["electric"], 3) if rec["hydrogen"] is not None else "无来源",
rec["tbox"] if rec["tbox"] is not None else "无来源",
rec["gps"] if rec["gps"] is not None else "无来源",
])
ws.insert_rows(1)
ws.merge_cells("A1:L1")
ws["A1"] = (
"1. 纯氢里程(km)、纯电里程(km)来源于广州平台统计,TBOX里程来源于车机上报的仪表盘里程统计,"
"GPS里程优先按两份7月GPS日明细汇总;测试车辆未命中时,回填7月最终考核源表中已确认的GPS实际里程\n"
"2. 广州平台总里程(km)=纯氢里程(km)+纯电里程(km)\n"
"3. 已确认广州平台暂未接入86辆普货车数据;无对应记录统一标记为“无来源”"
)
header_fill = PatternFill("solid", fgColor="1F4E78")
platform_fill = PatternFill("solid", fgColor="DDEBF7")
tbox_fill = PatternFill("solid", fgColor="E2F0D9")
gps_fill = PatternFill("solid", fgColor="E4DFEC")
test_fill = PatternFill("solid", fgColor="FFF2CC")
missing_fill = PatternFill("solid", fgColor="F2F2F2")
thin = Side(style="thin", color="B7B7B7")
for cell in ws[2]:
cell.fill = header_fill
cell.font = Font(color="FFFFFF", bold=True)
cell.alignment = Alignment(horizontal="center", vertical="center", wrap_text=True)
cell.border = Border(bottom=Side(style="medium", color="FFFFFF"))
for row in ws.iter_rows(min_row=3, max_row=ws.max_row):
is_test = row[6].value == ""
for cell in row:
cell.border = Border(left=thin, right=thin, top=thin, bottom=thin)
cell.alignment = Alignment(horizontal="center", vertical="center")
row[2].alignment = Alignment(horizontal="left", vertical="center")
if is_test:
for cell in row:
cell.fill = test_fill
else:
for cell in row[7:10]:
cell.fill = platform_fill
row[10].fill = tbox_fill
row[11].fill = gps_fill
for cell in row[7:12]:
if cell.value == "无来源":
cell.fill = missing_fill
cell.font = Font(color="7F7F7F", italic=True)
for row in range(3, ws.max_row + 1):
ws.cell(row, 4).number_format = "yyyy-mm-dd"
ws.cell(row, 5).number_format = "yyyy-mm-dd"
for col in range(8, 13):
ws.cell(row, col).number_format = "#,##0.0"
widths = [8, 14, 34, 15, 15, 11, 14, 17, 17, 23, 17, 17]
for col, width in enumerate(widths, 1):
ws.column_dimensions[get_column_letter(col)].width = width
ws["A1"].fill = PatternFill("solid", fgColor="FFF2CC")
ws["A1"].font = Font(color="7F6000", italic=True)
ws["A1"].alignment = Alignment(horizontal="left", vertical="center", wrap_text=True)
ws.row_dimensions[1].height = 58
ws.row_dimensions[2].height = 36
ws.freeze_panes = "A3"
ws.auto_filter.ref = f"A2:L{ws.max_row}"
ws.sheet_view.showGridLines = False
ws.sheet_properties.pageSetUpPr.fitToPage = True
ws.page_setup.orientation = "landscape"
ws.page_setup.fitToWidth = 1
ws.page_setup.fitToHeight = 0
ws.print_title_rows = "1:2"
OUTPUT.parent.mkdir(parents=True, exist_ok=True)
wb.save(OUTPUT)
print({
"output": str(OUTPUT),
"records": len(records),
"test_records": sum(r["test"] == "" for r in records),
"platform_missing": sum(r["hydrogen"] is None for r in records),
"tbox_missing": sum(r["tbox"] is None for r in records),
"gps_missing": sum(r["gps"] is None for r in records),
})