1017 Commits

Author SHA1 Message Date
lingniu
1e57fdfff1 feat: unify account directory metrics with semi ui 2026-07-19 12:10:36 +08:00
lingniu
7df015b0f0 feat: unify source diagnostics with semi ui 2026-07-19 12:07:12 +08:00
lingniu
031e22de79 feat: unify mileage metrics with semi ui 2026-07-19 12:02:10 +08:00
lingniu
2e87378c92 feat: unify history metrics with semi ui 2026-07-19 11:52:24 +08:00
lingniu
343e7c9b89 fix: hide expired session token errors on login 2026-07-19 11:27:34 +08:00
lingniu
61ca72fab2 feat: make account password login explicit 2026-07-19 11:25:57 +08:00
lingniu
a82bb28af9 feat: unify track replay summary 2026-07-19 11:18:16 +08:00
lingniu
1be76c1dfa feat: unify vehicle directory summary 2026-07-19 11:04:34 +08:00
lingniu
e02d8dba54 feat: simplify login and unify monitor summary 2026-07-19 10:48:08 +08:00
lingniu
dcc6b223d9 feat: unify monitor vehicle metric rail 2026-07-19 10:28:17 +08:00
lingniu
add47b7efc feat: unify vehicle live metric rail 2026-07-19 10:10:28 +08:00
lingniu
a6f6ce260f feat: unify operations health rail 2026-07-19 09:41:36 +08:00
lingniu
39c828ed32 feat: unify access decision rail 2026-07-19 09:25:06 +08:00
lingniu
8372ca5349 feat: unify alert operations queue 2026-07-19 09:05:32 +08:00
lingniu
26ccd2434a feat: focus operations review queue 2026-07-19 08:48:32 +08:00
lingniu
c3881bf25c feat: unify workspace dialogs 2026-07-19 08:32:48 +08:00
lingniu
49ae75937e feat: unify monitor action workbenches 2026-07-19 08:11:34 +08:00
lingniu
af71468e63 feat: unify mobile track workspace 2026-07-19 07:58:51 +08:00
lingniu
6795a8ea6d feat: unify export task workspace 2026-07-19 07:49:13 +08:00
lingniu
3e80da1b73 feat: unify filter workbenches 2026-07-19 07:36:08 +08:00
lingniu
af608b35e0 feat: unify global action workbenches 2026-07-19 07:13:09 +08:00
lingniu
3b5a66b687 feat: unify governance evidence workbenches 2026-07-19 06:51:23 +08:00
lingniu
f2fe5404d9 feat: unify track and telemetry workbenches 2026-07-19 06:29:55 +08:00
lingniu
4bd1ad8a85 feat: unify account permission workbenches 2026-07-19 06:07:26 +08:00
lingniu
860978fad8 feat: unify task and rule workbenches 2026-07-19 05:52:52 +08:00
lingniu
63b50732cb feat: unify detail side sheets 2026-07-19 05:38:17 +08:00
lingniu
7eca5ae352 feat: unify configuration side sheets 2026-07-19 05:16:55 +08:00
lingniu
2bae87f6a2 fix: keep mobile filter resets intentional 2026-07-19 04:57:31 +08:00
lingniu
9a87d14d6d feat: unify mobile filter sheets 2026-07-19 04:44:21 +08:00
lingniu
d3eb77e2c7 feat: refine global shell for mobile landscape 2026-07-19 04:29:58 +08:00
lingniu
712721bc73 feat: refine short landscape access workspace 2026-07-19 04:08:39 +08:00
lingniu
8168acc107 feat: refine short landscape operations 2026-07-19 03:47:55 +08:00
lingniu
2372d63823 feat: refine short landscape account governance 2026-07-19 03:26:48 +08:00
lingniu
bbacf3b913 feat: refine responsive vehicle record 2026-07-19 03:08:27 +08:00
lingniu
c695d6e431 feat: refine mobile vehicle directory 2026-07-19 02:37:40 +08:00
lingniu
f4f8bf9f4b feat: densify mobile monitor workspace 2026-07-19 02:15:10 +08:00
lingniu
dcfd78f73b feat: refine mobile track workspace 2026-07-19 01:50:14 +08:00
lingniu
3770afabec feat: clarify mobile mileage legend 2026-07-19 01:33:28 +08:00
lingniu
e7b49df8ed feat: guide history query start 2026-07-19 01:18:42 +08:00
lingniu
82f81eee33 feat: focus mobile alert triage 2026-07-19 01:02:39 +08:00
lingniu
862671c083 feat: refine mobile vehicle detail actions 2026-07-19 00:43:37 +08:00
lingniu
23b632a2e0 feat: unify mobile vehicle search 2026-07-19 00:17:47 +08:00
lingniu
c115b28bfe feat: unify mobile account filters 2026-07-18 23:59:11 +08:00
lingniu
045482c161 feat: unify mobile vehicle diagnostics 2026-07-18 23:41:26 +08:00
lingniu
c8814f471b feat: unify mobile history filters 2026-07-18 23:24:22 +08:00
lingniu
c358faf17d feat: optimize mobile access filters 2026-07-18 23:07:12 +08:00
lingniu
a7297abdaf refine Semi UI alert workflow 2026-07-18 22:58:07 +08:00
lingniu
2824c08eb6 refine Semi UI mileage filters 2026-07-18 22:41:46 +08:00
lingniu
b465d1ff7a refine Semi UI telemetry details 2026-07-18 22:28:00 +08:00
lingniu
76d35f54c5 refine mobile vehicle source evidence 2026-07-18 22:15:08 +08:00
lingniu
fd5e16c36c unify Semi UI track evidence 2026-07-18 22:05:21 +08:00
lingniu
e40eabac92 refine Semi UI monitor inspector 2026-07-18 21:53:35 +08:00
lingniu
911addd891 refine Semi UI vehicle flow 2026-07-18 21:33:32 +08:00
lingniu
22f1548f8a unify Semi UI protocol evidence 2026-07-18 21:10:45 +08:00
lingniu
14e87f143d refine Semi UI history evidence workspace 2026-07-18 20:51:50 +08:00
lingniu
9101d26005 refine Semi UI track replay workspace 2026-07-18 20:29:32 +08:00
lingniu
b5b235e394 refine Semi UI vehicle workspace 2026-07-18 20:04:46 +08:00
lingniu
e95fbe5c19 refine Semi UI account access readiness 2026-07-18 19:41:01 +08:00
lingniu
532524b992 refine Semi UI operations health hierarchy 2026-07-18 19:25:40 +08:00
lingniu
5d4aa92cc1 refine Semi UI access governance 2026-07-18 19:12:07 +08:00
lingniu
84684f2dc1 polish Semi UI alert subworkspaces 2026-07-18 18:56:01 +08:00
lingniu
db1079ae27 refine Semi UI alert disposition flow 2026-07-18 18:38:04 +08:00
lingniu
23669d8190 refine Semi UI track replay summary 2026-07-18 18:21:25 +08:00
lingniu
5bfdc7f637 refine Semi UI vehicle evidence 2026-07-18 18:07:15 +08:00
lingniu
801a9851ea refine Semi UI access evidence 2026-07-18 17:37:29 +08:00
lingniu
82d78926c6 refine Semi UI alert disposition 2026-07-18 17:27:08 +08:00
lingniu
29b9ce9594 refine Semi UI identity permissions 2026-07-18 17:20:25 +08:00
lingniu
dbd1d9d6ca refine Semi UI mobile vehicle grants 2026-07-18 17:09:58 +08:00
lingniu
f7b523453c refine Semi UI access evidence inspector 2026-07-18 16:50:20 +08:00
lingniu
8827a05e7e refine Semi UI operations diagnosis cockpit 2026-07-18 16:33:21 +08:00
lingniu
cedb4f66b4 refine Semi UI mobile alert rule editing 2026-07-18 16:19:40 +08:00
lingniu
f28dfebaae refine Semi UI alert mobile sheets 2026-07-18 16:00:18 +08:00
lingniu
f0c17eac4d refine shared Semi UI command tokens 2026-07-18 15:44:17 +08:00
lingniu
8a2edc8bc9 refine Semi UI access workspace 2026-07-18 15:27:47 +08:00
lingniu
b764ba1be8 refine Semi UI mobile history tools 2026-07-18 15:15:37 +08:00
lingniu
50b5107bc2 refine Semi UI mobile monitor 2026-07-18 14:59:55 +08:00
lingniu
fe056019eb refine Semi UI mileage workspace 2026-07-18 14:44:57 +08:00
lingniu
c57caabaa0 refine Semi UI track and vehicle workspaces 2026-07-18 14:34:01 +08:00
lingniu
01d165d771 refine Semi UI alert workspace 2026-07-18 14:12:44 +08:00
lingniu
32c8939eb5 refine Semi UI identity workflow 2026-07-18 13:59:33 +08:00
lingniu
10a072280b refine Semi UI operations health 2026-07-18 13:38:25 +08:00
lingniu
3b06dd86a8 refine Semi UI vehicle directory 2026-07-18 13:29:46 +08:00
lingniu
c7123f6fbd refine Semi UI access difference workspace 2026-07-18 13:14:39 +08:00
lingniu
ad81d108a7 refine Semi UI history evidence workspace 2026-07-18 13:00:33 +08:00
lingniu
8aa273e579 refine Semi UI mileage matrix workspace 2026-07-18 12:41:41 +08:00
lingniu
e5bf76ea15 refine Semi UI track replay workspace 2026-07-18 12:21:42 +08:00
lingniu
a00276816b refine Semi UI account governance 2026-07-18 12:07:56 +08:00
lingniu
f423240e6f polish operations quality workspace 2026-07-18 11:53:19 +08:00
lingniu
09e16e8d26 polish access difference workspace 2026-07-18 11:40:35 +08:00
lingniu
0802b1989a polish alert operations workspace 2026-07-18 11:24:24 +08:00
lingniu
622c4398ec polish single vehicle entry workspace 2026-07-18 11:09:27 +08:00
lingniu
5805635ccf unify Shanghai time across Semi UI workspaces 2026-07-18 10:51:55 +08:00
lingniu
bba6a621aa refine Semi UI history workspace 2026-07-18 10:39:00 +08:00
lingniu
3cd19b276d refine Semi UI account workspace 2026-07-18 10:30:28 +08:00
lingniu
41b5c41998 refine Semi UI access workspace 2026-07-18 10:17:05 +08:00
lingniu
c1c35f8962 refine Semi UI vehicle workspace 2026-07-18 10:00:15 +08:00
lingniu
04d5992252 refine Semi UI operations review flow 2026-07-18 09:35:16 +08:00
lingniu
71f334d259 refine Semi UI alert workspace 2026-07-18 09:19:23 +08:00
lingniu
1688ba1b50 unify Semi UI query metric rails 2026-07-18 09:02:54 +08:00
lingniu
fb7cb4c6d6 refine Semi UI track replay details 2026-07-18 08:49:29 +08:00
lingniu
31dd36ee70 refine Semi UI vehicle discovery on mobile 2026-07-18 08:41:39 +08:00
lingniu
9a49d015cf refine Semi UI account permission workspace 2026-07-18 08:28:23 +08:00
lingniu
aca0edf130 refine Semi UI vehicle detail workspace 2026-07-18 08:19:43 +08:00
lingniu
1adcff5ecb refine Semi UI mobile history workspace 2026-07-18 08:06:08 +08:00
lingniu
fb171fb0de fix mobile track workspace layout 2026-07-18 07:57:53 +08:00
lingniu
40fcb6fc7b refine Semi UI mileage workspace 2026-07-18 07:50:25 +08:00
lingniu
ffb1283807 refine Semi UI operations workspace 2026-07-18 07:32:32 +08:00
lingniu
0c6a3e4ef1 refine Semi UI alert workspace 2026-07-18 07:19:19 +08:00
lingniu
45c0d4d9a3 refine Semi UI map controls 2026-07-18 07:09:05 +08:00
lingniu
b54d5ba077 refine Semi UI telemetry evidence 2026-07-18 07:00:07 +08:00
lingniu
4b3bb3b3dd refine Semi UI vehicle directory 2026-07-18 06:47:39 +08:00
lingniu
4bca0482c1 refine Semi UI track playback dock 2026-07-18 06:29:46 +08:00
lingniu
f7cd3020f7 fix Semi UI account editor viewport 2026-07-18 06:19:32 +08:00
lingniu
24d3c1dcac refine Semi UI access filter workflow 2026-07-18 06:13:18 +08:00
lingniu
6bb2107f5e refine Semi UI history data workflow 2026-07-18 06:05:58 +08:00
lingniu
018ec20256 refine Semi UI responsive application shell 2026-07-18 05:50:04 +08:00
lingniu
32d97a0ddc refine Semi UI monitor vehicle inspector 2026-07-18 05:40:47 +08:00
lingniu
45820daa46 refine Semi UI monitor workspace 2026-07-18 05:31:24 +08:00
lingniu
9fa68e4ed7 refine Semi UI authentication workspace 2026-07-18 05:15:57 +08:00
lingniu
80f6880488 refine Semi UI account workspace 2026-07-18 05:02:31 +08:00
lingniu
800d956159 refine Semi UI operations workspace 2026-07-18 04:43:06 +08:00
lingniu
866dfdaf5f refine Semi UI alert and access workspaces 2026-07-18 04:26:37 +08:00
lingniu
cff2e07a57 refine Semi UI track and history workspaces 2026-07-18 04:02:49 +08:00
lingniu
8291f5ea69 refine Semi UI vehicle evidence hierarchy 2026-07-18 03:52:05 +08:00
lingniu
3cebfe1982 unify Semi UI workspace command hierarchy 2026-07-18 03:36:18 +08:00
lingniu
e39140c408 refine Semi UI account workspace 2026-07-18 03:22:35 +08:00
lingniu
91db15ebde refine Semi UI alert workspace 2026-07-18 03:13:34 +08:00
lingniu
bb3820bfaa refine Semi UI history workspace 2026-07-18 02:57:45 +08:00
lingniu
1a3a7d49ec refine Semi UI track replay workspace 2026-07-18 02:41:25 +08:00
lingniu
c29d41fd86 refine Semi UI vehicle detail workspace 2026-07-18 02:24:21 +08:00
lingniu
5a8eec16f3 refine Semi UI vehicle discovery workspace 2026-07-18 02:10:11 +08:00
lingniu
da9619617d refine Semi UI operations health workspace 2026-07-18 02:02:40 +08:00
lingniu
f3c1391322 polish Semi UI admin and operations workspaces 2026-07-18 01:57:06 +08:00
lingniu
1affd30366 polish Semi UI query workspaces 2026-07-18 01:43:37 +08:00
lingniu
6d63783630 unify Semi UI filter workspaces and pagination 2026-07-18 01:29:30 +08:00
lingniu
5c0a61f7bd refine Semi UI platform shell and states 2026-07-18 01:12:48 +08:00
lingniu
1cd92715a5 feat(web): unify Semi UI workspace states 2026-07-18 00:59:11 +08:00
lingniu
159c80b0ae feat(platform): harden telemetry pipeline and unify Semi UI workspaces 2026-07-18 00:26:36 +08:00
lingniu
65b4e4f055 feat(operations): support canonical source providers 2026-07-16 22:50:55 +08:00
lingniu
9a5e6e0c4f feat(operations): add auditable source provider maintenance 2026-07-16 20:01:08 +08:00
lingniu
c270140006 fix(master-data): enforce authoritative vehicle totals 2026-07-16 19:35:05 +08:00
lingniu
d67f42b4f4 feat(operations): add automated reconciliation center 2026-07-16 19:14:53 +08:00
lingniu
bbab018d55 docs(oneos): publish API integration runbook 2026-07-16 18:49:04 +08:00
lingniu
46f2026c03 feat(oneos): add signed scope API client 2026-07-16 18:49:01 +08:00
lingniu
1243efc7dd docs(monitor): record province aggregation release 2026-07-16 18:36:35 +08:00
lingniu
21d1baada5 feat(monitor): aggregate nationwide fleet by province 2026-07-16 18:36:31 +08:00
lingniu
ea4d576793 docs(access): record real source calibration 2026-07-16 18:10:48 +08:00
lingniu
d682b7ffe1 fix(access): expose master data maintenance queue 2026-07-16 18:10:00 +08:00
lingniu
ff6df370cb fix(access): report only real vehicle sources 2026-07-16 18:07:45 +08:00
lingniu
85de053962 docs(auth): record vehicle grant interval release 2026-07-16 18:03:22 +08:00
lingniu
ddc1bb2d96 feat(auth): manage vehicle grant intervals 2026-07-16 18:01:23 +08:00
lingniu
d6b857240c docs: record source diagnosis release 2026-07-16 17:39:07 +08:00
lingniu
55f8fbe113 fix(operations): stabilize empty policy audit 2026-07-16 17:37:29 +08:00
lingniu
96cad7eef7 feat(operations): add source diagnosis workspace 2026-07-16 17:35:10 +08:00
lingniu
df7d9799b3 docs: record customer demo gate release 2026-07-16 17:21:26 +08:00
lingniu
b2d469835e feat(auth): add customer demo acceptance gate 2026-07-16 17:20:51 +08:00
lingniu
7e9f46db2a docs: record live-first vehicle release 2026-07-16 17:15:28 +08:00
lingniu
acc29c3873 feat(vehicle): prioritize live status 2026-07-16 17:14:34 +08:00
lingniu
3e5c5772df docs: record monitor context release 2026-07-16 17:10:02 +08:00
lingniu
1cfc9c2bdd feat(monitor): preserve investigation context 2026-07-16 17:08:54 +08:00
lingniu
38b056f5f1 feat(history): scope export tasks to owners 2026-07-16 17:00:33 +08:00
lingniu
a2722e4afd feat(history): align fields quality and exports 2026-07-16 16:46:10 +08:00
lingniu
17c4591040 feat(platform): expose multi-source vehicle evidence 2026-07-16 16:35:04 +08:00
lingniu
196cfa018f docs: record monitor customer-list release 2026-07-16 16:10:23 +08:00
lingniu
b44dd3c45c fix(monitor): keep fleet totals on bound vehicles 2026-07-16 16:05:59 +08:00
lingniu
c224907fad feat(monitor): retain authorized vehicles without locations 2026-07-16 16:04:23 +08:00
lingniu
4688abadef feat(auth): enforce vehicle grant time boundaries 2026-07-16 15:44:29 +08:00
lingniu
a541d10c7b docs: align platform goal with 0716 meeting 2026-07-16 15:36:09 +08:00
lingniu
cda77555b7 docs: add vehicle platform meeting goal backlog 2026-07-16 15:20:14 +08:00
lingniu
48e2ce9fd4 fix: deduplicate account vehicle permissions 2026-07-16 14:16:22 +08:00
lingniu
a1195fb97d feat: add customer authentication and scoped RBAC 2026-07-16 13:58:28 +08:00
lingniu
6d6c9ce534 refactor(mileage): use fixed-column table on mobile 2026-07-16 13:07:43 +08:00
lingniu
4877c3fe8f feat(mileage): compact mobile results and add date presets 2026-07-16 12:53:32 +08:00
lingniu
c1b4703810 style(mileage): emphasize results across breakpoints 2026-07-16 12:44:53 +08:00
lingniu
069b324565 feat(platform): make mobile workflows task focused 2026-07-16 12:05:01 +08:00
lingniu
6d82a66d68 style(platform): unify workspace typography scale 2026-07-16 11:44:08 +08:00
lingniu
7299f42b0e docs(ops): document location source arbitration 2026-07-16 11:24:08 +08:00
lingniu
8487783564 fix(monitor): hold conflicting terminal coordinates 2026-07-16 11:16:22 +08:00
lingniu
b9fcf476ec fix(monitor): arbitrate conflicting location sources 2026-07-16 11:07:23 +08:00
lingniu
9384d6acf5 perf(monitor): isolate follow viewport events 2026-07-16 10:41:41 +08:00
lingniu
e0a5ba6f5d perf(monitor): throttle selected map centering 2026-07-16 10:35:52 +08:00
lingniu
cf9e2a9803 perf(monitor): synchronize map follow frames 2026-07-16 10:32:31 +08:00
lingniu
77f12be63e perf(monitor): reuse point animation buffers 2026-07-16 10:27:36 +08:00
lingniu
59f220885b feat(vehicle): live-update single vehicle map 2026-07-16 10:20:18 +08:00
lingniu
e0e7e28762 feat(monitor): match motion to report intervals 2026-07-16 09:55:10 +08:00
lingniu
3bbae099bc tune(monitor): extend point motion to five seconds 2026-07-16 09:41:08 +08:00
lingniu
26a0599611 tune(monitor): slow vehicle point transitions 2026-07-16 09:39:11 +08:00
lingniu
4ffaf95436 feat(monitor): animate live vehicle positions 2026-07-16 09:31:33 +08:00
lingniu
69c5ca09ba fix(web): use requested platform title 2026-07-16 09:19:05 +08:00
lingniu
c3b0996698 feat(web): apply Lingniu brand assets 2026-07-16 09:17:04 +08:00
lingniu
61c1bbc365 perf(mileage): stream workbook export rows 2026-07-16 09:08:04 +08:00
lingniu
b78cf1696f perf(api): make batch vehicle search exact 2026-07-16 08:58:32 +08:00
lingniu
251b53125d fix(api): bound mysql page sizes 2026-07-16 08:49:10 +08:00
lingniu
99d9dcca1b fix(mileage): bound daily date ranges 2026-07-16 08:44:00 +08:00
lingniu
91b5d76657 perf(monitor): bound map refresh fingerprints 2026-07-16 08:38:41 +08:00
lingniu
4c54cfa6ff perf(web): pause hidden live polling 2026-07-16 08:35:37 +08:00
lingniu
e06778e38c perf(web): reuse locale formatters 2026-07-16 08:30:23 +08:00
lingniu
22d069a404 fix(web): stabilize client file downloads 2026-07-16 08:23:45 +08:00
lingniu
4b2e743229 fix(monitor): harden mobile entry dialog 2026-07-16 08:20:15 +08:00
lingniu
473c9e1251 fix(history): respect export permission boundary 2026-07-16 08:13:36 +08:00
lingniu
6049d8c22b fix(access): hide unresolved identity queue from viewers 2026-07-16 08:10:05 +08:00
lingniu
7346cd0891 fix(access): respect read-only permission boundary 2026-07-16 08:08:06 +08:00
lingniu
7cf856fd51 fix(history): activate column visibility settings 2026-07-16 08:03:59 +08:00
lingniu
1a9b2ce962 fix(web): bound stalled mutations 2026-07-16 07:54:04 +08:00
lingniu
be1ff4ff0d fix(web): expose secondary query failures 2026-07-16 07:45:15 +08:00
lingniu
e9b7c0d31f fix(web): recover vehicle option lookups 2026-07-16 07:38:39 +08:00
lingniu
25e8427882 fix(web): activate contextual help and filter status 2026-07-16 07:33:55 +08:00
lingniu
d1764b3d66 fix(web): bound stalled route queries 2026-07-16 07:18:49 +08:00
lingniu
a0448d6615 fix(web): retry transient map load failures 2026-07-16 07:13:23 +08:00
lingniu
970ef5d7c0 perf(web): reschedule route preloads after navigation 2026-07-16 07:09:55 +08:00
lingniu
fa886abff9 fix(web): retry rejected route modules 2026-07-16 07:01:34 +08:00
lingniu
ec3071d59b fix(ops): recover stale history error health 2026-07-16 06:48:47 +08:00
lingniu
e99087f329 fix(tracks): sync query state with navigation 2026-07-16 06:41:06 +08:00
lingniu
f1f0bab590 perf(tracks): defer empty map runtime 2026-07-16 06:35:34 +08:00
lingniu
e6d26faad9 perf(tracks): coalesce address lookups 2026-07-16 06:22:37 +08:00
lingniu
41e2b1ab97 fix(auth): dedupe concurrent session expiry 2026-07-16 06:19:02 +08:00
lingniu
ee96b8803c perf(history): reduce export polling load 2026-07-16 06:14:32 +08:00
lingniu
97fe1704a2 fix(web): recover pre-render boot failures 2026-07-16 06:04:30 +08:00
lingniu
4302fc8d45 perf(monitor): throttle selected address lookups 2026-07-16 05:52:29 +08:00
lingniu
3f7619c7cd perf(monitor): defer QR generator 2026-07-16 05:41:05 +08:00
lingniu
9c6cd09413 perf(monitor): stop hidden selection polling 2026-07-16 05:34:19 +08:00
lingniu
d669531128 fix(web): bound lazy route recovery 2026-07-16 05:29:08 +08:00
lingniu
b21f9a60cd perf(monitor): skip metadata-only map work 2026-07-16 05:16:51 +08:00
lingniu
379f1ce941 fix(web): recover route errors on scope change 2026-07-16 05:08:40 +08:00
lingniu
c4f6587120 perf(web): release inactive mutation state 2026-07-16 05:04:17 +08:00
lingniu
6f0c15fe6d fix(statistics): isolate mileage query scopes 2026-07-16 04:59:13 +08:00
lingniu
25f3fa9c64 fix(monitor): isolate batch search scope 2026-07-16 04:52:35 +08:00
lingniu
14525887ea fix(web): isolate paginated filter scopes 2026-07-16 04:44:21 +08:00
lingniu
d95d7ab735 fix(history): isolate query scope transitions 2026-07-16 04:36:41 +08:00
lingniu
6b4a0726b2 fix(auth): isolate client cache by session 2026-07-16 04:26:20 +08:00
lingniu
b39917d9d6 docs(web): record bounded release rollout 2026-07-16 04:20:04 +08:00
lingniu
326d03cd6d ops(web): bound immutable release history 2026-07-16 04:18:15 +08:00
lingniu
8e8f3f9f5c perf(tracks): bound playback rendering work 2026-07-16 04:06:08 +08:00
lingniu
982acb7730 docs(web): record mileage DOM reduction 2026-07-16 03:49:28 +08:00
lingniu
5d3f5f1e59 perf(statistics): mount one responsive mileage matrix 2026-07-16 03:47:00 +08:00
lingniu
2f26fd25b9 perf(web): bound speculative route warming 2026-07-16 03:39:47 +08:00
lingniu
cad39d6326 perf(monitor): aggregate realtime workspace 2026-07-16 03:26:14 +08:00
lingniu
01150844df security(web): sanitize runtime map config 2026-07-16 03:15:25 +08:00
lingniu
4be4aaf265 perf(web): gate alert workspace queries 2026-07-16 03:06:16 +08:00
lingniu
d0e82cef87 perf(web): isolate Excel export worker 2026-07-16 02:58:18 +08:00
lingniu
02868344ca perf(web): contain offscreen vehicle rows 2026-07-16 02:51:12 +08:00
lingniu
7396544269 perf(web): warm every primary route 2026-07-16 02:35:10 +08:00
lingniu
d140a9782b perf(web): reconcile fleet map incrementally 2026-07-16 02:25:04 +08:00
lingniu
d995024e06 perf(web): release inactive address queries 2026-07-16 02:14:40 +08:00
lingniu
85e92e62da perf(api): bound reverse geocode requests 2026-07-16 02:10:10 +08:00
lingniu
1e80278fee feat(web): add visible batch vehicle search 2026-07-16 02:02:56 +08:00
lingniu
c493df37f1 docs: record portable release smoke gate 2026-07-16 01:56:07 +08:00
lingniu
986375d903 test(ops): support legacy Python HTTP server 2026-07-16 01:55:21 +08:00
lingniu
f38b630592 docs: record complete release asset gate 2026-07-16 01:54:26 +08:00
lingniu
3051caf327 ops: verify every release asset 2026-07-16 01:51:31 +08:00
lingniu
c86c9d1e31 docs: record production mileage export smoke 2026-07-16 01:47:39 +08:00
lingniu
c834a0406c perf(web): make mileage exports cancellable 2026-07-16 01:43:57 +08:00
lingniu
639564b96d perf(web): release track map listeners explicitly 2026-07-16 01:32:55 +08:00
lingniu
8ebffa4fee perf(web): warm frequent routes during idle time 2026-07-16 01:19:29 +08:00
lingniu
b93e165590 fix(web): recover transient map SDK failures 2026-07-16 01:09:56 +08:00
lingniu
e80e1dcea9 perf(web): release monitor resources on unmount 2026-07-16 00:59:57 +08:00
lingniu
c00f50551a perf(web): remove legacy theme from production bundle 2026-07-16 00:47:52 +08:00
lingniu
6506c0f1f0 fix(web): preserve route state and cancel stale reads 2026-07-16 00:29:47 +08:00
lingniu
53e1b57e86 fix(platform): harden queries and batch vehicle search 2026-07-16 00:14:16 +08:00
lingniu
c29ccdf2da fix(web): harden route transitions and monitor memory 2026-07-15 23:54:21 +08:00
lingniu
3fabcf181a feat(platform): consolidate production vehicle data workflows 2026-07-15 23:26:29 +08:00
lingniu
6cddc0a43d feat: add vehicle mileage statistics 2026-07-14 13:18:52 +08:00
lingniu
bb59303a4b feat: build vehicle data platform and production pipeline 2026-07-14 12:35:33 +08:00
lingniu
b452be3b94 fix(stats): backfill mqtt mileage from mileage frames 2026-07-08 18:39:20 +08:00
lingniu
469e9c75f6 fix(stats): calculate mileage without previous-day baseline 2026-07-08 17:26:34 +08:00
lingniu
536b0bfee6 refactor(stats): slim final daily mileage table 2026-07-08 16:58:25 +08:00
lingniu
da669fae94 fix(stats): track data sources before mileage sampling 2026-07-08 16:46:06 +08:00
lingniu
44e119331d fix(stats): preserve realtime baseline from current candidate 2026-07-08 16:19:50 +08:00
lingniu
bd67034927 fix(stats): guard backfill source-less mileage 2026-07-08 15:50:38 +08:00
lingniu
17937a1ad8 fix(stats): skip realtime mileage without source identity 2026-07-08 15:44:31 +08:00
lingniu
e0ae57828e fix: tighten source-aware mileage projection 2026-07-08 15:41:59 +08:00
lingniu
800ca1a8b1 feat(stats): expose selected mileage source 2026-07-08 15:19:43 +08:00
lingniu
6d633aa292 fix(stats): clear stale backfill candidates before projection 2026-07-08 15:12:27 +08:00
lingniu
a5eebfb32d fix(stats): clear stale backfill daily mileage rows 2026-07-08 15:06:25 +08:00
lingniu
804c238bff feat(stats): backfill mileage source candidates 2026-07-08 14:58:19 +08:00
lingniu
762c7265d7 feat(stats): write realtime mileage candidates 2026-07-08 14:49:01 +08:00
lingniu
67aa521dd9 fix(stats): guard daily mileage source selection marking 2026-07-08 14:45:46 +08:00
lingniu
bc2071ef42 fix(stats): enforce latest source endpoint only for projected daily mileage 2026-07-08 14:41:53 +08:00
lingniu
5f79df3fe2 fix(stats): project trusted mileage source endpoint from metadata 2026-07-08 14:39:40 +08:00
lingniu
99e81a289e docs: add task 3 report 2026-07-08 14:35:20 +08:00
lingniu
aa12317b54 feat(stats): project elected mileage source 2026-07-08 14:34:38 +08:00
lingniu
b7f6e47ce3 Fix candidate mileage review findings 2026-07-08 14:29:54 +08:00
lingniu
08e2d8c0f0 feat(stats): store source mileage candidates 2026-07-08 14:25:09 +08:00
lingniu
009e50a902 feat(stats): add vehicle data source metadata 2026-07-08 14:18:46 +08:00
lingniu
fe770132c7 docs(stats): plan multi-source mileage implementation 2026-07-08 14:12:40 +08:00
lingniu
c08c7308d3 docs(stats): design multi-source mileage statistics 2026-07-08 14:06:13 +08:00
lingniu
a60a628d25 fix(stats): elect trusted mileage source 2026-07-08 13:31:56 +08:00
lingniu
abfed27846 feat(stats): add daily last mileage backfill 2026-07-08 12:01:18 +08:00
lingniu
fdcac05559 fix(stats): map protocol mileage fields 2026-07-08 11:23:06 +08:00
lingniu
4e356876b1 fix(gateway): keep canonical stats fields in field events 2026-07-08 11:17:32 +08:00
lingniu
887c92cd70 feat(platform): focus vehicle service customer copy 2026-07-06 07:01:23 +08:00
lingniu
5f2b3672cb feat(platform): polish customer report delivery copy 2026-07-06 06:39:17 +08:00
lingniu
6bc3bf14de feat(platform): refine customer map views 2026-07-06 06:35:07 +08:00
lingniu
a46072a5e7 feat(platform): add customer alert action queue 2026-07-06 06:29:33 +08:00
lingniu
536d37ca9b feat(platform): align customer mileage service wording 2026-07-06 06:19:40 +08:00
lingniu
e847e1c7e8 feat(platform): add customer fleet command center 2026-07-06 06:10:06 +08:00
lingniu
00d416964b feat(platform): add alert closure workbench 2026-07-06 06:05:01 +08:00
lingniu
55a65d2bc6 feat(platform): add customer history export wizard 2026-07-06 05:58:05 +08:00
lingniu
8b7614d2d6 feat(platform): add customer daily operations home 2026-07-06 05:52:22 +08:00
lingniu
ebaf67a134 feat(platform): add mileage report delivery journey 2026-07-06 05:46:12 +08:00
lingniu
fe005a8296 feat(platform): add fleet service journey strip 2026-07-06 05:39:57 +08:00
lingniu
0cb1763326 feat(platform): add customer time window job command 2026-07-06 05:34:33 +08:00
lingniu
44b3839e83 feat(platform): add customer data delivery cover 2026-07-06 05:27:26 +08:00
lingniu
81014bf7ce feat(platform): add customer alert command center 2026-07-06 05:21:06 +08:00
lingniu
ce3a6225ec feat(platform): add customer mileage reconciliation command 2026-07-06 05:14:11 +08:00
lingniu
378706cc5c feat(platform): add customer map command center 2026-07-06 05:09:50 +08:00
lingniu
97c2db01f8 feat(platform): frame history replay as customer delivery 2026-07-06 05:04:06 +08:00
lingniu
e314f8330f feat(platform): frame mileage online impact for customers 2026-07-06 04:58:08 +08:00
lingniu
286840c19c feat(platform): align vehicle detail with service delivery 2026-07-06 04:53:34 +08:00
lingniu
7ac1390394 feat(platform): align dashboard copy with vehicle service delivery 2026-07-06 04:49:42 +08:00
lingniu
b9b5409acb feat(platform): align realtime copy with vehicle monitoring 2026-07-06 04:44:31 +08:00
lingniu
77c5730e77 feat(platform): frame navigation around vehicle monitoring 2026-07-06 04:40:17 +08:00
lingniu
aefd34bc59 feat(platform): highlight customer service modules 2026-07-06 04:35:27 +08:00
lingniu
2aa7b394a6 feat(platform): expose customer service scope actions 2026-07-06 04:30:47 +08:00
lingniu
f068fc772d feat(platform): surface customer next best actions 2026-07-06 04:24:18 +08:00
lingniu
6246887b86 feat(platform): collapse advanced dashboard workbench 2026-07-06 04:19:48 +08:00
lingniu
ad644d43b1 feat(platform): prioritize customer vehicle service path 2026-07-06 04:15:21 +08:00
lingniu
6b9f269b89 feat(platform): add dedicated time monitor workspace 2026-07-06 04:09:35 +08:00
lingniu
4ce466895a feat(platform): add history export acceptance checklist 2026-07-06 03:58:35 +08:00
lingniu
6f97181ec5 feat(platform): add mileage acceptance checklist 2026-07-06 03:51:30 +08:00
lingniu
ab8ae961e6 feat(platform): add map layer control panel 2026-07-06 03:46:29 +08:00
lingniu
68fed933d9 feat(platform): add time window next action strip 2026-07-06 03:40:22 +08:00
lingniu
3e9128b441 feat(platform): add notification orchestration strip 2026-07-06 03:32:00 +08:00
lingniu
08f2b62362 feat(platform): add history export next actions 2026-07-06 03:26:39 +08:00
lingniu
6a15503c7b feat(platform): add live map decision strip 2026-07-06 03:19:40 +08:00
lingniu
89214a6b7f feat(platform): add mileage definition delivery strip 2026-07-06 03:14:39 +08:00
lingniu
311a29a1dc feat(platform): add map status filter strip 2026-07-06 03:05:56 +08:00
lingniu
53a6aff662 feat(platform): add customer vehicle service hub 2026-07-06 03:01:08 +08:00
lingniu
e69d8ecd10 feat(platform): add history delivery conclusion strip 2026-07-06 02:56:08 +08:00
lingniu
22f945e0aa feat(platform): add realtime service overview 2026-07-06 02:50:14 +08:00
lingniu
268c350e63 feat(platform): add mileage conclusion strip 2026-07-06 02:44:05 +08:00
lingniu
a0158e1ec8 feat(platform): add trip review summary 2026-07-06 02:37:12 +08:00
lingniu
13fb0a1a49 feat(platform): center map page on vehicle service 2026-07-06 02:29:02 +08:00
lingniu
20b8fa425b feat(platform): add vehicle command center 2026-07-06 02:19:39 +08:00
lingniu
1d51c92d8c feat(platform): frame history protocol as source evidence 2026-07-06 02:11:44 +08:00
lingniu
26eeba713d feat(platform): rename realtime channel columns to evidence 2026-07-06 01:54:53 +08:00
lingniu
06904748cc feat(platform): frame realtime sources as evidence 2026-07-06 01:49:55 +08:00
lingniu
442afb8aa5 feat(platform): add single vehicle time window delivery 2026-07-06 01:38:00 +08:00
lingniu
af95f69388 feat(platform): sync dashboard customer time window 2026-07-06 01:27:31 +08:00
lingniu
d89f7f7168 feat(platform): fold dashboard explanatory playbooks 2026-07-06 01:18:27 +08:00
lingniu
221a67e4cb feat(platform): make dashboard map first 2026-07-06 01:13:31 +08:00
lingniu
6da5f77267 feat(platform): add mobile vehicle service dock 2026-07-06 01:06:59 +08:00
lingniu
cd5336f353 fix(platform): prioritize mobile vehicle service content 2026-07-06 00:55:24 +08:00
lingniu
af4d721eeb feat(platform): add customer vehicle service dashboard 2026-07-06 00:48:20 +08:00
lingniu
b8199dcb5a feat(platform): add map vehicle handoff card 2026-07-06 00:41:55 +08:00
lingniu
c00cb0218a feat(platform): add ten second vehicle triage 2026-07-06 00:34:07 +08:00
lingniu
2299588df6 feat(platform): add customer saved view library 2026-07-06 00:28:46 +08:00
lingniu
cca574d5eb feat(platform): separate internal evidence navigation 2026-07-06 00:18:08 +08:00
lingniu
79207d5096 feat(platform): add customer service blueprint 2026-07-06 00:12:09 +08:00
lingniu
0c33648b26 feat(platform): add export template library 2026-07-06 00:01:00 +08:00
lingniu
6d18e5a017 feat(platform): add map shift console 2026-07-05 23:55:27 +08:00
lingniu
7a02bf2c87 feat(platform): add history delivery readiness 2026-07-05 23:49:23 +08:00
lingniu
fbd9982427 feat(platform): add vehicle pool quick views 2026-07-05 23:41:32 +08:00
lingniu
07336e08bc feat(platform): expose amap service capabilities 2026-07-05 23:35:13 +08:00
lingniu
2e6d62c493 feat(platform): add vehicle health energy board 2026-07-05 23:27:11 +08:00
lingniu
4294ba424f feat(platform): add customer service sla board 2026-07-05 23:21:18 +08:00
lingniu
122ba079fb feat(platform): add amap vehicle service foundation 2026-07-05 23:15:57 +08:00
lingniu
4eda7b7e1b feat(platform): add alert action conclusion 2026-07-05 23:09:33 +08:00
lingniu
15a318112d feat(platform): add vehicle service cockpit 2026-07-05 23:03:14 +08:00
lingniu
934691e956 feat(platform): add mileage reconciliation conclusion 2026-07-05 22:55:15 +08:00
lingniu
c2e1341d99 feat(platform): add customer evidence package overview 2026-07-05 22:46:17 +08:00
lingniu
b516a532b7 feat(platform): add realtime time window task board 2026-07-05 22:38:35 +08:00
lingniu
3e834d2eb2 feat(platform): add map view modes 2026-07-05 22:27:40 +08:00
lingniu
aa286b810e feat(platform): add vehicle service delivery strip 2026-07-05 22:19:20 +08:00
lingniu
fc86119857 feat(platform): separate customer service flow 2026-07-05 22:12:25 +08:00
lingniu
eccede509f feat(platform): add delivery time presets 2026-07-05 22:02:29 +08:00
lingniu
7580fce07e feat(platform): add customer export workbench 2026-07-05 21:57:52 +08:00
lingniu
a68db5e0eb feat(platform): add dispatch operations home 2026-07-05 21:51:44 +08:00
lingniu
52a73cba2d feat(platform): add time window service path 2026-07-05 21:44:10 +08:00
lingniu
a26fd71bfa feat(platform): add fleet command center 2026-07-05 21:34:45 +08:00
lingniu
b892ffce81 feat(platform): add alert closure timeline 2026-07-05 21:29:38 +08:00
lingniu
059e4586a7 feat(platform): add realtime priority queue 2026-07-05 21:22:04 +08:00
lingniu
82faa046b7 feat(platform): add customer history evidence chain 2026-07-05 21:14:19 +08:00
lingniu
3efeeb07ac feat(platform): frame sources as evidence layer 2026-07-05 21:07:38 +08:00
lingniu
168493f76f feat(platform): surface map situation on dashboard 2026-07-05 21:03:30 +08:00
lingniu
dd5ae740e5 feat(platform): add customer vehicle service journey 2026-07-05 20:59:37 +08:00
lingniu
68c1c7cef4 feat(platform): polish customer time window controls 2026-07-05 20:54:04 +08:00
lingniu
2b246f414e feat(platform): frame notifications as customer closure 2026-07-05 20:45:22 +08:00
lingniu
1e4a11d806 feat(platform): separate ops quality as evidence layer 2026-07-05 20:40:29 +08:00
lingniu
59d55e7bd6 feat(platform): add customer fleet group views 2026-07-05 20:33:58 +08:00
lingniu
7348cc6186 feat(platform): anchor dashboard on amap vehicle service 2026-07-05 20:28:29 +08:00
lingniu
36ac5fe7c5 feat(platform): add customer report purpose navigation 2026-07-05 20:21:51 +08:00
lingniu
d10291a003 feat(platform): promote customer time window monitoring 2026-07-05 20:16:39 +08:00
lingniu
0beca74bcf feat(platform): center vehicle service around customers 2026-07-05 20:12:39 +08:00
lingniu
2765a3c491 feat(platform): add customer alert service desk 2026-07-05 20:07:19 +08:00
lingniu
dbdce8f312 feat(platform): add vehicle history service desk 2026-07-05 19:57:49 +08:00
lingniu
b73cbf8489 feat(platform): add customer mileage workbench 2026-07-05 19:54:14 +08:00
lingniu
6bfc08a841 feat(platform): add customer realtime command center 2026-07-05 19:49:30 +08:00
lingniu
30bc03a262 feat(platform): prioritize customer vehicle command center 2026-07-05 19:43:20 +08:00
lingniu
e1126a6fcc feat(platform): save reusable customer views 2026-07-05 19:38:30 +08:00
lingniu
2e0b13dfb5 feat(platform): copy customer service scope 2026-07-05 19:33:06 +08:00
lingniu
0c3c6e7b3d feat(platform): add customer time window presets 2026-07-05 19:28:25 +08:00
lingniu
0bfc47bc1c feat(platform): add global customer time window 2026-07-05 19:21:52 +08:00
lingniu
4085268971 feat(platform): preserve customer time window in tasks 2026-07-05 19:16:33 +08:00
lingniu
30c15ec8fd feat(platform): preserve vehicle context in topbar tasks 2026-07-05 19:11:10 +08:00
lingniu
08c275af34 feat(platform): preserve vehicle context in customer tasks 2026-07-05 19:07:50 +08:00
lingniu
e48ea7d22a feat(platform): add customer task dispatch 2026-07-05 19:03:12 +08:00
lingniu
33f1ed24e4 feat(platform): add alert recovery acceptance 2026-07-05 18:59:10 +08:00
lingniu
2852cdd7b6 feat(platform): add map area monitoring workspace 2026-07-05 18:49:05 +08:00
lingniu
ac4b7fcd30 feat(platform): add reusable history report view 2026-07-05 18:42:45 +08:00
lingniu
3e3bc117f7 feat(platform): add realtime vehicle snapshot 2026-07-05 18:32:19 +08:00
lingniu
a04681a65f feat(platform): add mileage reconciliation desk 2026-07-05 18:26:20 +08:00
lingniu
24cabb2122 feat(platform): add map dispatch actions 2026-07-05 18:19:15 +08:00
lingniu
91b60ed63a feat(platform): add single vehicle action desk 2026-07-05 18:13:57 +08:00
lingniu
098e927ada feat(platform): add vehicle priority queue 2026-07-05 18:09:21 +08:00
lingniu
261dbbacdb feat(platform): add customer export center 2026-07-05 18:04:10 +08:00
lingniu
3be803316a feat(platform): add customer primary navigation path 2026-07-05 17:57:37 +08:00
lingniu
680006e36f feat(platform): add customer vehicle monitor desk 2026-07-05 17:53:50 +08:00
lingniu
493834b834 feat(platform): add mileage delivery console 2026-07-05 17:48:18 +08:00
lingniu
67fe8567ef feat(platform): add vehicle asset operations console 2026-07-05 17:42:07 +08:00
lingniu
8dda3c0b85 feat(platform): add fleet operations cockpit 2026-07-05 17:35:30 +08:00
lingniu
2c6fca90d2 feat(platform): add live map service action rail 2026-07-05 17:29:29 +08:00
lingniu
2e20ec5c81 feat(platform): add custom time window monitor 2026-07-05 17:23:12 +08:00
lingniu
afbe29b8de feat(platform): add alert notification channel closure 2026-07-05 17:17:53 +08:00
lingniu
4b84fd8900 feat(platform): add vehicle archive summary 2026-07-05 17:13:02 +08:00
lingniu
8f9180f2ed feat(platform): add customer report cadence 2026-07-05 17:05:06 +08:00
lingniu
8765ac002a feat(platform): add customer delivery cockpit 2026-07-05 16:59:01 +08:00
lingniu
4e829da048 feat(platform): add live map dispatch desk 2026-07-05 16:53:21 +08:00
lingniu
bc1fb71944 feat(platform): add mileage delivery navigation 2026-07-05 16:47:03 +08:00
lingniu
89ab43a2dd feat(platform): add customer problem desk 2026-07-05 16:39:40 +08:00
lingniu
65ac26ef99 feat(platform): add history replay customer navigation 2026-07-05 16:34:16 +08:00
lingniu
3d2d065246 feat(platform): add vehicle asset task rail 2026-07-05 16:26:31 +08:00
lingniu
403a52ce29 feat(platform): add customer task navigation 2026-07-05 16:22:01 +08:00
lingniu
86961d08bb feat(platform): add notification customer questions 2026-07-05 16:15:27 +08:00
lingniu
044fd104cc feat(platform): add vehicle list customer questions 2026-07-05 16:10:03 +08:00
lingniu
06bfe79851 feat(platform): add realtime customer question flow 2026-07-05 16:05:12 +08:00
lingniu
0f6b47e97e feat(platform): add history customer question flow 2026-07-05 16:00:29 +08:00
lingniu
12aef4461a feat(platform): add vehicle detail customer questions 2026-07-05 15:54:32 +08:00
lingniu
50f277738b feat(platform): add customer mileage question flow 2026-07-05 15:50:28 +08:00
lingniu
e27d5513ed feat(platform): add customer question shortcuts 2026-07-05 15:46:32 +08:00
lingniu
416bf51746 feat(platform): add customer alert SLA view 2026-07-05 15:42:45 +08:00
lingniu
1f74ad063c feat(platform): reshape history delivery around customer workflows 2026-07-05 15:35:08 +08:00
lingniu
d74e52de2e feat(platform): align realtime time window review copy 2026-07-05 15:22:14 +08:00
lingniu
b4bcfbb77a feat(platform): sharpen dashboard around customer vehicle workflows 2026-07-05 15:19:10 +08:00
lingniu
dd6c63b72b feat(platform): refine navigation around vehicle statistics 2026-07-05 15:12:51 +08:00
lingniu
d6ae2688a8 feat(platform): align vehicle center with statistics query 2026-07-05 15:10:40 +08:00
lingniu
dab46d2595 feat(platform): align map workflows with statistics query 2026-07-05 15:06:50 +08:00
lingniu
2824982b60 feat(platform): streamline dashboard around vehicle workflows 2026-07-05 15:02:14 +08:00
lingniu
e829f143e7 feat(platform): align history exports with statistics query 2026-07-05 14:58:14 +08:00
lingniu
5c73cea263 feat(platform): align vehicle detail with statistics query 2026-07-05 14:50:54 +08:00
lingniu
597395852c feat(platform): position mileage as statistics query 2026-07-05 14:46:49 +08:00
lingniu
f3c6f1df7c feat(platform): add time window delivery decision 2026-07-05 14:41:29 +08:00
lingniu
96a419f5d4 feat(platform): add vehicle service delivery package 2026-07-05 14:36:04 +08:00
lingniu
ee257b4aa1 feat(platform): add notification reachability checks 2026-07-05 14:31:32 +08:00
lingniu
109ba09934 feat(platform): add history report templates 2026-07-05 14:25:41 +08:00
lingniu
e94ccc37d8 feat(platform): add map field command actions 2026-07-05 14:17:16 +08:00
lingniu
695cecc07b feat(platform): add mileage next actions 2026-07-05 14:12:58 +08:00
lingniu
4c110566f9 feat(platform): add realtime next actions 2026-07-05 14:07:29 +08:00
lingniu
2574a3bfa0 feat(platform): surface customer next actions 2026-07-05 14:00:13 +08:00
lingniu
c7866b3ad2 feat(platform): add trip replay workbench 2026-07-05 13:55:25 +08:00
lingniu
46a1b8c3a8 feat(platform): prioritize fleet monitoring dashboard 2026-07-05 13:42:25 +08:00
lingniu
43d6a0438e feat(platform): add vehicle live service view 2026-07-05 13:34:29 +08:00
lingniu
b3ec3def14 feat(platform): surface selected vehicle map actions 2026-07-05 13:28:07 +08:00
lingniu
2453d3e467 feat(platform): clarify customer data export flow 2026-07-05 13:24:19 +08:00
lingniu
9a16daae00 feat(platform): sharpen customer vehicle cockpit 2026-07-05 13:20:55 +08:00
lingniu
c0cc7fc54f feat(platform): add realtime time-window monitoring 2026-07-05 13:13:20 +08:00
lingniu
3cdded42c5 feat(platform): refine vehicle service workspace 2026-07-05 13:01:34 +08:00
lingniu
9f5a091b87 feat(platform): refine customer service navigation 2026-07-05 12:43:46 +08:00
lingniu
9c6973b4ee feat(platform): refine realtime vehicle monitoring workflow 2026-07-05 12:37:26 +08:00
lingniu
0b7fc31d52 feat(platform): streamline vehicle service dashboard 2026-07-05 12:32:04 +08:00
lingniu
993b1adb00 feat(platform): refine history query customer workflow 2026-07-05 12:26:46 +08:00
lingniu
a7e10f0983 feat(platform): refine mileage customer workflow 2026-07-05 12:17:20 +08:00
lingniu
0e86d45288 feat(platform): refine realtime vehicle monitoring copy 2026-07-05 12:05:43 +08:00
lingniu
cc3e90e0e9 feat(platform): align navigation with vehicle monitoring workflows 2026-07-05 11:59:51 +08:00
lingniu
358bb642f4 feat(platform): position dashboard as vehicle service center 2026-07-05 11:54:23 +08:00
lingniu
856e4f7dfc feat(platform): refine customer alert evidence copy 2026-07-05 11:49:07 +08:00
lingniu
2b423a3643 feat(platform): refine customer mileage evidence copy 2026-07-05 11:37:29 +08:00
lingniu
69496d6ca7 feat(platform): clarify customer history evidence copy 2026-07-05 05:20:13 +08:00
lingniu
488da08d39 feat(platform): reduce protocol wording in customer views 2026-07-05 05:11:44 +08:00
lingniu
4d2e1e3536 feat(platform): separate dashboard evidence layer 2026-07-05 05:05:48 +08:00
lingniu
d1b764e59b feat(platform): refine customer vehicle workbench navigation 2026-07-05 04:57:08 +08:00
lingniu
d762403677 feat(platform): add customer history export board 2026-07-05 04:50:35 +08:00
lingniu
b5f98bb6d8 feat(platform): add customer mileage query board 2026-07-05 04:46:46 +08:00
lingniu
bb6f577141 feat(platform): add customer alert closure board 2026-07-05 04:42:20 +08:00
lingniu
36dbc26ea7 feat(platform): add realtime customer monitoring path 2026-07-05 04:36:57 +08:00
lingniu
f5cfc86d63 feat(platform): sharpen customer vehicle workspace 2026-07-05 04:30:20 +08:00
lingniu
84a1f70284 feat(platform): refine vehicle detail evidence wording 2026-07-05 04:24:48 +08:00
lingniu
1fd4bdb1a2 feat(platform): refine realtime evidence wording 2026-07-05 04:13:44 +08:00
lingniu
6f3907c002 feat(platform): refine mileage evidence wording 2026-07-05 04:10:39 +08:00
lingniu
442e4e9390 feat(platform): refine customer source wording 2026-07-05 04:08:14 +08:00
lingniu
348c51398e feat(platform): add customer alert decision panel 2026-07-05 04:02:43 +08:00
lingniu
06d80f8147 feat(platform): add customer trajectory decision panel 2026-07-05 03:49:18 +08:00
lingniu
4445328581 feat(platform): add customer mileage decision panel 2026-07-05 03:43:56 +08:00
lingniu
8052bf4af4 feat(platform): add customer map decision panel 2026-07-05 03:35:58 +08:00
lingniu
5cb2509457 feat(platform): clarify customer service entry 2026-07-05 03:29:53 +08:00
lingniu
3097fd59eb feat(platform): add customer vehicle lookup path 2026-07-05 03:25:14 +08:00
lingniu
46d2421511 feat(platform): separate customer and ops navigation 2026-07-05 03:19:47 +08:00
lingniu
8b38b1411d feat(platform): remove default history vehicle 2026-07-05 03:17:20 +08:00
lingniu
1de268942c feat(platform): add time window service audit 2026-07-05 03:13:08 +08:00
lingniu
9d41c52cac feat(platform): add customer map monitoring package 2026-07-05 03:08:58 +08:00
lingniu
f4891525a6 feat(platform): add customer data delivery checklist 2026-07-05 03:03:08 +08:00
lingniu
0df1d61eb1 feat(platform): add customer time-window review package 2026-07-05 03:00:03 +08:00
lingniu
6cd95f7ab3 feat(platform): add customer vehicle service package 2026-07-05 02:56:29 +08:00
lingniu
1640edf8cf feat(platform): add mileage delivery package 2026-07-05 02:51:52 +08:00
lingniu
07663026e4 feat(platform): add customer delivery package 2026-07-05 02:48:02 +08:00
lingniu
21828c7b05 feat(platform): reorganize customer navigation 2026-07-05 02:44:34 +08:00
lingniu
7ec2bcf3f9 feat(platform): add map vehicle workbench 2026-07-05 02:42:00 +08:00
lingniu
7bef3c7e33 feat(platform): add vehicle service decision board 2026-07-05 02:36:59 +08:00
lingniu
e8d745949e feat(platform): add time window monitoring workbench 2026-07-05 02:33:17 +08:00
lingniu
f47475a1a9 feat(platform): add alert vehicle task board 2026-07-05 02:30:27 +08:00
lingniu
12caed6310 feat(platform): add history delivery task board 2026-07-05 02:27:00 +08:00
lingniu
751d290f1b feat(platform): add mileage service task board 2026-07-05 02:17:41 +08:00
lingniu
5d9576c5c7 feat(platform): add map monitoring task strip 2026-07-05 02:14:00 +08:00
lingniu
13ef1779ac feat(platform): add vehicle monitoring command center 2026-07-05 02:10:10 +08:00
lingniu
7657412b8c feat(platform): add single vehicle task board 2026-07-05 02:05:44 +08:00
lingniu
2c881dae7c feat(platform): add vehicle service task board 2026-07-05 01:55:54 +08:00
lingniu
7642f9c9eb feat(platform): add alert notification center 2026-07-05 01:45:20 +08:00
lingniu
f45cf90b70 feat(platform): add vehicle service desk 2026-07-05 01:36:46 +08:00
lingniu
f5d38253fe feat(platform): refine history query workbench 2026-07-05 01:29:45 +08:00
lingniu
72a4761bdb feat(platform): add customer mileage review console 2026-07-05 01:24:56 +08:00
lingniu
11710ec6fa feat(platform): customerize realtime map workspace 2026-07-05 01:16:45 +08:00
lingniu
83f033244c feat(platform): add dashboard time window monitor 2026-07-05 01:10:12 +08:00
lingniu
bb18a5075c feat(platform): streamline vehicle operations shell 2026-07-05 01:04:08 +08:00
lingniu
b824289cf6 feat(platform): customerize realtime monitoring view 2026-07-05 00:58:16 +08:00
lingniu
245af2c293 feat(platform): customerize dashboard operations language 2026-07-05 00:50:47 +08:00
lingniu
4acf86b073 feat(platform): refine vehicle detail operations view 2026-07-05 00:47:05 +08:00
lingniu
9a428198b4 feat(platform): customerize history export workflow 2026-07-05 00:39:24 +08:00
lingniu
74ccddb1a7 feat(platform): refine vehicle operations language 2026-07-05 00:31:58 +08:00
lingniu
0613fe9f7b feat(platform): streamline dashboard for vehicle service 2026-07-05 00:24:07 +08:00
lingniu
2ca656a3f3 feat(platform): refocus dashboard on vehicle operations 2026-07-05 00:03:12 +08:00
lingniu
6888f1a598 feat(platform): add source readiness handoff copy 2026-07-04 23:43:07 +08:00
lingniu
1780039722 feat(platform): surface source readiness on dashboard 2026-07-04 23:38:48 +08:00
lingniu
44c7748958 feat(platform): add source readiness operations view 2026-07-04 23:29:31 +08:00
lingniu
bf106a9730 feat(platform): add vehicle data flow workbench 2026-07-04 23:10:42 +08:00
lingniu
ee93d149e2 docs(platform): align vehicle center product goal 2026-07-04 23:03:50 +08:00
lingniu
f65b035be7 feat(platform): add online statistics operations board 2026-07-04 22:59:48 +08:00
lingniu
c0f5ec186e feat(platform): add notification coverage board 2026-07-04 22:49:49 +08:00
lingniu
4aa4e6086b feat(platform): add trajectory impact board 2026-07-04 22:45:55 +08:00
lingniu
a005dba2d4 feat(platform): add realtime operations impact board 2026-07-04 22:39:10 +08:00
lingniu
6803d51b78 feat(platform): add mileage impact scope board 2026-07-04 22:35:05 +08:00
lingniu
5dea611dd9 feat(platform): add unified vehicle service view 2026-07-04 22:28:08 +08:00
lingniu
24897850f3 feat(platform): add alert business impact board 2026-07-04 22:24:06 +08:00
lingniu
b3b4e2203a feat(platform): add vehicle service dossier 2026-07-04 22:17:41 +08:00
lingniu
ea18a058ac feat(platform): add mileage publish audit 2026-07-04 22:10:36 +08:00
lingniu
c27767bc5e feat(platform): add notification execution matrix 2026-07-04 22:03:21 +08:00
lingniu
ae42a4092b feat(platform): add dashboard map operations panel 2026-07-04 21:59:41 +08:00
lingniu
c73c46ca5d feat(platform): add ops capacity action board 2026-07-04 21:54:28 +08:00
lingniu
e6d639572d feat(platform): add vehicle source fusion matrix 2026-07-04 21:48:51 +08:00
lingniu
f0f7bc546e feat(platform): add focus vehicle service card 2026-07-04 21:42:04 +08:00
lingniu
5d57efabea feat(platform): add alert sla escalation board 2026-07-04 21:34:39 +08:00
lingniu
da3a639dd8 feat(platform): add amap reverse geocode for trajectory 2026-07-04 21:25:32 +08:00
lingniu
cd932dc097 feat(platform): add mileage BI publish package 2026-07-04 21:18:13 +08:00
lingniu
0fff9e3890 feat(platform): add alert dispatch board 2026-07-04 21:10:58 +08:00
lingniu
09c86022b2 feat(platform): add realtime source coverage board 2026-07-04 21:07:14 +08:00
lingniu
fa9c088dec feat(platform): add realtime duty handoff 2026-07-04 21:02:59 +08:00
lingniu
4077de9296 feat(platform): add trajectory review handoff 2026-07-04 20:57:28 +08:00
lingniu
df6404e789 feat(platform): add alert notification handoff 2026-07-04 20:54:22 +08:00
lingniu
bfb16c548e feat(platform): add vehicle service runbook 2026-07-04 20:51:32 +08:00
lingniu
431f4a4505 feat(platform): add vehicle scenario navigation 2026-07-04 20:47:29 +08:00
lingniu
e8f0ebff35 feat(platform): add online status handoff 2026-07-04 20:44:07 +08:00
lingniu
4b34a2bf4c feat(platform): add vehicle identity maintenance list 2026-07-04 20:39:58 +08:00
lingniu
87fa0f524c feat(platform): add ops capacity handoff 2026-07-04 20:37:20 +08:00
lingniu
c0b99e9a3a feat(platform): add alert notification drill package 2026-07-04 20:34:15 +08:00
lingniu
e6188e2569 feat(platform): export dashboard snapshot 2026-07-04 20:30:47 +08:00
lingniu
fc57dea551 feat(platform): export notification rules 2026-07-04 20:26:09 +08:00
lingniu
284d2e1741 feat(platform): add vehicle operations summary 2026-07-04 20:23:12 +08:00
lingniu
fb4bcbaa55 feat(platform): add mileage evidence package 2026-07-04 20:20:24 +08:00
lingniu
09e9cc937d feat(platform): add history evidence package 2026-07-04 20:16:58 +08:00
lingniu
bd0b69ddad feat(platform): add realtime issue checklist 2026-07-04 20:09:10 +08:00
lingniu
7a2a8e8a83 feat(platform): add alert escalation checklist 2026-07-04 20:06:35 +08:00
lingniu
06d9dc1562 feat(platform): add vehicle dispatch checklist 2026-07-04 20:03:56 +08:00
lingniu
259601ed7b feat(platform): add operations handoff summary 2026-07-04 19:57:33 +08:00
lingniu
b0451982a1 feat(platform): add alert evidence package 2026-07-04 19:53:28 +08:00
lingniu
9e38112ece feat(platform): add vehicle detail workbench 2026-07-04 19:49:24 +08:00
lingniu
ea76ac22da feat(platform): add vehicle service workbench 2026-07-04 19:44:51 +08:00
lingniu
3ccc280b02 feat(platform): add mileage closure workspace 2026-07-04 19:40:37 +08:00
lingniu
ff04a5c93f feat(platform): flag trajectory playback anomalies 2026-07-04 19:34:58 +08:00
lingniu
0b433c2592 feat(platform): add alert handling command queue 2026-07-04 19:30:55 +08:00
lingniu
41c5e0b338 feat(platform): add realtime operations shortcuts 2026-07-04 19:25:37 +08:00
lingniu
99261997db feat(platform): summarize vehicle service verdicts 2026-07-04 19:17:16 +08:00
lingniu
75d232e773 feat(platform): organize console around vehicle service 2026-07-04 19:13:28 +08:00
lingniu
cbe7882d51 feat(platform): expose structured capacity metrics 2026-07-04 19:10:04 +08:00
lingniu
68b260c704 fix(gateway): guard nats stream retention and writer timeouts 2026-07-04 19:00:18 +08:00
lingniu
504a49a13c feat(platform): align console with vehicle operations center 2026-07-04 18:53:52 +08:00
lingniu
48642726aa feat(platform): center console on vehicle service 2026-07-04 18:40:36 +08:00
lingniu
bdad6f7bf7 feat(platform): surface notification priority queue 2026-07-04 18:17:09 +08:00
lingniu
b8c8ff35fc feat(platform): link vehicle center workflows 2026-07-04 18:13:14 +08:00
lingniu
f840d8be4a feat(platform): link realtime table map selection 2026-07-04 18:07:48 +08:00
lingniu
483d92719b feat(platform): link trajectory playback selection 2026-07-04 18:04:45 +08:00
lingniu
89e954cad9 feat(platform): use alert events api surface 2026-07-04 18:01:45 +08:00
lingniu
1235e76c32 feat(platform): promote alert events surface 2026-07-04 17:58:33 +08:00
lingniu
eb8f15ca18 feat(platform): expose amap server readiness 2026-07-04 17:53:03 +08:00
lingniu
aa384732ef docs(platform): refine vehicle center product goal 2026-07-04 17:49:06 +08:00
lingniu
3b817ef874 feat(platform): add online vehicle statistics detail 2026-07-04 17:47:38 +08:00
lingniu
d266396c47 feat(platform): surface global alert pressure 2026-07-04 17:37:44 +08:00
lingniu
2d4b397acd feat(platform): add parsed fields history view 2026-07-04 17:31:44 +08:00
lingniu
747894ce1d feat(platform): split trajectory replay and history query 2026-07-04 17:23:48 +08:00
lingniu
a65db893d2 feat(platform): show realtime data freshness 2026-07-04 17:14:28 +08:00
lingniu
34a9def4ff feat(platform): add realtime auto refresh controls 2026-07-04 17:07:51 +08:00
lingniu
5ca6b51d0b feat(platform): surface map situation from dashboard 2026-07-04 17:04:32 +08:00
lingniu
255afe8a84 feat(platform): add map situation workspace 2026-07-04 17:01:49 +08:00
lingniu
9ac2b90d12 feat(platform): add online statistics summary 2026-07-04 16:57:11 +08:00
lingniu
d7f1a9279f feat(platform): add vehicle-first statistics overview 2026-07-04 16:44:55 +08:00
lingniu
bdb2ebc723 feat(platform): add trajectory autoplay controls 2026-07-04 16:40:10 +08:00
lingniu
318beccaaf feat(platform): add ops quality workspace 2026-07-04 16:31:51 +08:00
lingniu
c060317fb0 feat(platform): add notification rules workspace 2026-07-04 16:27:37 +08:00
lingniu
33da8ba5a4 feat(platform): advance vehicle operations center 2026-07-04 16:22:05 +08:00
lingniu
775073a95c feat(platform): align shell with vehicle operations center 2026-07-04 16:13:46 +08:00
lingniu
0c7910d6b9 docs(platform): define vehicle service platform blueprint 2026-07-04 16:09:51 +08:00
lingniu
19f260d94a feat(platform): include release in vehicle summary 2026-07-04 16:06:03 +08:00
lingniu
a01a958e60 feat(platform): include release in alert runbook 2026-07-04 15:59:20 +08:00
lingniu
7d6d1afbf2 feat(platform): add alert policy runbook copy 2026-07-04 15:56:48 +08:00
lingniu
fd93ce8d97 feat(platform): include alert escalation in digest 2026-07-04 15:53:29 +08:00
lingniu
1838da4b02 feat(platform): add alert policy escalation 2026-07-04 15:50:45 +08:00
lingniu
ef45a75612 feat(platform): include release in alert digest 2026-07-04 15:46:22 +08:00
lingniu
eaf7fcaa0e feat(platform): show release in quality health 2026-07-04 15:40:41 +08:00
lingniu
d72da88c3f feat(platform): show release in topbar 2026-07-04 15:37:48 +08:00
lingniu
ff36cb4bd1 feat(platform): expose deployed release 2026-07-04 15:34:37 +08:00
lingniu
1f069178d6 feat(platform): expose amap runtime health 2026-07-04 15:30:51 +08:00
lingniu
5bbd110743 feat(platform): show mileage closure check 2026-07-04 15:24:03 +08:00
lingniu
d133b187ad feat(platform): show trajectory evidence quality 2026-07-04 15:20:23 +08:00
lingniu
4742361c7f feat(platform): show realtime amap integration status 2026-07-04 15:17:14 +08:00
lingniu
46024151a0 feat(platform): add vehicle quality evidence actions 2026-07-04 15:12:26 +08:00
lingniu
c96144e92a feat(platform): copy priority alert notification 2026-07-04 15:06:02 +08:00
lingniu
21294e8b79 feat(platform): surface priority alert evidence 2026-07-04 15:02:27 +08:00
lingniu
cac978f46f feat(platform): add operations workflow cockpit 2026-07-04 14:58:42 +08:00
lingniu
df37277717 feat(platform): show realtime coverage metrics 2026-07-04 14:53:35 +08:00
lingniu
10c598e013 feat(platform): show mileage confidence metrics 2026-07-04 14:48:58 +08:00
lingniu
cb49798b6d feat(platform): add quality escalation clock 2026-07-04 14:45:39 +08:00
lingniu
b7dd65e435 feat(platform): proxy amap security requests 2026-07-04 14:39:57 +08:00
lingniu
6e5ceafcb7 feat(platform): show capability status summaries 2026-07-04 14:33:12 +08:00
lingniu
f6710514e3 feat(platform): separate history capability entry 2026-07-04 14:28:27 +08:00
lingniu
c03e890096 feat(platform): add vehicle capability matrix 2026-07-04 14:23:50 +08:00
lingniu
d6c2edce7e feat(platform): expose vehicle operations shortcuts 2026-07-04 14:14:06 +08:00
lingniu
67ebf89db9 feat(platform): link mileage statistics summary 2026-07-04 14:09:27 +08:00
lingniu
a91eefd614 feat(platform): copy quality issue notification 2026-07-04 14:05:48 +08:00
lingniu
e515a4251e feat(platform): show vehicle location posture 2026-07-04 13:58:35 +08:00
lingniu
1365ce2a30 feat(platform): copy vehicle diagnostic summary 2026-07-04 13:52:57 +08:00
lingniu
c4d25593c1 feat(platform): copy quality dispatch ticket 2026-07-04 13:46:41 +08:00
lingniu
b545da3d11 feat(platform): open amap route from trajectory 2026-07-04 13:43:49 +08:00
lingniu
1da75a5d75 feat(platform): copy realtime operations summary 2026-07-04 13:40:48 +08:00
lingniu
d77754b0a0 feat(platform): copy trajectory playback summary 2026-07-04 13:35:28 +08:00
lingniu
ff833bc065 feat(platform): add quality notification plan 2026-07-04 13:31:36 +08:00
lingniu
817590c104 feat(platform): copy vehicle governance summary 2026-07-04 13:23:56 +08:00
lingniu
9d35f65be6 feat(platform): copy mileage statistics summary 2026-07-04 13:19:08 +08:00
lingniu
bcc9355400 feat(platform): copy alert priority digest 2026-07-04 13:15:57 +08:00
lingniu
47dfd18462 feat(platform): surface amap operations shortcuts 2026-07-04 13:12:44 +08:00
lingniu
2cff1637ae feat(platform): export realtime vehicle page 2026-07-04 13:07:56 +08:00
lingniu
c6875e12a5 feat(platform): export mileage statistics page 2026-07-04 13:03:29 +08:00
lingniu
8c35bf9f2c feat(platform): include evidence links in alert notifications 2026-07-04 13:00:10 +08:00
lingniu
783c8eba5e feat(platform): copy quality alert notification 2026-07-04 12:56:46 +08:00
lingniu
89ddb82f46 feat(platform): prevent unscoped raw field lookups 2026-07-04 12:53:07 +08:00
lingniu
34116142e6 fix(platform): guard unscoped raw field queries 2026-07-04 12:45:53 +08:00
lingniu
773338cce9 feat(platform): link alert queue to raw evidence 2026-07-04 12:42:55 +08:00
lingniu
a0f17f35c4 feat(platform): link alert queue to trajectory evidence 2026-07-04 12:39:29 +08:00
lingniu
2e9d3fa5eb feat(platform): add alert realtime map action 2026-07-04 12:36:48 +08:00
lingniu
bed79e164d feat(platform): link quality issues to raw evidence 2026-07-04 12:32:51 +08:00
lingniu
a960d1fdab feat(platform): link locations to raw evidence 2026-07-04 12:28:09 +08:00
lingniu
5a133b0bcd feat(platform): surface realtime source consistency 2026-07-04 12:23:53 +08:00
lingniu
8dbf91ca25 feat(platform): link raw frames to mileage evidence 2026-07-04 12:18:06 +08:00
lingniu
2991a01738 feat(platform): link mileage rows to raw evidence 2026-07-04 12:14:08 +08:00
lingniu
2465bd7f16 feat(platform): link quality issues to realtime status 2026-07-04 12:10:28 +08:00
lingniu
4ef71c37f2 feat(platform): link realtime rows to quality evidence 2026-07-04 12:06:46 +08:00
lingniu
276da59194 feat(platform): link vehicle rows to quality evidence 2026-07-04 12:01:28 +08:00
lingniu
fb6d8c323a feat(platform): link quality issues to mileage evidence 2026-07-04 11:57:55 +08:00
lingniu
f305066eff feat(platform): link history locations to mileage evidence 2026-07-04 11:54:48 +08:00
lingniu
2ba30b21cd docs(platform): define vehicle service center direction 2026-07-04 11:48:57 +08:00
lingniu
b03f9b1a3a feat(platform): link quality issues to history evidence 2026-07-04 11:43:41 +08:00
lingniu
27e99f21d8 feat(platform): link mileage rows to trajectory evidence 2026-07-04 11:40:46 +08:00
lingniu
8327eeb2f5 feat(platform): add realtime map actions 2026-07-04 11:37:01 +08:00
lingniu
442ca3a275 feat(platform): add trajectory playback controls 2026-07-04 11:32:14 +08:00
lingniu
e057dc151f feat(platform): add realtime map service queue 2026-07-04 11:27:22 +08:00
lingniu
02011035ae feat(platform): prioritize alert handling 2026-07-04 11:22:06 +08:00
lingniu
e94d0e348f feat(platform): add vehicle command center 2026-07-04 11:15:56 +08:00
lingniu
688cf0a155 feat(platform): export vehicle service dossier 2026-07-04 11:07:46 +08:00
lingniu
8dc64bd294 feat(platform): export vehicle coverage results 2026-07-04 11:02:33 +08:00
lingniu
cd9d160f3e feat(platform): export history query results 2026-07-04 10:59:47 +08:00
lingniu
312382c95d feat(platform): enhance mileage statistics workspace 2026-07-04 10:56:10 +08:00
lingniu
adb9d146d9 feat(platform): integrate runtime amap vehicle maps 2026-07-04 10:52:11 +08:00
lingniu
bb1b2f1c46 feat(platform): add alert notification operations workspace 2026-07-04 10:43:32 +08:00
lingniu
be5a0cf92c feat(platform): add trajectory playback workspace 2026-07-04 10:38:10 +08:00
lingniu
89c4fffd5d feat(platform): reshape telematics operations console 2026-07-04 10:32:51 +08:00
lingniu
1f116ccaa8 feat(platform): share vehicle service verdict logic 2026-07-04 10:25:41 +08:00
lingniu
295a7df396 feat(platform): summarize vehicle service verdict 2026-07-04 10:21:34 +08:00
lingniu
e0de4b353a feat(platform): guide mileage anomaly handling 2026-07-04 10:17:19 +08:00
lingniu
56f07c3488 feat(platform): recommend quality issue actions 2026-07-04 10:13:26 +08:00
lingniu
782918b15f feat(platform): prioritize vehicle action queue 2026-07-04 10:09:52 +08:00
lingniu
e59e5f7c11 feat(platform): enrich vehicle source evidence matrix 2026-07-04 10:06:43 +08:00
lingniu
7925be676b feat(platform): show archive gaps on vehicle detail 2026-07-04 10:00:35 +08:00
lingniu
adff22ef74 feat(platform): show archive gaps per vehicle 2026-07-04 09:56:18 +08:00
lingniu
4791a1fd9c feat(platform): surface archive gap actions 2026-07-04 09:53:08 +08:00
lingniu
b5a72ce0ec feat(platform): filter vehicle archive gaps 2026-07-04 09:50:41 +08:00
lingniu
b127b22a4b feat(platform): surface incomplete vehicle archives 2026-07-04 09:45:03 +08:00
lingniu
29f84099b2 feat(platform): filter vehicles by archive status 2026-07-04 09:37:51 +08:00
lingniu
817e5324f6 feat(platform): show vehicle archive completeness 2026-07-04 09:33:45 +08:00
lingniu
d4ff7d8b83 feat(platform): show canonical vehicle archive 2026-07-04 09:31:13 +08:00
lingniu
cb45c44a05 feat(platform): surface dashboard action queue 2026-07-04 09:28:17 +08:00
lingniu
16dd9feb1c feat(platform): recommend vehicle detail actions 2026-07-04 09:24:46 +08:00
lingniu
5952bb1a6f feat(platform): summarize vehicle action queue 2026-07-04 09:15:02 +08:00
lingniu
c4ff0079cf feat(platform): make vehicle action recommendations filterable 2026-07-04 09:10:54 +08:00
lingniu
a68672fe16 feat(platform): recommend vehicle service actions 2026-07-04 09:07:47 +08:00
lingniu
d8bea4eb0c feat(platform): add vehicle service posture drilldowns 2026-07-04 09:05:23 +08:00
lingniu
ad932565ed feat(platform): highlight unified vehicle service posture 2026-07-04 09:01:34 +08:00
lingniu
f651e48434 feat(platform): expose capacity check findings 2026-07-04 08:58:40 +08:00
lingniu
e29f8768ca feat(platform): expose active connection capacity 2026-07-04 08:52:57 +08:00
lingniu
b26a37f6f7 feat(platform): report capacity check health 2026-07-04 08:49:44 +08:00
lingniu
ff8a868abd fix(platform): avoid false storage error while health loads 2026-07-04 08:47:11 +08:00
lingniu
1f777c4040 feat(platform): report redis online key health 2026-07-04 08:44:26 +08:00
lingniu
d671184c09 fix(platform): keep dashboard metrics on preview failures 2026-07-04 08:40:35 +08:00
lingniu
7c1ad16018 fix(platform): return empty raw frames for missing tdengine tables 2026-07-04 08:36:37 +08:00
lingniu
6ce46f3011 fix(platform): return empty history for missing tdengine tables 2026-07-04 08:33:41 +08:00
lingniu
7c81285be8 feat(platform): show active history filters 2026-07-04 08:30:12 +08:00
lingniu
208d48115d feat(platform): show active mileage filters 2026-07-04 08:26:41 +08:00
lingniu
1c75f1d5f1 feat(platform): show active realtime filters 2026-07-04 08:23:12 +08:00
lingniu
a34a2620c7 feat(platform): show active vehicle filters 2026-07-04 08:20:24 +08:00
lingniu
fc02376272 feat(platform): show active quality filters 2026-07-04 08:17:32 +08:00
lingniu
c923ceefad feat(platform): drill quality issues by type 2026-07-04 08:13:42 +08:00
lingniu
a1e384f78d feat(platform): drill quality issues by source 2026-07-04 08:11:01 +08:00
lingniu
7a3f007340 feat(platform): open quality governance from vehicle detail 2026-07-04 08:07:47 +08:00
lingniu
5fa7858662 fix(platform): unify degraded service wording 2026-07-04 08:03:24 +08:00
lingniu
4251ea5235 fix(platform): align topbar missing source status 2026-07-04 07:58:24 +08:00
lingniu
1d691306db feat(platform): keep canonical source summary slots 2026-07-04 07:54:38 +08:00
lingniu
422ac684ed fix(platform): count source online by protocol 2026-07-04 07:50:37 +08:00
lingniu
ec00dd0166 feat(platform): summarize vehicle source diagnosis 2026-07-04 07:47:57 +08:00
lingniu
36b75b4c0e feat(platform): show source slots on vehicle detail 2026-07-04 07:44:23 +08:00
lingniu
999d36c94d feat(platform): expose source slots in realtime vehicles 2026-07-04 07:39:09 +08:00
lingniu
f93c8d6ccc feat(platform): unify source status tags 2026-07-04 07:34:18 +08:00
lingniu
c528376ef9 feat(platform): expose source slots in vehicle service 2026-07-04 07:30:56 +08:00
lingniu
2cfb9e551f feat(platform): open vehicle list from dashboard coverage 2026-07-04 07:25:13 +08:00
lingniu
86b65f682d feat(platform): show source consistency on dashboard coverage 2026-07-04 07:20:26 +08:00
lingniu
455e288fb0 feat(platform): make source consistency actionable 2026-07-04 07:17:19 +08:00
lingniu
c9a809d13d feat(platform): expose source consistency in vehicle coverage 2026-07-04 07:12:54 +08:00
lingniu
df78bb5e20 fix(platform): keep realtime source when opening vehicle 2026-07-04 07:08:42 +08:00
lingniu
84f578bcc7 feat(platform): add vehicle detail realtime shortcut 2026-07-04 07:04:55 +08:00
lingniu
0366144780 feat(platform): add vehicle detail raw shortcut 2026-07-04 07:01:35 +08:00
lingniu
db0b979ffb feat(platform): persist mileage filters in route 2026-07-04 06:58:53 +08:00
lingniu
266b20b97a feat(platform): persist history filters in route 2026-07-04 06:54:29 +08:00
lingniu
b9534af233 feat(platform): persist realtime source filters 2026-07-04 06:45:38 +08:00
lingniu
2feaf246be fix(platform): load vehicle context by source 2026-07-04 06:41:55 +08:00
lingniu
c70af50cec fix(platform): keep vehicle list source filter 2026-07-04 06:38:06 +08:00
lingniu
5234eb5542 fix(platform): keep dashboard source when opening vehicle 2026-07-04 06:34:23 +08:00
lingniu
2997009ac2 fix(platform): keep quality source when opening vehicle 2026-07-04 06:30:27 +08:00
lingniu
9fc6f72b2b fix(platform): keep realtime source when opening vehicle 2026-07-04 06:25:42 +08:00
lingniu
857c96734d fix(platform): keep mileage source when opening vehicle 2026-07-04 06:23:20 +08:00
lingniu
6dd00f4ebb fix(platform): keep history source when opening vehicle 2026-07-04 06:20:29 +08:00
lingniu
b9a173f8d2 feat(platform): open raw evidence from vehicle detail 2026-07-04 06:17:27 +08:00
lingniu
8cd90f629b feat(platform): summarize missing vehicle sources 2026-07-04 06:13:38 +08:00
lingniu
1c210897bb feat(platform): filter vehicles from missing source tags 2026-07-04 06:08:58 +08:00
lingniu
111154ed39 feat(platform): link missing source diagnosis to vehicles 2026-07-04 06:05:40 +08:00
lingniu
eb7c47b4f5 fix(platform): align vehicle detail service statuses 2026-07-04 06:02:32 +08:00
lingniu
a60f87c9e7 feat(platform): expose missing source diagnosis 2026-07-04 06:00:06 +08:00
lingniu
bf17c7437e fix(platform): clarify vehicle service status labels 2026-07-04 05:52:11 +08:00
lingniu
6229f837a0 fix(platform): degrade vehicles with missing sources 2026-07-04 05:46:56 +08:00
lingniu
5e7c4b56ef feat(platform): expose missing sources per vehicle 2026-07-04 05:42:39 +08:00
lingniu
bc6715b35b feat(platform): summarize missing vehicle sources 2026-07-04 05:38:17 +08:00
lingniu
6ac980adad feat(platform): filter dashboard by missing source 2026-07-04 05:34:17 +08:00
lingniu
7c3b506fe7 feat(platform): preserve missing source filters 2026-07-04 05:30:40 +08:00
lingniu
999d58da1d feat(platform): filter vehicles by missing source 2026-07-04 05:28:29 +08:00
lingniu
ef35137236 feat(platform): clarify missing source role 2026-07-04 05:24:25 +08:00
lingniu
d51c83db2a feat(platform): expose missing vehicle source slots 2026-07-04 05:21:54 +08:00
lingniu
ff74375c45 feat(platform): unify vehicle source matrix 2026-07-04 05:17:49 +08:00
lingniu
0e7374b039 feat(platform): preserve history source evidence 2026-07-04 05:14:21 +08:00
lingniu
31529230c0 feat(platform): copy vehicle service detail links 2026-07-04 05:10:17 +08:00
lingniu
65e3c53f69 feat(platform): show vehicle service evidence chain 2026-07-04 05:07:43 +08:00
lingniu
9e35c470a9 feat(platform): copy vehicle filter links 2026-07-04 05:03:21 +08:00
lingniu
4d9422a434 feat(platform): label dashboard protocols as evidence 2026-07-04 04:58:27 +08:00
lingniu
59bda6a6eb feat(platform): open unified vehicle service by default 2026-07-04 04:56:03 +08:00
lingniu
2dfe0fbf3a feat(platform): frame realtime as vehicle service 2026-07-04 04:53:21 +08:00
lingniu
04645c888d feat(platform): copy quality filter links 2026-07-04 04:50:01 +08:00
lingniu
753faca269 feat(platform): persist quality filters in route 2026-07-04 04:45:52 +08:00
lingniu
b53cfbabfe feat(platform): filter quality issues by type 2026-07-04 04:42:12 +08:00
lingniu
7ed5d7620b feat(platform): align quality issues with vehicle service 2026-07-04 04:38:09 +08:00
lingniu
669390e8bb feat(platform): surface no-source quality issues 2026-07-04 04:34:32 +08:00
lingniu
40b55600c7 feat(platform): include no-data service status 2026-07-04 04:28:27 +08:00
lingniu
40397efa4a feat(platform): expose no-data vehicles 2026-07-04 04:25:29 +08:00
lingniu
7f9d327f38 feat(platform): expose single-source vehicles 2026-07-04 04:18:20 +08:00
lingniu
0b53abf63d feat(platform): label source evidence 2026-07-04 04:15:42 +08:00
lingniu
e077d99a5e feat(platform): diagnose source consistency 2026-07-04 04:12:23 +08:00
lingniu
7b801b6f71 feat(platform): surface overview consistency 2026-07-04 04:06:54 +08:00
lingniu
db70f58a7c feat(platform): expose source consistency 2026-07-04 04:01:33 +08:00
lingniu
353b71ebbb feat(platform-web): show source consistency 2026-07-04 03:57:26 +08:00
lingniu
2e50f43ff4 feat(platform-web): show vehicle analysis scope 2026-07-04 03:52:55 +08:00
lingniu
a0772be4a6 feat(platform-web): add source quick actions 2026-07-04 03:48:55 +08:00
lingniu
ede44ee3d9 feat(platform-web): make vehicle summary filters actionable 2026-07-04 03:46:12 +08:00
lingniu
b7e8aae317 feat(platform): add filtered vehicle coverage summary 2026-07-04 03:42:19 +08:00
lingniu
6683ce4df7 feat(platform-web): summarize vehicle list results 2026-07-04 03:37:20 +08:00
lingniu
a2ccc4a498 feat(platform-web): link summary kpis to vehicle filters 2026-07-04 03:32:51 +08:00
lingniu
948dc774d0 feat(platform-web): surface vehicle service summary 2026-07-04 03:27:35 +08:00
lingniu
2872e8c94d feat(platform): add vehicle service summary 2026-07-04 03:23:21 +08:00
lingniu
a3b86c7617 fix(platform): preserve fuzzy vehicle overview search 2026-07-04 03:19:29 +08:00
lingniu
5d49675582 perf(platform): reuse batch path for vehicle overview 2026-07-04 03:15:31 +08:00
lingniu
e35565f993 feat(platform): batch vehicle service overview lookup 2026-07-04 03:12:21 +08:00
lingniu
f996fa0df9 feat(platform): add batch vehicle overview api 2026-07-04 03:05:27 +08:00
lingniu
1a86e5d38c feat(platform): return service status in vehicle overview 2026-07-04 03:00:50 +08:00
lingniu
8dfbdc83e5 feat(platform-web): use lightweight overview for vehicle context 2026-07-04 02:57:11 +08:00
lingniu
6c745bf48b feat(platform): add lightweight vehicle overview api 2026-07-04 02:52:56 +08:00
lingniu
38e1bf422a feat(platform): add vehicle service overview 2026-07-04 02:49:06 +08:00
lingniu
131f59360c test(platform-web): cover vehicle service navigation context 2026-07-04 02:44:20 +08:00
lingniu
d67540c457 feat(platform-web): resolve vehicle context from detail hash 2026-07-04 02:39:31 +08:00
lingniu
80bd726455 feat(platform-web): show searched vehicle identity context 2026-07-04 02:35:55 +08:00
lingniu
337e9f3dac feat(platform-web): show searched vehicle service status 2026-07-04 02:33:04 +08:00
lingniu
58b125912e feat(platform): include service status in vehicle resolution 2026-07-04 02:29:44 +08:00
lingniu
e7de41894b feat(platform): filter vehicle rows by service status 2026-07-04 02:26:39 +08:00
lingniu
cbd02956ff feat(platform): expose service status on vehicle rows 2026-07-04 02:23:57 +08:00
lingniu
826fa76008 feat(platform-web): show dashboard coverage filter context 2026-07-04 02:19:59 +08:00
lingniu
a6edc02be5 feat(platform-web): drill into service status coverage 2026-07-04 02:16:19 +08:00
lingniu
b36d0480ee feat(platform-web): filter dashboard coverage by service status 2026-07-04 02:13:53 +08:00
lingniu
4c27cfa28d feat(platform-web): show dashboard row service status 2026-07-04 02:11:32 +08:00
lingniu
6b1e36a5a6 feat(platform): expose realtime service status 2026-07-04 02:08:31 +08:00
lingniu
526377c13f feat(platform): expose coverage service status 2026-07-04 02:03:12 +08:00
lingniu
5d401a8f0b feat(platform): summarize service status on dashboard 2026-07-04 01:59:05 +08:00
lingniu
14408d135d feat(platform): filter vehicles by service status 2026-07-04 01:54:30 +08:00
lingniu
b7de38605d feat(platform-web): show vehicle list service status 2026-07-04 01:50:06 +08:00
lingniu
aa3b39f5e3 feat(platform): summarize vehicle service status 2026-07-04 01:47:24 +08:00
lingniu
8e800f795c feat(platform): add vehicle service API 2026-07-04 01:43:26 +08:00
lingniu
5000d5cd1c feat(platform-web): show vehicle detail source scope 2026-07-04 01:39:18 +08:00
lingniu
dc3e12328e feat(platform-web): link analysis pages to vehicle service 2026-07-04 01:35:15 +08:00
lingniu
9886fe2481 feat(platform-web): carry detail source to analysis links 2026-07-04 01:31:50 +08:00
lingniu
a12d86e6f4 feat(platform-web): switch vehicle detail sources 2026-07-04 01:28:08 +08:00
lingniu
edca567d52 feat(platform-web): keep realtime source in vehicle links 2026-07-04 01:25:13 +08:00
lingniu
55834a4651 feat(platform-web): keep protocol when opening vehicle rows 2026-07-04 01:22:41 +08:00
lingniu
5f84139c00 feat(platform-web): apply protocol in shared history links 2026-07-04 01:18:02 +08:00
lingniu
657674f233 feat(platform-web): preserve protocol in vehicle links 2026-07-04 01:15:39 +08:00
lingniu
8e5cb5b8c7 feat(platform-web): support shareable vehicle routes 2026-07-04 01:13:07 +08:00
lingniu
beb5034b9b feat(platform): include identity resolution in vehicle detail 2026-07-04 01:09:35 +08:00
lingniu
703a7b112f feat(platform): resolve vehicle identity before navigation 2026-07-04 01:06:24 +08:00
lingniu
3e76210d7d feat(platform): expose vehicle service runtime health 2026-07-04 01:02:07 +08:00
lingniu
377cfcdfca feat(platform-api): return traceable timeout errors 2026-07-04 00:57:53 +08:00
lingniu
4bfaf8c4c8 feat(platform-api): enforce request timeout 2026-07-04 00:55:43 +08:00
lingniu
081f8b5099 feat(platform-web): include trace id in api errors 2026-07-04 00:52:42 +08:00
lingniu
bbbf589d60 feat(platform-web): surface backend api errors 2026-07-04 00:50:17 +08:00
lingniu
894d200a16 feat(platform-web): query raw frames with post body 2026-07-04 00:48:13 +08:00
lingniu
a4ff210c82 docs(platform): define vehicle keyword API contract 2026-07-04 00:45:29 +08:00
lingniu
20e2781f35 feat(platform): use vehicle keyword for data queries 2026-07-04 00:42:40 +08:00
lingniu
f62fe13647 feat(platform-api): resolve vehicle keywords for data queries 2026-07-04 00:37:47 +08:00
lingniu
0f371ab3c9 feat(platform): focus unresolved vehicle detail on binding 2026-07-04 00:29:02 +08:00
lingniu
3cb2d3479b feat(platform): skip vehicle sections for unresolved lookup 2026-07-04 00:22:39 +08:00
lingniu
f30eb4fbff feat(platform): explain unresolved vehicle binding 2026-07-04 00:19:25 +08:00
lingniu
cb1a3097d8 feat(platform): separate vehicle lookup key from VIN 2026-07-04 00:16:04 +08:00
lingniu
52b3040506 feat(platform): show unresolved vehicle lookup state 2026-07-04 00:11:31 +08:00
lingniu
ff1fb4d2ac refactor(platform): centralize vehicle lookup rules 2026-07-04 00:08:35 +08:00
lingniu
0503b03aa1 feat(platform): clarify quality vehicle lookup action 2026-07-04 00:05:25 +08:00
lingniu
bfdfb5761a feat(platform): link quality issues to vehicle lookup 2026-07-04 00:02:33 +08:00
lingniu
a5ca34797d feat(platform): show vehicle detail lookup key 2026-07-03 23:59:34 +08:00
lingniu
8af622e7f6 feat(platform): support vehicle keyword detail lookup 2026-07-03 23:57:11 +08:00
lingniu
d4fdd7527b feat(platform): summarize vehicle source coverage 2026-07-03 23:51:40 +08:00
lingniu
6c53172ba7 feat(platform): center navigation on vehicle service 2026-07-03 23:49:01 +08:00
lingniu
fe92b34356 feat(platform): show quality issue identity on dashboard 2026-07-03 23:44:53 +08:00
lingniu
562c26f061 feat(platform): show vehicle-level latest dashboard 2026-07-03 23:41:49 +08:00
lingniu
fd60d2644e feat(platform): link quality issues to vehicle service 2026-07-03 23:39:35 +08:00
lingniu
532b0d9aec feat(platform): open analysis from vehicle detail 2026-07-03 23:37:12 +08:00
lingniu
37ca94ab5f feat(platform): link realtime to vehicle service 2026-07-03 23:34:14 +08:00
lingniu
05c0643a69 feat(platform): link records to vehicle service 2026-07-03 23:32:19 +08:00
lingniu
73d32d3b0e feat(platform): copy quality issue identity 2026-07-03 23:30:09 +08:00
lingniu
c9e94639d1 feat(platform): expose quality issue identity 2026-07-03 23:28:10 +08:00
lingniu
54d74178ea feat(platform): summarize quality issues 2026-07-03 23:25:39 +08:00
lingniu
4ca1bdb8fe feat(platform): summarize mileage queries 2026-07-03 23:21:51 +08:00
lingniu
a4e9a8afb8 feat(platform): include plates in raw history 2026-07-03 23:17:23 +08:00
lingniu
a644a6617c feat(platform): enrich history locations with plates 2026-07-03 23:14:40 +08:00
lingniu
55b069e27d feat(platform): summarize realtime in vehicle detail 2026-07-03 23:12:27 +08:00
lingniu
37f9d7a307 feat(platform): aggregate realtime by vehicle 2026-07-03 23:09:28 +08:00
lingniu
71b714bbf7 feat(platform): color topbar link health 2026-07-03 23:03:42 +08:00
lingniu
b73292b97e feat(platform): poll topbar link health 2026-07-03 23:01:01 +08:00
lingniu
909f7d6720 feat(platform): refresh quality link health 2026-07-03 22:59:15 +08:00
lingniu
ea367fe581 feat(platform): refresh topbar link issues from quality 2026-07-03 22:57:21 +08:00
lingniu
b3168dda2a feat(platform): show link issue count in topbar 2026-07-03 22:54:44 +08:00
lingniu
94ea30223b fix(platform): remove fake healthy topbar 2026-07-03 22:52:52 +08:00
lingniu
3b29851c1e feat(platform): link dashboard quality preview 2026-07-03 22:50:40 +08:00
lingniu
ad76eb45e4 feat(platform): preview quality issues on dashboard 2026-07-03 22:48:51 +08:00
lingniu
d3c2d8dac5 fix(platform): label storage health as reads 2026-07-03 22:46:46 +08:00
lingniu
98b23df2ed feat(platform): highlight quality link health 2026-07-03 22:45:17 +08:00
lingniu
354c146506 fix(platform): verify storage health by reads 2026-07-03 22:43:36 +08:00
lingniu
2c24d4734c fix(platform): avoid fake redis online keys 2026-07-03 22:41:25 +08:00
lingniu
91c2168cb0 fix(platform): avoid fake kafka lag 2026-07-03 22:38:50 +08:00
lingniu
4ff94a4e92 fix(platform): count dashboard frames from tdengine 2026-07-03 22:36:39 +08:00
lingniu
2efb62d8d1 fix(platform): show vehicle quality total 2026-07-03 22:34:50 +08:00
lingniu
42fec61b4b fix(platform): align dashboard quality count 2026-07-03 22:33:19 +08:00
lingniu
e68028799e fix(platform): count dashboard vehicles by vin 2026-07-03 22:31:25 +08:00
lingniu
1293035045 feat(platform): count vehicle list pages 2026-07-03 22:29:46 +08:00
lingniu
28d1a68823 feat(platform): paginate quality issues 2026-07-03 22:27:40 +08:00
lingniu
39e2788e26 feat(platform): paginate daily mileage 2026-07-03 22:23:10 +08:00
lingniu
72aa20b347 feat(platform): paginate history queries 2026-07-03 22:19:14 +08:00
lingniu
ea96b6909a feat(platform): paginate realtime locations 2026-07-03 22:15:16 +08:00
lingniu
c34996dcfb feat(platform): paginate vehicle coverage 2026-07-03 22:12:04 +08:00
lingniu
9359254d28 feat(platform): show vehicles as service coverage 2026-07-03 22:06:59 +08:00
lingniu
2fcce04c5a feat(platform): surface vehicle quality issues 2026-07-03 22:04:12 +08:00
lingniu
2d064e3c53 feat(platform): filter vehicle coverage 2026-07-03 21:59:03 +08:00
lingniu
110041a9db feat(platform): add vehicle coverage view 2026-07-03 21:54:09 +08:00
lingniu
6352289c25 feat(platform): summarize vehicle source status 2026-07-03 21:47:30 +08:00
lingniu
a3a91ef1f8 feat(platform): resolve vehicle detail keyword 2026-07-03 21:40:40 +08:00
lingniu
92d7020b36 feat(platform): route vehicle searches to detail 2026-07-03 21:37:32 +08:00
lingniu
d4bd70fdf4 feat(platform): add vehicle detail aggregate api 2026-07-03 21:34:50 +08:00
lingniu
0438d054f6 feat(platform): make vehicle detail source agnostic 2026-07-03 21:31:11 +08:00
lingniu
f9b8182949 feat(platform): query tdengine history data 2026-07-03 21:22:21 +08:00
lingniu
2de569f104 feat(platform): connect mysql production store 2026-07-03 21:09:43 +08:00
lingniu
859bc3e9ee feat(platform): add vehicle data management console 2026-07-03 20:55:54 +08:00
lingniu
0d8916df47 docs: add vehicle data platform implementation plan 2026-07-03 20:25:09 +08:00
lingniu
8acd3806c9 docs: add vehicle data platform design spec 2026-07-03 20:13:15 +08:00
lingniu
c735372cc1 ops(go): schedule capacity health checks 2026-07-03 19:59:10 +08:00
lingniu
3edd7462c7 feat(go): add capacity health check tool 2026-07-03 19:54:29 +08:00
lingniu
68b729fa7b fix(go): singleflight tdengine child table ensures 2026-07-03 19:49:04 +08:00
lingniu
3a0d9bd840 feat(go): batch redis fast writer projections 2026-07-03 19:45:00 +08:00
lingniu
81e050ec1c feat(go): qualify tdengine history writes by database 2026-07-03 19:39:33 +08:00
lingniu
fb34969602 fix(go): keep fast writer tdengine pool safe by default 2026-07-03 19:33:45 +08:00
lingniu
06866b0fa7 feat(go): expose gateway connection close metrics 2026-07-03 19:28:06 +08:00
lingniu
d08bf17111 feat(go): expose fast writer nats consumer metrics 2026-07-03 19:24:02 +08:00
lingniu
c43b87d648 feat(go): expose fast writer batch pending metrics 2026-07-03 19:20:09 +08:00
lingniu
3b694671fa feat(go): batch nats fast writer tdengine writes 2026-07-03 19:17:02 +08:00
lingniu
2db05d100d feat(go): expose fast writer stage latency histogram 2026-07-03 19:12:58 +08:00
lingniu
07d17e5adf feat(go): expose stat writer latency histogram 2026-07-03 19:08:49 +08:00
lingniu
3900144ea4 feat(go): expose realtime store latency histogram 2026-07-03 19:05:58 +08:00
lingniu
afb301cb7a feat(go): expose async sink publish latency histogram 2026-07-03 19:02:24 +08:00
lingniu
018015f9de feat(go): expose bridge batch pressure metrics 2026-07-03 18:58:37 +08:00
lingniu
16ee09180c feat(go): expose gateway frame duration histogram 2026-07-03 18:54:00 +08:00
lingniu
71905aca3d feat(go): expose history batch pending metrics 2026-07-03 18:46:40 +08:00
lingniu
1cbe29c5fc feat(go): generate unique load simulator frames 2026-07-03 18:43:15 +08:00
lingniu
09f3e68891 feat(go): batch tdengine history writes 2026-07-03 18:37:38 +08:00
lingniu
3b0c5fbd5a ops: record gateway async sink metrics rollout 2026-07-03 18:30:50 +08:00
lingniu
fb03c80a95 feat(go): expose gateway async sink metrics 2026-07-03 18:28:07 +08:00
lingniu
f8fbeb8936 test(go): record 100k gateway connection baseline 2026-07-03 18:24:07 +08:00
lingniu
65232812a1 test(go): record 10k gateway connection baseline 2026-07-03 18:09:53 +08:00
lingniu
291bffa340 feat(go): add load simulator hold mode 2026-07-03 18:01:40 +08:00
lingniu
c1fe1fbfb5 ops: record 100k hardening ecs rollout 2026-07-03 17:59:44 +08:00
lingniu
79e591f864 feat(go): add 100k ingest hardening baseline 2026-07-03 17:55:22 +08:00
lingniu
378a5d5e57 feat(go): reuse parsed fields across projections 2026-07-03 17:38:50 +08:00
lingniu
368b18fc96 feat(go): flatten mysql realtime snapshots 2026-07-03 16:00:40 +08:00
lingniu
2b5de65e31 feat(go): remove mysql realtime kv 2026-07-03 15:52:53 +08:00
lingniu
f292d4c0d6 feat(go): make identity binding read only 2026-07-03 15:45:47 +08:00
lingniu
1d79dc25e0 feat(go): add oem to identity binding schema 2026-07-03 15:31:52 +08:00
lingniu
9994f2baf4 feat(go): improve realtime pipeline resilience 2026-07-03 15:20:22 +08:00
lingniu
b8299faefa feat(go): add realtime pipeline observability 2026-07-03 15:06:27 +08:00
lingniu
28c81611a1 feat(go): add post raw frame query 2026-07-03 14:54:52 +08:00
lingniu
c0d3b41550 feat(go): filter raw parsed fields 2026-07-03 14:49:53 +08:00
lingniu
a0a4997e79 fix(go): avoid gb32960 decimal float tails 2026-07-03 14:47:11 +08:00
lingniu
96e9382d73 fix(go): normalize history parsed field names 2026-07-03 14:42:10 +08:00
lingniu
6624930cc1 perf(go): gzip history raw field responses 2026-07-03 14:38:48 +08:00
lingniu
6318c10c41 perf(go): optimize tdengine raw frame queries 2026-07-03 14:25:48 +08:00
lingniu
535acdb2a6 feat(go): publish realtime fields stream 2026-07-03 14:08:32 +08:00
lingniu
e1bf2d72e1 refactor(go): map realtime fields to semantic names 2026-07-03 13:51:57 +08:00
lingniu
c359930009 refactor(go): simplify redis realtime projections 2026-07-03 13:42:15 +08:00
lingniu
a96490618c fix(go): collapse gb32960 auxiliary snapshot 2026-07-03 13:17:31 +08:00
lingniu
112b02e8c5 fix(go): store protocol parsed fields in realtime kv 2026-07-03 12:08:12 +08:00
lingniu
7bf78f91ba fix(go): project full yutong mqtt data to realtime kv 2026-07-03 12:01:32 +08:00
lingniu
0344c65e5b feat(go): add nats fast writer 2026-07-03 11:59:33 +08:00
lingniu
76f5a20b79 perf(go): decouple realtime mysql projection 2026-07-03 11:49:13 +08:00
lingniu
6fb6262d0d feat(go): add realtime kv projection 2026-07-03 11:34:52 +08:00
lingniu
b948a46e64 fix(go): normalize gb32960 realtime vendor fragments 2026-07-03 11:20:17 +08:00
lingniu
8789a92428 fix(go): merge gb32960 realtime snapshot payloads 2026-07-03 11:14:57 +08:00
lingniu
0fa5343ab3 fix(go): preserve sparse realtime location fields 2026-07-03 10:56:38 +08:00
lingniu
9c212815ef fix(go): filter realtime mysql snapshots by business events 2026-07-03 10:42:32 +08:00
lingniu
02075b5ea0 fix(go): resolve jt808 locations through registration 2026-07-03 10:30:20 +08:00
lingniu
cabdac1fe2 feat(go): enrich realtime snapshot context 2026-07-03 10:17:10 +08:00
lingniu
75fbd92de9 fix(go): write tdengine timestamps as epoch millis 2026-07-03 09:17:53 +08:00
lingniu
a16043c032 fix(go): guard realtime mysql against stale events 2026-07-03 09:03:46 +08:00
lingniu
c2e4c2b955 perf(go): trim frame id from location history 2026-07-03 08:56:04 +08:00
lingniu
c916120ea2 perf(go): skip empty parsed raw payloads 2026-07-03 08:51:30 +08:00
lingniu
d4c164480f perf(go): avoid mqtt raw hex duplication 2026-07-03 08:48:07 +08:00
lingniu
1830798d2e fix(go): cache mileage after successful write 2026-07-03 08:43:58 +08:00
lingniu
8f5afd20a6 perf(go): deduplicate unchanged mileage samples 2026-07-03 08:39:48 +08:00
lingniu
c1640b4332 perf(go): skip empty realtime payload snapshots 2026-07-03 08:35:14 +08:00
lingniu
97362e3205 perf(go): start realtime projector at latest offset 2026-07-03 08:30:18 +08:00
lingniu
4c34c1221b refactor(go): make raw topics the realtime source 2026-07-03 08:25:14 +08:00
lingniu
c2d058bf75 refactor(go): keep realtime cache vin scoped 2026-07-03 08:12:21 +08:00
lingniu
400047e08c perf(go): cache identity binding lookups 2026-07-03 08:07:32 +08:00
lingniu
0b9e803139 perf(go): simplify identity binding lookup 2026-07-03 08:01:00 +08:00
lingniu
2af2ba3367 perf(go): simplify realtime plate lookup 2026-07-03 07:57:04 +08:00
lingniu
0aa71402d1 docs(go): document realtime total semantics 2026-07-03 07:53:30 +08:00
lingniu
b3ea2bd91f refactor(go): make realtime totals opt in 2026-07-03 07:50:14 +08:00
lingniu
eced1873cd refactor(go): drop unused realtime geo index 2026-07-03 07:46:21 +08:00
lingniu
1c0d7dc03e perf(go): cache realtime plate lookups 2026-07-03 07:42:24 +08:00
lingniu
9345fe60b5 refactor(go): drop redundant daily mileage vin index 2026-07-03 07:36:53 +08:00
lingniu
cb63ceec2e refactor(go): make daily metric totals opt in 2026-07-03 07:32:09 +08:00
lingniu
b149660b4b refactor(go): make history totals opt in 2026-07-03 07:28:02 +08:00
lingniu
bb0d001d72 refactor(go): simplify identity table keys 2026-07-03 07:22:18 +08:00
lingniu
04bafb4cd6 refactor(go): simplify daily mileage table key 2026-07-03 07:16:49 +08:00
lingniu
e89c9fd98f refactor(go): simplify realtime mysql current tables 2026-07-03 07:11:40 +08:00
lingniu
7c41b81654 refactor(go): remove duplicate mileage history api 2026-07-02 21:34:12 +08:00
lingniu
e27af63025 refactor(go): simplify history location identity 2026-07-02 21:28:29 +08:00
lingniu
ef2be36fce refactor(go): remove source endpoint from realtime snapshots 2026-07-02 21:16:39 +08:00
lingniu
04ca0d15d8 refactor(go): keep protocol realtime snapshots lightweight 2026-07-02 21:12:33 +08:00
lingniu
d1e28b9e64 refactor(go): avoid duplicate realtime protocol payloads 2026-07-02 21:09:07 +08:00
lingniu
da78a989b5 docs: record go topic configuration baseline 2026-07-02 21:05:46 +08:00
lingniu
bec6326103 refactor(go): make daily mileage vin only 2026-07-02 21:02:02 +08:00
lingniu
b23ae3410c refactor(go): remove realtime schema compatibility cleanup 2026-07-02 20:55:24 +08:00
lingniu
2ae9794dda refactor(go): default to go vehicle topics 2026-07-02 20:22:47 +08:00
lingniu
3248336bea docs: add production data plane inventory 2026-07-02 20:14:56 +08:00
lingniu
3c9678491a feat(go): bootstrap minimal identity schema 2026-07-02 20:06:38 +08:00
lingniu
593dc08870 refactor(go): simplify daily mileage storage 2026-07-02 20:01:03 +08:00
lingniu
abc1b9870a docs: record storage simplification deployment 2026-07-02 19:51:48 +08:00
lingniu
29cd044a57 refactor(go): remove duplicate raw fields json 2026-07-02 19:47:32 +08:00
lingniu
0bbe87e6fc refactor(go): stop duplicate mileage history writes 2026-07-02 19:42:53 +08:00
lingniu
be559c0f99 docs: define minimal vehicle storage contract 2026-07-02 19:38:24 +08:00
lingniu
8fd38eb87e docs: add vehicle ingest production runbook 2026-07-02 19:32:52 +08:00
lingniu
df96ec8879 feat(go): expose nats bridge pending metrics 2026-07-02 19:27:36 +08:00
lingniu
1f4a23ff69 feat(go): expose kafka consumer lag metrics 2026-07-02 19:23:01 +08:00
lingniu
41ff0bcef4 feat(go): track gateway active connections 2026-07-02 19:16:25 +08:00
lingniu
266a74f225 docs: document go service observability endpoints 2026-07-02 19:12:17 +08:00
lingniu
896e841902 feat(go): expose runtime ingest metrics 2026-07-02 19:06:24 +08:00
lingniu
1dfd540700 feat(go): add platform principles and health probes 2026-07-02 18:53:02 +08:00
lingniu
0bfd1191cb feat: add realtime snapshot and location APIs 2026-07-02 18:41:23 +08:00
lingniu
44a6bee4ac refactor(go): trim realtime mysql business columns 2026-07-02 17:16:00 +08:00
lingniu
12c15896d0 refactor(go): key realtime mysql tables by vin 2026-07-02 16:47:29 +08:00
lingniu
5333ed34b7 refactor(go): keep realtime mysql tables core only 2026-07-02 16:42:40 +08:00
lingniu
75b36f4011 fix(go): backfill realtime plate from binding 2026-07-02 16:36:52 +08:00
lingniu
bc399d2819 feat(go): cache realtime locations in mysql 2026-07-02 16:31:58 +08:00
lingniu
d0289b1b48 fix(go): flatten jt808 additional fields 2026-07-02 16:23:03 +08:00
lingniu
5b2e1abcdd fix(go): skip empty frames in realtime snapshot 2026-07-02 16:11:58 +08:00
lingniu
dac6718ee4 feat(go): persist realtime snapshots to mysql 2026-07-02 16:06:29 +08:00
lingniu
f4335641db fix(go): merge realtime raw arrays by serial 2026-07-02 15:48:14 +08:00
lingniu
f3b4cbbbb2 docs: add go data flow diagram 2026-07-02 15:48:14 +08:00
lingniu
c27150cc15 feat(go): add nats kafka ingest bridge 2026-07-02 15:07:44 +08:00
lingniu
079bd76d38 chore(go): tune gateway async publish defaults 2026-07-02 14:22:11 +08:00
lingniu
5145d156bc feat(go): decouple gateway publish with async sink 2026-07-02 14:11:43 +08:00
lingniu
5a958616a3 fix(go): batch durable kafka replay 2026-07-02 14:01:50 +08:00
lingniu
c939cc6b0c fix(go): stream durable spool batches 2026-07-02 13:47:53 +08:00
lingniu
b27f909109 fix(go): normalize tdengine phone tags 2026-07-02 13:43:15 +08:00
lingniu
75e7c3fbe5 fix(go): batch durable spool replay 2026-07-02 13:29:49 +08:00
lingniu
bc9025d566 fix(go): protect durable spool replay during shutdown 2026-07-02 13:25:15 +08:00
lingniu
640f70636d fix(go): protect kafka message processing during shutdown 2026-07-02 13:20:46 +08:00
lingniu
a80eb2eab4 fix(go): decouple mqtt message publishing from shutdown context 2026-07-02 13:14:15 +08:00
lingniu
d10d5d75c3 fix(go): decouple frame publishing from shutdown context 2026-07-02 13:10:38 +08:00
lingniu
924903feb4 fix(go): require kafka write acknowledgements 2026-07-02 13:06:18 +08:00
lingniu
ba5a28e636 fix(go): replay raw before unified events 2026-07-02 12:18:37 +08:00
lingniu
7a9a6f441a feat(go): persist full raw payload chunks 2026-07-02 12:11:24 +08:00
lingniu
3a67f6e05f feat: cache protocol realtime raw 2026-07-02 11:19:46 +08:00
lingniu
e7d024405a fix: throttle jt808 registration touches 2026-07-02 10:24:44 +08:00
lingniu
a66f765e16 chore: keep go branch go-only 2026-07-02 09:58:44 +08:00
1095 changed files with 252766 additions and 61366 deletions

View File

@@ -1,5 +0,0 @@
.git
.idea
.worktrees
archive
**/.DS_Store

5
.gitignore vendored
View File

@@ -40,3 +40,8 @@ hs_err_pid*
replay_pid*
.worktrees/
# Local frontend caches and generated review artifacts
.vite/
outputs/
vehicle-data-platform/apps/web/public/app-config.js

View File

@@ -0,0 +1,47 @@
## final-review-fix (2026-07-08 15:41 CST)
### Changes
- Made realtime duplicate suppression source-aware in `go/vehicle-gateway/internal/stats/daily_metric.go` by keying the in-memory mileage cache with `vin + protocol + source_key + stat_date`, so identical mileage from different sources is retained while same-source repeats are still skipped.
- Added optional previous-baseline querying and per-source/day caching in `Writer`; realtime candidates now:
- become `NO_PREVIOUS_BASELINE` with reason `missing_previous_source` when the same-source previous-day candidate is absent,
- use previous-day `latest_total_mileage_km` as `first_total_mileage_km`,
- compute `daily_mileage_km` from previous-day baseline when present,
- become `INVALID_DELTA` with reason `outside_daily_range` when the delta is negative or above `maxSelectedDailyMileageKM`.
- Added generic stale-final cleanup to `ProjectDailyMileage` in `go/vehicle-gateway/internal/stats/source_mileage.go` so `vehicle_daily_mileage` is deleted when no candidate remains selected for the exact `(vin, stat_date, protocol)` target.
- Changed `BACKFILL_METHOD` default from `scan` to `last_diff` in `go/vehicle-gateway/cmd/stats-backfill/main.go`.
- Expanded regression coverage for source-aware realtime dedupe, realtime baseline/no-baseline behavior, projector cleanup SQL, stale-final cleanup through backfill, and the backfill method default.
### Test Results
- `cd /Users/lingniu/project/ai-coding/lingniu-vehicle-ingest/go/vehicle-gateway && go test ./internal/stats -run 'TestWriter|TestProjectDailyMileage|TestSource' -count=1`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats 0.435s`
- `cd /Users/lingniu/project/ai-coding/lingniu-vehicle-ingest/go/vehicle-gateway && go test ./cmd/stats-backfill ./internal/stats -count=1`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/stats-backfill 0.643s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats 0.279s`
- `cd /Users/lingniu/project/ai-coding/lingniu-vehicle-ingest/go/vehicle-gateway && go test ./... -count=1`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/capacity-check 0.419s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/gateway 0.397s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/history-writer 0.728s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/load-sim 0.698s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/nats-fast-writer 0.688s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/nats-kafka-bridge 0.683s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/realtime-api 0.574s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/stat-writer 0.534s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/stats-backfill 0.634s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/capacity 0.834s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope 0.738s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/eventbus 0.758s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/gateway 0.483s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/health 0.384s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/history 0.502s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/identity 0.631s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/loadsim 0.640s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics 0.626s`
- `? lingniu-vehicle-ingest/go/vehicle-gateway/internal/observability [no test files]`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/protocol/gb32960 0.622s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/protocol/jt808 0.371s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/protocol/yutongmqtt 0.384s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/realtime 0.496s`
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats 0.443s`
- `? lingniu-vehicle-ingest/go/vehicle-gateway/internal/topics [no test files]`

View File

@@ -0,0 +1,54 @@
# Task 1 Report: Source Metadata Schema And Identity Helpers
## Outcome
- Implemented Task 1 in `go/vehicle-gateway/internal/stats`.
- Added source metadata schema, identity helpers, and focused tests.
## RED Evidence
Initial focused test run failed as expected because the new helpers did not exist yet:
```bash
cd /Users/lingniu/project/ai-coding/lingniu-vehicle-ingest/go/vehicle-gateway
go test ./internal/stats -run 'TestNormalizeSourceIP|TestNewSourceIdentity|TestUpsertDataSource' -count=1
```
Result:
- `undefined: NormalizeSourceIP`
- `undefined: NewSourceIdentity`
- `undefined: SourceIdentity`
- `undefined: UpsertDataSource`
## GREEN Evidence
After implementation, the focused tests passed:
```bash
cd /Users/lingniu/project/ai-coding/lingniu-vehicle-ingest/go/vehicle-gateway
go test ./internal/stats -run 'TestNormalizeSourceIP|TestNewSourceIdentity|TestUpsertDataSource' -count=1
```
Result:
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats 0.927s`
Full package verification also passed:
```bash
cd /Users/lingniu/project/ai-coding/lingniu-vehicle-ingest/go/vehicle-gateway
go test ./internal/stats -count=1
```
Result:
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats 0.355s`
## Files Changed
- `go/vehicle-gateway/internal/stats/schema.go`
- `go/vehicle-gateway/internal/stats/source.go`
- `go/vehicle-gateway/internal/stats/source_test.go`
## Notes
- `DataSourceTableSQL` was added verbatim to `schema.go` for the future schema wiring step.
- `UpsertDataSource` follows the brief exactly, including the nil exec panic, empty-source short circuit, and SQL shape.
## Self-Review
- The implementation is tightly scoped to Task 1.
- The new tests cover IP normalization, identity construction, and the SQL shape for the upsert helper.
- No additional concerns at this stage.

View File

@@ -0,0 +1,62 @@
# Task 2 Report: Candidate Mileage Schema And Writer
## Outcome
- Implemented Task 2 in `go/vehicle-gateway/internal/stats`.
- Added candidate mileage schema, sample-to-candidate mapping, and the upsert writer.
## RED Evidence
The focused candidate tests failed before implementation because the new symbols did not exist:
```bash
cd /Users/lingniu/project/ai-coding/lingniu-vehicle-ingest/go/vehicle-gateway
go test ./internal/stats -run 'TestSourceKey|TestUpsertSourceMileage' -count=1
```
Result:
- `undefined: SourceKey`
- `undefined: SourceMileageSample`
- `undefined: QualityOK`
- `undefined: UpsertSourceMileage`
## GREEN Evidence
After implementation, the focused candidate tests passed:
```bash
cd /Users/lingniu/project/ai-coding/lingniu-vehicle-ingest/go/vehicle-gateway
go test ./internal/stats -run 'TestSourceKey|TestUpsertSourceMileage' -count=1
```
Result:
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats 0.556s`
Package verification also passed:
```bash
cd /Users/lingniu/project/ai-coding/lingniu-vehicle-ingest/go/vehicle-gateway
go test ./internal/stats -count=1
```
Result:
- `ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats 0.235s`
## Files Changed
- `go/vehicle-gateway/internal/stats/schema.go`
- `go/vehicle-gateway/internal/stats/daily_metric.go`
- `go/vehicle-gateway/internal/stats/source_mileage.go`
- `go/vehicle-gateway/internal/stats/source_mileage_test.go`
## Self-Review
- The candidate schema matches the brief's table shape and indexes.
- `SamplesFromEnvelope` now carries `EventTime` and `DeviceID`, which the new candidate mapping needs.
- The upsert SQL is focused on the candidate table and reuses the shared `Execer` interface.
- No unrelated stats files were modified.
## Review Fix Addendum
- Preserved manual `platform_name` values when the candidate upsert receives blank runtime input by switching to `COALESCE(NULLIF(TRIM(VALUES(platform_name)), ''), platform_name)`.
- Bootstrapped `vehicle_data_source`, `vehicle_daily_mileage_source`, and `vehicle_daily_mileage` before running alter statements in `Writer.EnsureSchema`.
- Added a defensive blank `SourceIP` guard in `UpsertSourceMileage` so malformed candidate rows are skipped instead of written.
- Removed the SQL comment that existed only to satisfy a string-match test and updated the focused assertions to check the real `daily_mileage_km` and `platform_name` SQL expressions.
- Verification run:
- `go test ./internal/stats -run 'TestUpsertSourceMileage|TestWriterEnsuresSchema' -count=1`
- `go test ./internal/stats -count=1`
- Both passed.

View File

@@ -0,0 +1,110 @@
# Task 3 Report: Final Mileage Election Projector
## Outcome
Implemented `ProjectDailyMileage` in `go/vehicle-gateway/internal/stats/source_mileage.go` and added focused projector coverage in `go/vehicle-gateway/internal/stats/source_mileage_test.go`.
## RED
Before implementation, the new projector test failed as expected because the function did not exist yet:
```text
internal/stats/source_mileage_test.go:87:9: undefined: ProjectDailyMileage
```
That confirmed the test was exercising the missing task-3 surface rather than passing accidentally.
## GREEN
After implementation:
- `go test ./internal/stats -run TestProjectDailyMileageSelectsCandidateAndMarksSource -count=1`
- `go test ./internal/stats -count=1`
Both passed.
## What Changed
- Added `maxSelectedDailyMileageKM = 1000`.
- Added `ProjectDailyMileage(...)` with three SQL steps:
- clear `is_selected` for the VIN/date/protocol slice,
- project the elected row from `vehicle_daily_mileage_source` into `vehicle_daily_mileage`,
- mark the winning source row as selected.
- Kept the candidate identity intact: `source_key` still carries protocol + phone/device + source IP, while the data-source join uses protocol + source IP.
- Added a focused test that asserts the clear/project/mark sequence and the key SQL fragments.
## Files Changed
- `go/vehicle-gateway/internal/stats/source_mileage.go`
- `go/vehicle-gateway/internal/stats/source_mileage_test.go`
## Self-Review
- The projector stays aligned with the existing stats schema and helper conventions.
- The SQL ordering and join logic preserve the intended source identity split; there is no IP-only simplification.
- I did not add an unrelated schema migration because the current repository state already contained the needed `vehicle_daily_mileage_source` and `vehicle_daily_mileage` columns/indexes for this projector.
- The only notable follow-up risk is that the final selection query relies on MySQL-style `INSERT ... SELECT ... LIMIT 1 ON DUPLICATE KEY UPDATE` behavior, which matches the rest of the stats package but should still be exercised against the target DB in an integration run.
## Commit
`aa12317b5` - `feat(stats): project elected mileage source`
## Fix (reviewer findings)
Applied the reviewer correction for the trusted source endpoint contract:
- In `go/vehicle-gateway/internal/stats/source_mileage.go`, `projectDailyMileageSQL` now writes `trusted_source_endpoint` from `vehicle_data_source.latest_source_endpoint` via `COALESCE(NULLIF(ds.latest_source_endpoint, ''), s.source_endpoint)` so latest connection metadata is authoritative with a simple fallback.
- Replaced hardcoded quality predicate with the existing constant using `s.quality_status = '` + QualityOK + `'` in `projectDailyMileageSQL`.
- Updated `go/vehicle-gateway/internal/stats/source_mileage_test.go` to assert both the endpoint expression and constant-based quality predicate.
Validation commands:
```bash
cd go/vehicle-gateway && go test ./internal/stats -run TestProjectDailyMileageSelectsCandidateAndMarksSource -count=1
# PASS: lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats
cd go/vehicle-gateway && go test ./internal/stats -count=1
# PASS: lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats
```
## Fix (reviewer findings #2)
Corrected the remaining `projectDailyMileageSQL` contract issue so `trusted_source_endpoint` uses only `vehicle_data_source.latest_source_endpoint`:
- In `go/vehicle-gateway/internal/stats/source_mileage.go`, changed the projection from:
`COALESCE(NULLIF(ds.latest_source_endpoint, ''), s.source_endpoint)`
to:
`ds.latest_source_endpoint`.
- In `go/vehicle-gateway/internal/stats/source_mileage_test.go`, extended `TestProjectDailyMileageSelectsCandidateAndMarksSource` assertions to:
- require `ds.latest_source_endpoint` is selected, and
- explicitly verify no fallback expression using `s.source_endpoint` remains.
This enforces the strict source-of-truth rule that endpoint data for final projection must come from latest connection metadata only, even when it is null/empty.
## Fix (reviewer findings #3)
Addressed stale candidate selection re-flagging by narrowing source marking to the same eligibility rules used by projection:
- In `go/vehicle-gateway/internal/stats/source_mileage.go`, `markSelectedSourceSQL` now:
- No longer joins `vehicle_daily_mileage` to discover the selected source.
- Selects the same single winning candidate directly from `vehicle_daily_mileage_source`, constrained by:
- VIN/date/protocol
- `quality_status = QualityOK`
- mileage bounds (`0..maxSelectedDailyMileageKM`)
- enabled source metadata
- same trust/sample/time ordering
- Marks only that candidate as `is_selected = 1`.
- In `go/vehicle-gateway/internal/stats/source_mileage_test.go`, `TestProjectDailyMileageSelectsCandidateAndMarksSource` now verifies:
- the mark query includes candidate filters and ordering, including `s2.quality_status`, `s2.daily_mileage_km`, and `COALESCE(ds.enabled, 1) = 1`.
- it no longer references `vehicle_daily_mileage` as a join source.
- it passes the selection window arg through to marking (`maxSelectedDailyMileageKM`) and uses seven bind args.
Validation commands:
```bash
cd go/vehicle-gateway && go test ./internal/stats -run TestProjectDailyMileage -count=1
# PASS: lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats
cd go/vehicle-gateway && go test ./internal/stats -count=1
# PASS: lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats
```

View File

@@ -0,0 +1,6 @@
Status: Completed Task 4.
Commit: 821ea8e39 (`feat(stats): write realtime mileage candidates`).
Test Summary: `go test ./internal/stats -run TestWriterAppendWritesSourceCandidateAndProjection -count=1` (PASS) and `go test ./internal/stats -count=1` (PASS).
Concerns: `Writer.Append` now always runs projector projection per sample event, even when source identity is unavailable; this preserves final row refresh behavior and matches existing schema flow.
Files changed: `go/vehicle-gateway/internal/stats/daily_metric.go`, `go/vehicle-gateway/internal/stats/daily_metric_test.go`.
Report path: `.superpowers/sdd/task-4-report.md`.

View File

@@ -0,0 +1,207 @@
# Task 5 Report: Backfill Candidate Rows And Final Projection
## Summary
Implemented Task 5 in `go/vehicle-gateway/cmd/stats-backfill` so backfill now writes source candidates into `vehicle_daily_mileage_source` and re-projects `vehicle_daily_mileage` through the shared projector instead of writing final rows directly.
## TDD Evidence
1. Added the failing regression in `go/vehicle-gateway/cmd/stats-backfill/main_test.go`:
- `TestDailySourceLastBuildsCandidateKeysBySourceIP`
- Updated the existing trusted-source test to the new `normalizedSourceKey(protocol, phone, deviceID, endpoint)` signature.
2. Verified the red state with:
```bash
go test ./cmd/stats-backfill -run 'TestChooseTrustedSource|TestDailySourceLastBuildsCandidateKeysBySourceIP' -count=1
```
Observed failure:
```text
too many arguments in call to normalizedSourceKey
have (string, string, string, string)
want (string, string)
```
3. Implemented the production changes.
4. Verified green with:
```bash
go test ./cmd/stats-backfill -run 'TestChooseTrustedSource|TestDailySourceLastBuildsCandidateKeysBySourceIP' -count=1
go test ./cmd/stats-backfill ./internal/stats -count=1
```
Both commands passed.
## Files Changed
- `go/vehicle-gateway/cmd/stats-backfill/main.go`
- `go/vehicle-gateway/cmd/stats-backfill/main_test.go`
## What Changed
### 1. Source-key normalization now matches runtime semantics
- Changed backfill `normalizedSourceKey` to call:
```go
stats.SourceKey(envelope.Protocol(protocol), phone, deviceID, stats.NormalizeSourceIP(endpoint))
```
- This means candidate keys now use:
- protocol
- phone or device ID
- normalized source IP
- Port changes on the same source endpoint no longer produce different source keys.
### 2. Scan-mode backfill now preserves per-source candidates
- Expanded raw-frame reads to include `phone`, `device_id`, and `source_endpoint`.
- Populated `envelope.FrameEnvelope` with those fields before calling `stats.SamplesFromEnvelope`.
- Changed scan-mode aggregation key from:
```text
vin + stat_date + protocol
```
to:
```text
vin + stat_date + protocol + source_key
```
- Extended `metricAgg` to carry:
- `SourceKey`
- `Phone`
- `DeviceID`
- `SourceEndpoint`
- `FirstEventTime`
- `LatestEventTime`
- `QualityStatus`
- `QualityReason`
This keeps scan-mode candidates source-separated instead of mixing sources together.
### 3. Backfill writes candidates + re-projects final rows
- Replaced the direct `vehicle_daily_mileage` batch upsert path.
- `writeAggregates` now, per aggregate:
1. Upserts `vehicle_data_source` using protocol + normalized source IP.
2. Upserts a `vehicle_daily_mileage_source` candidate row.
3. Calls `stats.ProjectDailyMileage` for the VIN/date/protocol.
- Quality handling:
- default `OK`
- `INVALID_DELTA` when daily delta is `< 0` or `> 1000`
- pre-set non-OK statuses from `last_diff` are preserved
### 4. `last_diff` now preserves every source candidate
- `buildLastDiffAggregates` no longer picks a single trusted source.
- It now creates one aggregate per current-source candidate using:
```text
vin + stat_date + protocol + source_key
```
- If a same-source previous-day baseline exists:
- `FirstKM = previous.TotalKM`
- `LatestKM = current.TotalKM`
- `QualityStatus = OK`
- `QualityReason = "same_source_previous_day"`
- If there is no same-source previous-day baseline:
- `FirstKM = current.TotalKM`
- `LatestKM = current.TotalKM`
- `QualityStatus = NO_PREVIOUS_BASELINE`
- `QualityReason = "missing_previous_source"`
This preserves candidates without projecting them by default, matching the projectors `quality_status = 'OK'` filter.
### 5. Reset now clears candidate rows too
- Updated `resetStats` to delete both:
- `vehicle_daily_mileage_source`
- `vehicle_daily_mileage`
within the requested date/protocol scope.
This prevents stale candidates from affecting a subsequent rebuild run.
## Test Results
```bash
go test ./cmd/stats-backfill -run 'TestChooseTrustedSource|TestDailySourceLastBuildsCandidateKeysBySourceIP' -count=1
ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/stats-backfill 0.572s
```
```bash
go test ./cmd/stats-backfill ./internal/stats -count=1
ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/stats-backfill 0.174s
ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats 0.557s
```
## Concerns / Notes
- `chooseTrustedSource` remains in place for its existing unit test, but the `last_diff` aggregation path no longer relies on it.
- `ProjectDailyMileage` still follows the shared runtime semantics: it selects only `quality_status = 'OK'` candidates and does not synthesize cross-source deltas.
## Review Fix: Stale Final Rows
### Bug Addressed
When all candidates for a backfill target `(vin, stat_date, protocol)` are non-qualifying (`NO_PREVIOUS_BASELINE` / `INVALID_DELTA`), `ProjectDailyMileage` runs but inserts no final row, leaving any prior `vehicle_daily_mileage` row in place. That violated the expectation that a run with no selected candidate should not keep a historical final row for that key.
### Fix Applied
- Added helper in `cmd/stats-backfill/main.go`:
- `clearBackfillFinalMileage(ctx, db, vin, statDate, protocol)`
- Executes `DELETE FROM vehicle_daily_mileage WHERE vin = ? AND stat_date = ? AND protocol = ?`
- Guarded against empty key values.
- Updated `writeAggregates` to call the helper once per exact `(vin, stat_date, protocol)` target before calling `ProjectDailyMileage`.
- Kept existing runtime writer behavior unchanged; only backfill projection sequencing changed.
### Tests Added
- `TestClearBackfillFinalMileageClearsExactKey` in `cmd/stats-backfill/main_test.go`
- Verifies the SQL targets exactly the intended triplet.
- `TestWriteAggregatesClearsFinalBeforeProjection` in `cmd/stats-backfill/main_test.go`
- Verifies backfill writes still occur and final-row clear is executed before projection.
### Updated Evidence
```bash
go test ./cmd/stats-backfill -run 'TestChooseTrustedSource|TestDailySourceLastBuildsCandidateKeysBySourceIP|Test.*Reset|Test.*Projection|Test.*Final' -count=1
ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/stats-backfill 0.447s
go test ./cmd/stats-backfill ./internal/stats -count=1
ok lingniu-vehicle-ingest/go/vehicle-gateway/cmd/stats-backfill 0.188s
ok lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats 0.547s
```
## Follow-up Fix: Non-Reset Backfill Idempotency
### Additional Bug
The prior fix only cleared `vehicle_daily_mileage` before projection. In non-reset backfills, stale `vehicle_daily_mileage_source` rows for the same `(vin, stat_date, protocol)` could still be selected by `ProjectDailyMileage`, causing old OK candidates to win against fresh backfill input.
### Additional Fix
- Added `clearBackfillTargetMileage` in `go/vehicle-gateway/cmd/stats-backfill/main.go` to delete both tables for the exact target key:
- `DELETE FROM vehicle_daily_mileage_source WHERE vin = ? AND stat_date = ? AND protocol = ?`
- `DELETE FROM vehicle_daily_mileage WHERE vin = ? AND stat_date = ? AND protocol = ?`
- Updated `writeAggregates` so this clear happens at most once per `(vin, stat_date, protocol)` and before the first candidate upsert for that target.
- Kept existing backfill runtime flow intact: still upserts sources/candidates and calls `ProjectDailyMileage` for each target aggregate.
### Tests Added/Updated
- `TestClearBackfillTargetMileageClearsExactKey`:
- Verifies both table deletes use the exact VIN/date/protocol tuple.
- `TestWriteAggregatesClearsTargetRowsBeforeStaleCandidateUpsert`:
- Verifies clear operations happen before candidate writes and projection, and that the stale-candidate flow is tested by using a non-qualifying candidate status.
### Verification
```bash
go test ./cmd/stats-backfill -run 'TestChooseTrustedSource|TestDailySourceLastBuildsCandidateKeysBySourceIP|Test.*Reset|Test.*Projection|Test.*Final|Test.*Candidate' -count=1
go test ./cmd/stats-backfill ./internal/stats -count=1
```

View File

@@ -1,113 +0,0 @@
# Changelog
本文件记录本仓库的显著变更。日期使用 `YYYY-MM-DD` 格式。
格式参考 [Keep a Changelog](https://keepachangelog.com/zh-CN/1.1.0/),版本遵循 [SemVer](https://semver.org/lang/zh-CN/)。
> 说明v0.1.0 为本仓库 git 初始化时的基线快照。下面列出的"主题改动"实际上都被压在
> `chore: initial import of lingniu-vehicle-ingest` 这一个 commit 里——这是 git
> 初始化前已经完成的工作,无法事后拆分。从下一次提交开始走"小步分主题提交"流程。
---
## [Unreleased]
### Added —— 原始报文冷存Raw Archive接入 EventSink 扇出
- 新增 `VehicleEvent.RawArchive` sealed 变体(`ingest-api`),携带一帧入站的原始字节 +
`command` / `infoType` 供按协议命令分片归档。
- `Dispatcher.dispatch` 在 interceptor **之前**按 `RawFrame.rawBytes()` 产出一条
`RawArchive` 事件,保证 dedup / rate-limit **不过滤** raw archive满足"每帧必留痕"目标。
- 新增 `ArchiveEventSink implements EventSink``sink-archive`
- `accepts``VehicleEvent.RawArchive`
- 写入 key `yyyy/MM/dd/<source>/<vin>/<eventId>.bin`UTC 日期分片)
- 同 eventId 重复写按 put 语义覆盖
- 失败 completeExceptionallyEventBus 打 WARN
- AutoConfig 在 `ArchiveStore` bean 存在时自动装配
- `KafkaEventSink.accepts(RawArchive)` 返回 `false`:本期 Kafka 不投递原始字节,
只落 ArchiveStore。未来需要 Kafka 回填 URI 时再开(`TopicRouter` / `EnvelopeMapper`
已补好 switch 分支)。
- 覆盖测试 `ArchiveEventSinkTest` —— accepts 过滤 / 稳定 key 写入 / 幂等覆盖 3 case。
### Changed —— GB/T 32960 Body Parser 单块异常隔离
- `Gb32960BodyParser` 主循环对单信息块的 parser 异常不再放弃整帧:
- 固定长度块(`fixedLength ≥ 0`):回滚 reader → 按 `fixedLen` 截取字节兜成
`InfoBlock.Raw``continue` 继续解析后续块;
- 变长块(`fixedLength == -1`)或剩余字节不足:剩余字节全部兜成 `Raw` 后终止循环;
- `parser` 契约违反(声明 `fixedLen=X` 但实际读了 `Y`)仍抛 `DecodeException`
以暴露 parser bug不走 lenient 兜底,避免静默误解析长期误判;
- 只捕获 `DecodeException` / `BufferUnderflowException` / `IndexOutOfBoundsException`
`RuntimeException` 继续向上抛,保留 bug 可见性。
- 新增配置 `lingniu.ingest.gb32960.parse.lenient-block-failure`(默认 `true`
需要回退严格模式时设为 `false`
- `InfoBlock.Raw` record 结构**未变**,下游 `Gb32960EventMapper``findBlock(Class)`
类型匹配,新增 Raw 不影响业务事件映射。
- 新增测试 `Gb32960BodyParserIsolationTest` —— 3 种隔离场景 + 严格模式回退,共 4 case。
---
## [0.1.0] — 2026-04-15
初始基线。多模块 Spring Boot 车辆遥测接入服务。
### Added
- **GB/T 32960.3 接入**`protocol-gb32960`Netty server、帧分帧器、消息解码器、
V2016/V2025 双版本信息体 parser、平台登入鉴权、VIN 白名单、读空闲告警。
- **JT/T 808 / JT/T 1078 / JSATL12 接入**`protocol-jt808` / `protocol-jt1078` /
`protocol-jsatl12`)。
- **MQTT 入站**`inbound-mqtt`)。
- **Disruptor 事件总线**`ingest-core`):高吞吐解耦解码与下游 sink。
- **Sink**:本地归档(`sink-archive`、Kafka`sink-kafka`,含 protobuf envelope
- **会话状态**`session-core`)。
- **可观测性**Micrometer/Actuator 装配(`observability`)。
- **终端控制命令网关**`command-gateway`)。
### Removed
- 旧文件型事件索引模块和 DuckDB 依赖;历史查询统一收敛到 TDengine 与 `archive://...` 原始报文引用。
### Fixed —— GB/T 32960 应答帧时间字段
- `Gb32960MessageDecoder``VEHICLE_LOGIN` / `VEHICLE_LOGOUT` /
`PLATFORM_LOGIN` / `PLATFORM_LOGOUT` 命令预解析 body 首 6B 采集时间,写入
`header.eventTime`。修复前 handler 回 ACK 时落到 `Instant.now()`,导致应答时间与原
报文时间不一致,违反 §6.3.2 应答规则。
### Added —— Netty 读空闲告警
- 配置项 `lingniu.ingest.gb32960.idle-read-seconds`(默认 60s0 禁用)。
- pipeline 注入 `IdleStateHandler(idleReadSeconds, 0, 0)`,仅打 WARN 日志不断链,
用于排查"对端登入成功后长时间不推数据"场景。
### Changed —— GB32960 信息体 parser 包结构重构
-`codec/parser/` 下的 16 个共用类按版本严格拆分到 `parser/v2016/`10 类)
`parser/v2025/`11 类)两个独立子包。
- 取消 `Vehicle/DriveMotor/Engine/Position``v2016()/v2025()` 工厂方法,每个
版本独立类,零代码共用。
- `Alarm` 解析中跨版本静态调用 `readFaultList` 也已改为各版本类内部私有方法。
- `Gb32960AutoConfiguration` 同步更新所有 import 与 `@Bean` 工厂。
### Removed —— 违反 2016 标准的 typeCode 注册
- 经核对 GB/T 32960.3-2016 附录 B 表 B.3`0x30~0x7F` 整段为预留区,
**2016 标准里没有 0x30 / 0x31 / 0x32 任何字段定义**
- 删除 `FuelCellStackV2016BlockParser`(曾把 V2025 字段布局硬套到 V2016 帧的 0x30 段,
解出来的电压/电流远超规范上限,是非物理值)。
- `Gb32960AutoConfiguration` 移除 `gb32960V2016FuelCellStack` bean。
- `Gb32960DecoderGoldenTest` 中 749B 真实生产帧的断言由
"`isEmpty()`(不应有 Raw 兜底)"翻转为 "`isPresent()`peer 越界使用 0x30+ 预留
typeCode应有 Raw 兜底)"。这才是符合 2016 国标的正确预期。
### Diagnostics —— BodyParser hex dump 增强
- `Gb32960BodyParser` 在遇到未知 typeCode 触发 unknown 路径时,额外打一条 WARN
内容为整段 info-block 区域的 16 字节/行 hex dump并在 `typeStartPos` 字节后插入
`<` 标记,便于人工反查上一个块的字段对齐。
### Notes —— 待办
- peer 在 V2016 帧(`2323` 起始)里下发 0x30/0x31/0x32 是越界使用 2016 预留区段。
建议推动对端:
1. 切换到 V2025 帧头(`2424`
2. 或迁移到 `0x80~0xFE` 用户自定义区段;
3. 或提供私有协议文档,再决定是否实现 vendor-specific parser。
- `Gb32960BodyParser` 当前对 unknown 块的策略是"吞掉剩余所有字节为单个 Raw 后终止
循环",未做单块异常隔离。后续可改为对每个 parser.parse 调用加 try/catch +
position 回滚,实现单块失败不拖垮整帧。
- `platformAuthorizer.authenticate` 是 EventLoop 同步调用,存在 worker 阻塞风险,
建议挪到独立业务线程池(与 0x30 议题无关,但属于同模块潜在风险)。
[0.1.0]: #010--2026-04-15

View File

@@ -1,105 +0,0 @@
# 架构决策记录ADR 汇总)
> 本文件记录 lingniu-vehicle-ingest v2 的关键架构决策,后续每次重大调整追加条目(不删除旧条目)。
## ADR-001 消息通道Kafka
- **Status**: Accepted
- **Context**: 需要高吞吐、严格单车有序、成熟生态
- **Decision**: 使用 Kafka分区 key = VIN
- **Consequences**: 下游消费者需使用消费组 + 分区内顺序消费;生产链路只支持 Kafka。
## ADR-002 线上消息格式Protobuf
- **Status**: Accepted
- **Context**: 需要向前兼容、体积小、性能好
- **Decision**: 线上用 Protobuf调试/诊断保留 JSON 序列化能力
- **Consequences**: 需维护 `.proto` schema`ingest-api` 模块负责生成 Java stub
## ADR-003 command-gateway 独立模块
- **Status**: Accepted
- **Context**: 下行命令链路不应与接入进程耦合
- **Decision**: 拆出独立 `command-gateway` 模块,复用 `session-core`;本次重构同步交付
- **Consequences**: HTTP 对外接口(原 JT808Controller / JT1078Controller迁移到此模块
## ADR-004 信达 PushRemoved
- **Status**: Superseded by ADR-011
- **Context**: 旧实现硬编码凭证、无重连、0300/0401 空实现
- **Decision**: 删除信达 Push 源码、Maven profile、私仓依赖和部署入口。
- **Consequences**: 信达 Push 后续已从源码和构建面删除,优化优先投向 GB32960、JT808、宇通 MQTT、历史和统计链路。
## ADR-005 JT1078 / JSATL12 进第一批
- **Status**: Superseded by ADR-012
- **Decision**: 第一批迁移即覆盖 JT1078 信令 + JSATL12 报警附件上传
- **Consequences**: Phase 2 工期相应拉长JSATL12 需要对象存储后端(本地 / S3 / OSS
## ADR-006 部署形态GB32960 三应用拆分
- **Status**: Superseded by ADR-011
- **Context**: 原 `bootstrap-all` 一体化启动已被 GB32960 拆分部署替代
- **Decision**: 当时生产默认部署 `gb32960-ingest-app``vehicle-history-app``vehicle-analytics-app`;旧 `bootstrap-all` 模块删除
- **Consequences**: 该三应用边界已由 ADR-011 的三协议接入 + 历史 + 统计生产面取代;协议接入、历史查询、统计消费仍保持独立发布和回滚。
## ADR-007 Java 25 + 虚拟线程 + Disruptor
- **Status**: Accepted
- **Decision**:
- Netty EventLoop 只做解码与 RingBuffer 投递,严禁阻塞
- 业务 Handler 走虚拟线程(`Thread.ofVirtual()`
- 协议间背压通过 Disruptor RingBuffer 表达
- 禁用 `synchronized`,使用 `ReentrantLock` 避免虚拟线程 pinning
- **Consequences**: 需开启 `jdk.VirtualThreadPinned` JFR 监控
## ADR-008 协议接入应用不直接写业务库
- **Status**: Accepted
- **Decision**: GB32960、JT808、Yutong MQTT 等协议接入应用只负责收、解析、校验、规整和投递 Kafka不直接写业务表。
- **Consequences**: 业务落库由独立 Kafka 消费应用承接:`vehicle-history-app` 写入 TDengine 历史与 RAW JSON`vehicle-analytics-app` 写入 MySQL `vehicle_stat_metric`
## ADR-009 构建工具Maven
- **Status**: Accepted
- **Decision**: 沿用 Maven统一 BOM 管理版本Spotless 格式化ArchUnit 守护分层
## ADR-010 框架Spring Boot 3.5
- **Status**: Accepted
- **Decision**: Spring Boot 3.5.x支持 JDK 25AutoConfiguration 用于按需启停
## ADR-011 默认生产面:三协议接入 + 历史 + 统计
- **Status**: Accepted
- **Context**: 生产接入范围已经从 GB32960 三应用拆分,扩展为三条活跃接入链路和两个消费应用;信达 Push 已废弃并删除,不应再出现在构建、部署或优化目标里。
- **Decision**:
1. 默认生产面包含 GB32960、JT808、Yutong MQTT、vehicle-history-app、vehicle-analytics-app。
2. Xinda Push 源码、Maven profile、Woodpecker 镜像发布和历史消费绑定全部移除。
3. `command-gateway` 和 JT1078 继续作为可选能力,通过 `optional-command-gateway` profile 显式启用。
- **Consequences**: 默认构建和部署保持精简;后续性能、可靠性、字段解析和存储优化都服务三条活跃协议链路。
## ADR-012 JSATL12 附件上传Optional only
- **Status**: Accepted
- **Context**: 当前默认生产面只部署 GB32960、JT808、Yutong MQTT、vehicle-history-app、vehicle-analytics-appJSATL12 附件上传没有独立生产 app也不在 Portainer 和 Woodpecker 活跃镜像列表中。
- **Decision**:
1. `protocol-jsatl12` 不进入默认 Maven reactor。
2. JSATL12 仅保留在 `optional-attachments` profile需要附件上传能力时显式构建。
3. 默认优化和验证优先覆盖活跃接入链路,附件上传能力保持源码可用但不增加默认构建面。
- **Consequences**: 默认构建更轻,生产部署边界更清晰;附件上传相关测试需要通过 `-Poptional-attachments` 显式运行。
## ADR-013 最新状态服务Optional only
- **Status**: Accepted
- **Context**: Redis 最新状态查询是独立消费能力,但当前默认 Portainer 部署和 Woodpecker 镜像发布只包含三条接入链路、history 和 analytics`vehicle-state-service` 没有独立 app也不应增加默认构建面。
- **Decision**:
1. `vehicle-state-service` 不进入默认 Maven reactor。
2. vehicle-state-service 仅保留在 `optional-latest-state` profile需要 Redis 最新状态查询能力时显式构建。
3. 默认优化和验证优先覆盖活跃接入、TDengine 历史和 MySQL 指标链路。
- **Consequences**: 默认构建和部署边界继续收窄;最新状态能力保持源码可用,但其测试需要通过 `-Poptional-latest-state` 显式运行。
## ADR-014 raw-archive-store 原型已删除
- **Status**: Accepted
- **Context**: 默认生产 raw bytes 写入由 `sink-archive` 负责,历史查询通过 TDengine `raw_frames``archive://...` 引用追溯;`raw-archive-store` 只是未部署的独立读写原型,继续保留会造成 raw archive 路径歧义。
- **Decision**:
1. 删除 `raw-archive-store` 模块和对应 optional profile。
2. 默认生产链路只保留 `sink-archive`、Kafka raw topic、TDengine raw_frames 这条 raw 路径。
3. 如需新的 archive store 能力,先以生产链路需求重新设计,不恢复旧原型。
- **Consequences**: 默认构建面和可选构建面继续收窄raw bytes 写入职责集中在 `sink-archive`
## ADR-015 文件型事件索引Removed
- **Status**: Accepted
- **Context**: 默认历史查询已收敛到 TDengine `raw_frames``vehicle_locations` 和按需解码;旧文件型索引会增加一套无生产部署的查询和依赖边界。
- **Decision**:
1. 删除旧文件型事件索引模块和 profile。
2. 父 POM 不再管理旧索引驱动依赖。
3. 历史查询和 RAW 回放统一以 TDengine + `archive://...` 引用为准。
- **Consequences**: 默认构建面和可选构建面都更小;需要历史查询时只维护 TDengine 一条路径。

View File

@@ -1,19 +0,0 @@
FROM eclipse-temurin:25-jre
ARG APP_NAME
ARG APP_VERSION
ENV APP_NAME=${APP_NAME}
ENV APP_VERSION=${APP_VERSION}
ENV JAVA_OPTS="--sun-misc-unsafe-memory-access=allow -Dio.netty.transport.noNative=true"
RUN apt-get update \
&& apt-get install -y --no-install-recommends curl \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app
COPY modules/apps/${APP_NAME}/target/${APP_NAME}.jar /app/app.jar
EXPOSE 20100 20200 20310 20400 20500 32960 808
ENTRYPOINT ["sh", "-c", "exec java $JAVA_OPTS -jar /app/app.jar"]

104
README.md
View File

@@ -1,104 +0,0 @@
# lingniu-vehicle-ingest
> 羚牛车辆数据接入平台 v2
## 设计目标
- **协议接入统一抽象**GB/T 32960、JT/T 808、Yutong MQTT 是默认生产接入JT/T 1078、JSATL12 作为显式 profile 的可选能力;信达 Push 已废弃并删除源码
- **原子能力化**:每个协议一个独立 Maven 模块 + 独立 AutoConfiguration + 独立配置开关
- **业务 / 实时彻底解耦**:协议接入应用只负责收、解析、校验、规整和投递 Kafka历史与统计由独立 Kafka 消费应用落库
- **高并发低延迟**Netty + Disruptor + Java 25 虚拟线程;目标单节点 ≥ 5 万 msg/sP99 < 50 ms
- **可观测、可回放、可灰度**:统一 traceId、Envelope、幂等键、DLQ、原始报文冷存
## 技术栈
| 类别 | 选型 |
|---|---|
| 语言 | Java **25** |
| 框架 | Spring Boot 3.5.3 |
| 网络 | Netty 4.2.9.Final |
| 并发 | Disruptor 4 + 虚拟线程 |
| 事件流 | **Kafka**唯一生产消息通道vin 分区,保证单车有序) |
| 线上消息格式 | **Protobuf**,调试走 JSON |
| MQTT 客户端 | Eclipse Paho 1.2.5 |
| 会话索引 | Redis |
| 熔断/限流 | Resilience4j 2 |
| 可观测 | Micrometer + Prometheus |
| 冷存 | 本地文件系统 |
| 构建 | Maven |
## 模块划分
```
lingniu-vehicle-ingest/
├── modules/
│ ├── core/
│ │ ├── ingest-api/ SPI + sealed 领域事件 + 注解
│ │ ├── ingest-codec-common/ 公共编解码工具BCD/CRC/BCC/bit utils
│ │ ├── ingest-core/ Pipeline / Dispatcher / Disruptor / Session 桥
│ │ ├── session-core/ 设备会话 + 鉴权 + Token + Redis SessionStore
│ │ ├── vehicle-identity/ 跨协议车辆身份解析 + MySQL 外部标识绑定
│ │ └── observability/ metrics / health
│ ├── protocols/
│ │ ├── protocol-gb32960/ GB/T 32960
│ │ ├── protocol-jt808/ JT/T 808统一身份映射 + 事件 metadata 内部 VIN + 注册/鉴权/心跳/注销/位置/批量位置/参数/属性/媒体/透传/未知上行和坏帧兜底/断链清理会话/下行分包)
│ │ ├── protocol-jt1078/ JT/T 1078optional-command-gateway profile808 信令按需桥接 + 常用下行信令编码 + TCP/UDP RTP 媒体流分段归档 + 事件 metadata 内部 VIN + 坏 RTP 统一 RawArchive/Passthrough + 归档失败兜底 + SIM 身份映射)
│ │ └── protocol-jsatl12/ 苏标主动安全报警附件optional-attachments profile
│ ├── inbound/
│ │ └── inbound-mqtt/ MQTT 接入endpoint 生命周期 + profile 注册扩展 + 统一身份映射 + PEM 双向 TLS + 未知 profile/解析失败/profile异常/连接订阅失败兜底 + 统一 UNKNOWN 身份 metadata
│ ├── sinks/
│ │ ├── sink-kafka/ Kafka producer + Protobuf Envelope
│ │ └── sink-archive/ 原始报文冷存
│ ├── services/
│ │ ├── event-history-service/ Kafka 全字段事件消费 + 历史查询
│ │ ├── vehicle-state-service/ Kafka 全字段事件消费 + Redis 热状态查询optional-latest-state profile
│ │ └── vehicle-stat-service/ Kafka 全字段事件消费 + 可配置日统计
│ ├── testing/
│ │ └── vehicle-identity-test-support/ 协议测试共享身份解析夹具
│ └── apps/
│ ├── command-gateway/ 可选 HTTP → 设备下行命令optional-command-gateway profile
│ ├── gb32960-ingest-app/ GB32960 TCP 接入 + Kafka 投递
│ ├── jt808-ingest-app/ JT808 TCP 接入 + Kafka 投递
│ ├── yutong-mqtt-app/ 宇通 MQTT 接入 + Kafka 投递
│ ├── vehicle-history-app/ TDengine 历史查询 + RAW JSON
│ └── vehicle-analytics-app/ JT808 每日里程指标消费
├── docs/ 架构文档、模块图、实施计划
└── reference/ 参考资料
```
## 快速开始
```bash
# 要求JDK 25, Maven 3.9+
mvn -v
mvn -pl :gb32960-ingest-app,:jt808-ingest-app,:yutong-mqtt-app,:vehicle-history-app,:vehicle-analytics-app -am package -Dmaven.test.skip=true
```
生产服务只在 ECS/Portainer 上运行本机只用于源码检查、Maven 构建和单元测试。部署、健康检查和真实流量验证步骤见 `docs/operations/gb32960-service-split-runbook.md``docs/operations/vehicle-ingest-tdengine-verification.md`
信达 Push 已废弃并删除源码,不参与 Maven reactor、CI、Portainer 部署和历史消费链路。
JSATL12 附件上传不在默认 Maven reactor 中;需要时使用 `-Poptional-attachments` 构建。
Command Gateway/JT1078 下行与音视频信令不在默认 Maven reactor 中;需要时使用 `-Poptional-command-gateway` 构建。
Redis 最新状态查询不在默认 Maven reactor 中;需要时使用 `-Poptional-latest-state` 构建 `vehicle-state-service`
## 迁移说明
本项目是 `lingniu-vehicle-data-reception` 的 v2 重构,采用 strangler fig 渐进式迁移,旧项目保留只读参考。迁移路径与决策参见 `../REFRACTOR_PLAN.md`
## 架构文档
- 目标架构:`docs/target-architecture.md`
- 内部字段模型:`docs/vehicle-telemetry-internal-fields.md`
- 模块与数据流:`docs/module-data-flow.html`
## 核心原则
1. **接入与业务落库分离**:协议接入应用只负责收、解析、校验、规整和投递 Kafka`vehicle-history-app` 写入 TDengine 历史与 RAW JSON`vehicle-analytics-app` 将 JT808 `daily_mileage_km` 写入 MySQL `vehicle_stat_metric`
2. **Kafka 唯一消息通道**:生产链路只支持 Kafka接入、历史、统计统一通过 Kafka Envelope 解耦
3. **协议即插拔**:每个 `protocol-*` / `inbound-*` 都可独立开关(`lingniu.ingest.<name>.enabled`
4. **顺序保证**:同一 VIN 严格有序Disruptor hash + Kafka 分区 key
5. **幂等消费**Envelope 带 `eventId`,下游去重
6. **原始可回放**:协议接入会把 `rawArchiveKey/rawArchiveUri` 追加到事件 metadataKafka Envelope、TDengine `raw_frames` 和导出都使用 `archive://...` 逻辑 URI 追溯原始 bytes实际文件由 `sink-archive` 管理

View File

@@ -0,0 +1,10 @@
FROM python:3.11-slim
RUN pip install --no-cache-dir \
--index-url https://mirrors.aliyun.com/pypi/simple \
ddddocr==1.6.1
WORKDIR /app
COPY ocr.py /app/ocr.py
USER nobody
ENTRYPOINT ["python", "/app/ocr.py"]

22
deploy/feichi-ocr/ocr.py Normal file
View File

@@ -0,0 +1,22 @@
import re
import sys
import ddddocr
def main() -> int:
image = sys.stdin.buffer.read()
if not image:
print("captcha image is empty", file=sys.stderr)
return 2
result = ddddocr.DdddOcr(show_ad=False).classification(image)
code = re.sub(r"[^0-9A-Za-z]", "", result).upper()
if not re.fullmatch(r"[0-9A-Z]{4}", code):
print(f"invalid OCR result: {code!r}", file=sys.stderr)
return 3
print(code)
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,12 @@
services:
nats:
image: nats:2.10-alpine
container_name: lingniu-nats
restart: always
command: ["-c", "/etc/nats/nats-server.conf"]
ports:
- "172.17.111.56:4222:4222"
- "172.17.111.56:8222:8222"
volumes:
- /opt/lingniu-nats/conf/nats-server.conf:/etc/nats/nats-server.conf:ro
- /opt/lingniu-nats/data:/data

View File

@@ -0,0 +1,13 @@
server_name: lingniu-vehicle-nats
port: 4222
http_port: 8222
jetstream {
store_dir: "/data/jetstream"
max_mem_store: 1G
max_file_store: 200G
}
authorization {
timeout: 2
}

View File

@@ -1,117 +0,0 @@
version: "3.8"
# Go sidecar stack for the vehicle ingest redesign.
# It is intentionally separate from deploy/portainer/docker-compose.yml so the
# current Java production stack can keep running until Go evidence is captured.
x-common-env: &common-env
TZ: Asia/Shanghai
KAFKA_BROKERS: ${KAFKA_BROKERS:-172.17.111.56:9092}
KAFKA_TOPIC_GB32960_RAW: ${KAFKA_TOPIC_GB32960_RAW:-vehicle.raw.gb32960.v1}
KAFKA_TOPIC_JT808_RAW: ${KAFKA_TOPIC_JT808_RAW:-vehicle.raw.jt808.v1}
KAFKA_TOPIC_YUTONG_MQTT_RAW: ${KAFKA_TOPIC_YUTONG_MQTT_RAW:-vehicle.raw.yutong-mqtt.v1}
KAFKA_TOPIC_UNIFIED: ${KAFKA_TOPIC_UNIFIED:-vehicle.event.unified.v1}
x-restart: &restart-policy
restart: unless-stopped
networks:
- vehicle-ingest
logging:
driver: json-file
options:
max-size: "100m"
max-file: "5"
services:
go-vehicle-gateway:
<<: *restart-policy
image: ${LINGNIU_GO_IMAGE_REGISTRY:-crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com}/${LINGNIU_GO_IMAGE_NAMESPACE:-oneos}/vehicle-gateway-go:${LINGNIU_GO_IMAGE_VERSION:?set LINGNIU_GO_IMAGE_VERSION}
container_name: go-vehicle-gateway
command: ["/app/gateway"]
mem_limit: ${GO_GATEWAY_MEM_LIMIT:-512m}
environment:
<<: *common-env
GB32960_TCP_ADDR: ":32960"
JT808_TCP_ADDR: ":808"
TCP_READ_BUFFER_BYTES: ${TCP_READ_BUFFER_BYTES:-65536}
TCP_IDLE_TIMEOUT_SECONDS: ${TCP_IDLE_TIMEOUT_SECONDS:-180}
TCP_MAX_CONNECTIONS: ${TCP_MAX_CONNECTIONS:-20000}
YUTONG_MQTT_ENABLED: ${YUTONG_MQTT_ENABLED:-false}
YUTONG_MQTT_ENDPOINT: ${YUTONG_MQTT_ENDPOINT:-yutong}
YUTONG_MQTT_URI: ${YUTONG_MQTT_URI:-}
YUTONG_MQTT_TOPICS: ${YUTONG_MQTT_TOPICS:-/ytforward/shln/+}
YUTONG_MQTT_QOS: ${YUTONG_MQTT_QOS:-2}
YUTONG_MQTT_CLIENT_ID: ${YUTONG_MQTT_CLIENT_ID:-lingniu-go-yutong-mqtt}
YUTONG_MQTT_USERNAME: ${YUTONG_MQTT_USERNAME:-}
YUTONG_MQTT_PASSWORD: ${YUTONG_MQTT_PASSWORD:-}
YUTONG_MQTT_CLEAN_SESSION: ${YUTONG_MQTT_CLEAN_SESSION:-false}
YUTONG_MQTT_KEEP_ALIVE_SECONDS: ${YUTONG_MQTT_KEEP_ALIVE_SECONDS:-20}
YUTONG_MQTT_CONNECTION_TIMEOUT_SECONDS: ${YUTONG_MQTT_CONNECTION_TIMEOUT_SECONDS:-10}
YUTONG_MQTT_TLS_CA_PEM: ${YUTONG_MQTT_TLS_CA_PEM:-}
YUTONG_MQTT_TLS_CLIENT_PEM: ${YUTONG_MQTT_TLS_CLIENT_PEM:-}
YUTONG_MQTT_TLS_CLIENT_KEY: ${YUTONG_MQTT_TLS_CLIENT_KEY:-}
YUTONG_MQTT_TLS_HOSTNAME_VERIFICATION_ENABLED: ${YUTONG_MQTT_TLS_HOSTNAME_VERIFICATION_ENABLED:-true}
IDENTITY_MYSQL_DSN: ${IDENTITY_MYSQL_DSN:-}
VEHICLE_IDENTITY_TABLE: ${VEHICLE_IDENTITY_TABLE:-vehicle_identity_binding}
ports:
- "${GO_GB32960_TCP_PORT:-32960}:32960"
- "${GO_JT808_TCP_PORT:-808}:808"
volumes:
- "${KAFKA_SPOOL_HOST_DIR:-/opt/lingniu-go/spool/gateway}:${KAFKA_SPOOL_DIR:-/data/spool/gateway}"
- "${YUTONG_MQTT_CERT_HOST_DIR:-/opt/lingniuServices/certificate/yutong/vehicledatareception}:${YUTONG_MQTT_CERT_CONTAINER_DIR:-/opt/lingniuServices/certificate/yutong/vehicledatareception}:ro"
go-history-writer:
<<: *restart-policy
image: ${LINGNIU_GO_IMAGE_REGISTRY:-crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com}/${LINGNIU_GO_IMAGE_NAMESPACE:-oneos}/vehicle-gateway-go:${LINGNIU_GO_IMAGE_VERSION:?set LINGNIU_GO_IMAGE_VERSION}
container_name: go-history-writer
command: ["/app/history-writer"]
mem_limit: ${GO_HISTORY_WRITER_MEM_LIMIT:-768m}
environment:
<<: *common-env
KAFKA_GROUP: ${KAFKA_GROUP_GO_HISTORY:-go-history-writer}
KAFKA_TOPICS: ${KAFKA_TOPICS_GO_HISTORY:-vehicle.raw.gb32960.v1,vehicle.raw.jt808.v1,vehicle.raw.yutong-mqtt.v1}
TDENGINE_DRIVER: ${TDENGINE_DRIVER:-taosWS}
TDENGINE_DSN: ${TDENGINE_DSN:?set TDENGINE_DSN, example root:password@ws(172.17.111.57:6041)/lingniu_vehicle_ts}
TDENGINE_DATABASE: ${TDENGINE_DATABASE:-lingniu_vehicle_ts}
TDENGINE_ENSURE_SCHEMA: ${TDENGINE_ENSURE_SCHEMA:-true}
go-stat-writer:
<<: *restart-policy
image: ${LINGNIU_GO_IMAGE_REGISTRY:-crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com}/${LINGNIU_GO_IMAGE_NAMESPACE:-oneos}/vehicle-gateway-go:${LINGNIU_GO_IMAGE_VERSION:?set LINGNIU_GO_IMAGE_VERSION}
container_name: go-stat-writer
command: ["/app/stat-writer"]
mem_limit: ${GO_STAT_WRITER_MEM_LIMIT:-512m}
environment:
<<: *common-env
KAFKA_GROUP: ${KAFKA_GROUP_GO_STAT:-go-stat-writer}
KAFKA_TOPICS: ${KAFKA_TOPICS_GO_STAT:-vehicle.raw.gb32960.v1,vehicle.raw.jt808.v1}
MYSQL_DSN: ${MYSQL_DSN:?set MYSQL_DSN}
MYSQL_ENSURE_SCHEMA: ${MYSQL_ENSURE_SCHEMA:-true}
LOCAL_TZ: ${LOCAL_TZ:-Asia/Shanghai}
go-realtime-api:
<<: *restart-policy
image: ${LINGNIU_GO_IMAGE_REGISTRY:-crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com}/${LINGNIU_GO_IMAGE_NAMESPACE:-oneos}/vehicle-gateway-go:${LINGNIU_GO_IMAGE_VERSION:?set LINGNIU_GO_IMAGE_VERSION}
container_name: go-realtime-api
command: ["/app/realtime-api"]
mem_limit: ${GO_REALTIME_API_MEM_LIMIT:-512m}
environment:
<<: *common-env
HTTP_ADDR: ":20210"
KAFKA_GROUP: ${KAFKA_GROUP_GO_REALTIME:-go-realtime-api}
KAFKA_TOPICS: ${KAFKA_TOPICS_GO_REALTIME:-vehicle.event.unified.v1}
REDIS_ADDR: ${REDIS_ADDR:-r-bp1u741kij7e51i481.redis.rds.aliyuncs.com:6379}
REDIS_USERNAME: ${REDIS_USERNAME:-}
REDIS_PASSWORD: ${REDIS_PASSWORD:?set REDIS_PASSWORD}
REDIS_DB: ${REDIS_DB:-50}
ONLINE_TTL_SECONDS: ${ONLINE_TTL_SECONDS:-600}
MYSQL_DSN: ${MYSQL_DSN:-}
TDENGINE_DRIVER: ${TDENGINE_DRIVER:-taosWS}
TDENGINE_DSN: ${TDENGINE_DSN:-}
TDENGINE_DATABASE: ${TDENGINE_DATABASE:-lingniu_vehicle_ts}
ports:
- "${GO_REALTIME_HTTP_PORT:-20210}:20210"
networks:
vehicle-ingest:
name: vehicle-ingest

View File

@@ -1,216 +0,0 @@
version: "3.8"
# Default production stack: GB32960, JT808, Yutong MQTT, history, analytics.
# Legacy and optional services stay outside this compose file.
# Set LINGNIU_IMAGE_VERSION in Portainer Stack env, for example:
# main-0.1.0-SNAPSHOT-22468e7
# Override LINGNIU_IMAGE_REGISTRY / LINGNIU_IMAGE_NAMESPACE only when publishing to another registry.
x-common-env: &common-env
TZ: Asia/Shanghai
JAVA_OPTS: ${JAVA_OPTS:---sun-misc-unsafe-memory-access=allow -Dio.netty.transport.noNative=true -XX:+UseZGC -XX:MaxRAMPercentage=75}
NACOS_CONFIG_ENABLED: ${NACOS_CONFIG_ENABLED:-true}
NACOS_SERVER_ADDR: ${NACOS_SERVER_ADDR:-127.0.0.1:8848}
NACOS_NAMESPACE: ${NACOS_NAMESPACE:-}
NACOS_GROUP: ${NACOS_GROUP:-DEFAULT_GROUP}
NACOS_CONFIG_FILE_EXTENSION: ${NACOS_CONFIG_FILE_EXTENSION:-yml}
NACOS_REFRESH_ENABLED: ${NACOS_REFRESH_ENABLED:-true}
NACOS_USERNAME: ${NACOS_USERNAME:-}
NACOS_PASSWORD: ${NACOS_PASSWORD:-}
KAFKA_BROKERS: ${KAFKA_BROKERS:-172.17.111.56:9092}
x-identity-mysql-env: &identity-mysql-env
VEHICLE_IDENTITY_MYSQL_JDBC_URL: ${MYSQL_JDBC_URL:-}
VEHICLE_IDENTITY_MYSQL_USERNAME: ${MYSQL_USERNAME:-}
VEHICLE_IDENTITY_MYSQL_PASSWORD: ${MYSQL_PASSWORD:-}
x-redis-session-env: &redis-session-env
REDIS_HOST: ${REDIS_HOST:-r-bp1u741kij7e51i481.redis.rds.aliyuncs.com}
REDIS_PORT: ${REDIS_PORT:-6379}
REDIS_DATABASE: ${REDIS_DATABASE:-50}
REDIS_USERNAME: ${REDIS_USERNAME:-}
REDIS_PASSWORD: ${REDIS_PASSWORD:-}
x-readiness-healthcheck: &readiness-healthcheck
interval: 30s
timeout: 5s
retries: 3
start_period: 60s
services:
gb32960-ingest-app:
image: ${LINGNIU_IMAGE_REGISTRY:-crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com}/${LINGNIU_IMAGE_NAMESPACE:-oneos}/gb32960-ingest-app:${LINGNIU_IMAGE_VERSION:?set LINGNIU_IMAGE_VERSION}
container_name: gb32960-ingest-app
restart: unless-stopped
mem_limit: ${GB32960_MEM_LIMIT:-768m}
environment:
<<:
- *common-env
- *identity-mysql-env
- *redis-session-env
HTTP_PORT: 20100
GB32960_PORT: 32960
KAFKA_NODE_ID: ${KAFKA_NODE_ID:-gb32960-ingest-portainer}
KAFKA_TOPIC_GB32960_EVENT: ${KAFKA_TOPIC_GB32960_EVENT:-vehicle.event.gb32960.v1}
KAFKA_TOPIC_GB32960_RAW: ${KAFKA_TOPIC_GB32960_RAW:-vehicle.raw.gb32960.v1}
KAFKA_TOPIC_GB32960_DLQ: ${KAFKA_TOPIC_GB32960_DLQ:-vehicle.dlq.gb32960.v1}
GB32960_AUTH_ENABLED: ${GB32960_AUTH_ENABLED:-false}
GB32960_PLATFORM_USER_HYUNDAI: ${GB32960_PLATFORM_USER_HYUNDAI:-Hyundai}
GB32960_PLATFORM_PWD_HYUNDAI: ${GB32960_PLATFORM_PWD_HYUNDAI:-}
GB32960_PLATFORM_IP_HYUNDAI: ${GB32960_PLATFORM_IP_HYUNDAI:-}
GB32960_PLATFORM_USER_YUEJIN: ${GB32960_PLATFORM_USER_YUEJIN:-YueJin}
GB32960_PLATFORM_PWD_YUEJIN: ${GB32960_PLATFORM_PWD_YUEJIN:-}
GB32960_PLATFORM_IP_YUEJIN: ${GB32960_PLATFORM_IP_YUEJIN:-}
GB32960_TLS_ENABLED: ${GB32960_TLS_ENABLED:-false}
SINK_ARCHIVE_PATH: ${GB32960_ARCHIVE_PATH:-/data/archive}
ports:
- "${GB32960_HTTP_PORT:-20100}:20100"
- "${GB32960_TCP_PORT:-32960}:32960"
volumes:
- gb32960-ingest-data:/data
networks:
- vehicle-ingest
healthcheck:
<<: *readiness-healthcheck
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:20100/actuator/health/readiness >/dev/null"]
jt808-ingest-app:
image: ${LINGNIU_IMAGE_REGISTRY:-crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com}/${LINGNIU_IMAGE_NAMESPACE:-oneos}/jt808-ingest-app:${LINGNIU_IMAGE_VERSION:?set LINGNIU_IMAGE_VERSION}
container_name: jt808-ingest-app
restart: unless-stopped
mem_limit: ${JT808_MEM_LIMIT:-768m}
environment:
<<:
- *common-env
- *identity-mysql-env
- *redis-session-env
HTTP_PORT: 20400
JT808_PORT: 808
KAFKA_NODE_ID: ${KAFKA_NODE_ID_JT808:-jt808-ingest-portainer}
KAFKA_TOPIC_JT808_EVENT: ${KAFKA_TOPIC_JT808_EVENT:-vehicle.event.jt808.v1}
KAFKA_TOPIC_JT808_RAW: ${KAFKA_TOPIC_JT808_RAW:-vehicle.raw.jt808.v1}
KAFKA_TOPIC_JT808_DLQ: ${KAFKA_TOPIC_JT808_DLQ:-vehicle.dlq.jt808.v1}
SINK_ARCHIVE_PATH: ${JT808_ARCHIVE_PATH:-/data/archive}
ports:
- "${JT808_HTTP_PORT:-20400}:20400"
- "${JT808_TCP_PORT:-808}:808"
volumes:
- jt808-ingest-data:/data
networks:
- vehicle-ingest
healthcheck:
<<: *readiness-healthcheck
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:20400/actuator/health/readiness >/dev/null"]
yutong-mqtt-app:
image: ${LINGNIU_IMAGE_REGISTRY:-crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com}/${LINGNIU_IMAGE_NAMESPACE:-oneos}/yutong-mqtt-app:${LINGNIU_IMAGE_VERSION:?set LINGNIU_IMAGE_VERSION}
container_name: yutong-mqtt-app
restart: unless-stopped
mem_limit: ${YUTONG_MQTT_MEM_LIMIT:-512m}
environment:
<<:
- *common-env
- *identity-mysql-env
HTTP_PORT: 20500
YUTONG_MQTT_ENABLED: ${YUTONG_MQTT_ENABLED:-false}
YUTONG_MQTT_AUTO_STARTUP: ${YUTONG_MQTT_AUTO_STARTUP:-true}
KAFKA_NODE_ID: ${KAFKA_NODE_ID_MQTT:-yutong-mqtt-portainer}
KAFKA_TOPIC_YUTONG_MQTT_EVENT: ${KAFKA_TOPIC_YUTONG_MQTT_EVENT:-vehicle.event.mqtt-yutong.v1}
KAFKA_TOPIC_YUTONG_MQTT_RAW: ${KAFKA_TOPIC_YUTONG_MQTT_RAW:-vehicle.raw.mqtt-yutong.v1}
KAFKA_TOPIC_YUTONG_MQTT_DLQ: ${KAFKA_TOPIC_YUTONG_MQTT_DLQ:-vehicle.dlq.mqtt-yutong.v1}
YUTONG_MQTT_ENDPOINT_NAME: ${YUTONG_MQTT_ENDPOINT_NAME:-yutong}
YUTONG_MQTT_URI: ${YUTONG_MQTT_URI:-}
YUTONG_MQTT_TOPIC: "${YUTONG_MQTT_TOPIC:-#}"
YUTONG_MQTT_QOS: ${YUTONG_MQTT_QOS:-1}
YUTONG_MQTT_CLIENT_ID: ${YUTONG_MQTT_CLIENT_ID:-lingniu-yutong-mqtt}
YUTONG_MQTT_USERNAME: ${YUTONG_MQTT_USERNAME:-}
YUTONG_MQTT_PASSWORD: ${YUTONG_MQTT_PASSWORD:-}
YUTONG_MQTT_CLEAN_SESSION: ${YUTONG_MQTT_CLEAN_SESSION:-false}
YUTONG_MQTT_KEEP_ALIVE_SECONDS: ${YUTONG_MQTT_KEEP_ALIVE_SECONDS:-20}
YUTONG_MQTT_CONNECTION_TIMEOUT_SECONDS: ${YUTONG_MQTT_CONNECTION_TIMEOUT_SECONDS:-10}
YUTONG_MQTT_TLS_CA_PEM: ${YUTONG_MQTT_TLS_CA_PEM:-}
YUTONG_MQTT_TLS_CLIENT_PEM: ${YUTONG_MQTT_TLS_CLIENT_PEM:-}
YUTONG_MQTT_TLS_CLIENT_KEY: ${YUTONG_MQTT_TLS_CLIENT_KEY:-}
YUTONG_MQTT_TLS_HOSTNAME_VERIFICATION_ENABLED: ${YUTONG_MQTT_TLS_HOSTNAME_VERIFICATION_ENABLED:-true}
SINK_ARCHIVE_PATH: ${YUTONG_MQTT_ARCHIVE_PATH:-/data/archive}
ports:
- "${YUTONG_MQTT_HTTP_PORT:-20500}:20500"
volumes:
- yutong-mqtt-data:/data
networks:
- vehicle-ingest
healthcheck:
<<: *readiness-healthcheck
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:20500/actuator/health/readiness >/dev/null"]
vehicle-history-app:
image: ${LINGNIU_IMAGE_REGISTRY:-crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com}/${LINGNIU_IMAGE_NAMESPACE:-oneos}/vehicle-history-app:${LINGNIU_IMAGE_VERSION:?set LINGNIU_IMAGE_VERSION}
container_name: vehicle-history-app
restart: unless-stopped
mem_limit: ${VEHICLE_HISTORY_MEM_LIMIT:-1536m}
environment:
<<: *common-env
HTTP_PORT: 20200
KAFKA_CONSUMER_ENABLED: ${KAFKA_CONSUMER_ENABLED_HISTORY:-true}
KAFKA_CONSUMER_CLIENT_ID_PREFIX: ${KAFKA_CONSUMER_CLIENT_ID_PREFIX_HISTORY:-vehicle-history}
KAFKA_CONSUMER_MAX_POLL_RECORDS: ${KAFKA_CONSUMER_MAX_POLL_RECORDS_HISTORY:-2000}
KAFKA_CONSUMER_MAX_POLL_INTERVAL_MILLIS: ${KAFKA_CONSUMER_MAX_POLL_INTERVAL_MILLIS_HISTORY:-1800000}
KAFKA_TOPIC_GB32960_EVENT: ${KAFKA_TOPIC_GB32960_EVENT:-vehicle.event.gb32960.v1}
KAFKA_TOPIC_GB32960_RAW: ${KAFKA_TOPIC_GB32960_RAW:-vehicle.raw.gb32960.v1}
KAFKA_TOPIC_JT808_EVENT: ${KAFKA_TOPIC_JT808_EVENT:-vehicle.event.jt808.v1}
KAFKA_TOPIC_JT808_RAW: ${KAFKA_TOPIC_JT808_RAW:-vehicle.raw.jt808.v1}
KAFKA_TOPIC_YUTONG_MQTT_EVENT: ${KAFKA_TOPIC_YUTONG_MQTT_EVENT:-vehicle.event.mqtt-yutong.v1}
KAFKA_TOPIC_YUTONG_MQTT_RAW: ${KAFKA_TOPIC_YUTONG_MQTT_RAW:-vehicle.raw.mqtt-yutong.v1}
KAFKA_TOPIC_HISTORY_DLQ: ${KAFKA_TOPIC_HISTORY_DLQ:-vehicle.dlq.history.v1}
KAFKA_GROUP_HISTORY: ${KAFKA_GROUP_HISTORY:-vehicle-history}
TDENGINE_HISTORY_ENABLED: ${TDENGINE_HISTORY_ENABLED:-true}
TDENGINE_JDBC_URL: ${TDENGINE_JDBC_URL:-jdbc:TAOS-WS://172.17.111.57:6041/vehicle_ts}
TDENGINE_HISTORY_DATABASE: ${TDENGINE_HISTORY_DATABASE:-vehicle_ts}
TDENGINE_USERNAME: ${TDENGINE_USERNAME:-root}
TDENGINE_PASSWORD: ${TDENGINE_PASSWORD:-taosdata}
TDENGINE_CONNECTION_TIMEOUT_MILLIS: ${TDENGINE_CONNECTION_TIMEOUT_MILLIS:-30000}
TDENGINE_MIN_IDLE: ${TDENGINE_MIN_IDLE:-1}
TDENGINE_MAX_POOL_SIZE: ${TDENGINE_MAX_POOL_SIZE:-32}
ports:
- "${VEHICLE_HISTORY_HTTP_PORT:-20200}:20200"
networks:
- vehicle-ingest
healthcheck:
<<: *readiness-healthcheck
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:20200/actuator/health/readiness >/dev/null"]
vehicle-analytics-app:
image: ${LINGNIU_IMAGE_REGISTRY:-crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com}/${LINGNIU_IMAGE_NAMESPACE:-oneos}/vehicle-analytics-app:${LINGNIU_IMAGE_VERSION:?set LINGNIU_IMAGE_VERSION}
container_name: vehicle-analytics-app
restart: unless-stopped
mem_limit: ${VEHICLE_ANALYTICS_MEM_LIMIT:-1024m}
environment:
<<: *common-env
HTTP_PORT: 20310
KAFKA_CONSUMER_ENABLED: ${KAFKA_CONSUMER_ENABLED_ANALYTICS:-true}
KAFKA_CONSUMER_CLIENT_ID_PREFIX: ${KAFKA_CONSUMER_CLIENT_ID_PREFIX_ANALYTICS:-vehicle-analytics}
KAFKA_CONSUMER_MAX_POLL_INTERVAL_MILLIS: ${KAFKA_CONSUMER_MAX_POLL_INTERVAL_MILLIS_ANALYTICS:-900000}
KAFKA_TOPIC_JT808_EVENT: ${KAFKA_TOPIC_JT808_EVENT:-vehicle.event.jt808.v1}
KAFKA_TOPIC_JT808_DLQ: ${KAFKA_TOPIC_JT808_DLQ:-vehicle.dlq.jt808.v1}
KAFKA_GROUP_STAT: ${KAFKA_GROUP_STAT:-vehicle-stat}
VEHICLE_STAT_ENABLED: ${VEHICLE_STAT_ENABLED:-true}
VEHICLE_STAT_ZONE_ID: ${VEHICLE_STAT_ZONE_ID:-Asia/Shanghai}
VEHICLE_STAT_JT808_MILEAGE_ENABLED: ${VEHICLE_STAT_JT808_MILEAGE_ENABLED:-true}
MYSQL_JDBC_URL: ${MYSQL_JDBC_URL:-}
MYSQL_USERNAME: ${MYSQL_USERNAME:-}
MYSQL_PASSWORD: ${MYSQL_PASSWORD:-}
ports:
- "${VEHICLE_ANALYTICS_HTTP_PORT:-20310}:20310"
networks:
- vehicle-ingest
healthcheck:
<<: *readiness-healthcheck
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:20310/actuator/health/readiness >/dev/null"]
networks:
vehicle-ingest:
name: vehicle-ingest
volumes:
gb32960-ingest-data:
jt808-ingest-data:
yutong-mqtt-data:

View File

@@ -1,66 +0,0 @@
# Go Vehicle Gateway Native Deployment
本目录用于 ECS 原生部署。当前 goal 后续不再使用 Docker/Portainer 作为 Go 接入链路的部署方式。
## Runtime Layout
```text
/opt/lingniu-go-native/
current -> /opt/lingniu-go-native/releases/<git-short-sha>
releases/<git-short-sha>/
gateway
history-writer
stat-writer
realtime-api
env/
gateway.env
history-writer.env
stat-writer.env
realtime-api.env
spool/gateway/
```
## Services
| systemd unit | Binary | Purpose |
|---|---|---|
| `lingniu-go-gateway.service` | `gateway` | GB32960 TCP `32960`、JT808 TCP `808`、宇通 MQTT 接入,写 Kafka RAW/unified |
| `lingniu-go-history-writer.service` | `history-writer` | 消费 RAW topic写 TDengine `raw_frames` 和核心时序表 |
| `lingniu-go-stat-writer.service` | `stat-writer` | 消费 RAW topic写 MySQL `vehicle_daily_metric` |
| `lingniu-go-realtime-api.service` | `realtime-api` | 消费 unified topic写 Redis并提供 realtime/raw/stat 查询 API |
## Build
在开发机或 CI 上构建 Linux amd64 二进制:
```bash
cd go/vehicle-gateway
CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags='-s -w' -o /tmp/lingniu-go-native/gateway ./cmd/gateway
CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags='-s -w' -o /tmp/lingniu-go-native/history-writer ./cmd/history-writer
CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags='-s -w' -o /tmp/lingniu-go-native/stat-writer ./cmd/stat-writer
CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags='-s -w' -o /tmp/lingniu-go-native/realtime-api ./cmd/realtime-api
```
## Cutover Notes
1. 先上传二进制到新 release 目录并更新 `current` symlink。
2. 从现有生产 env 生成四个 systemd 专用 env 文件;不要把密钥写入仓库。
3. 停止旧 Docker Go 容器或 Java 容器,释放 `808``32960``20210`
4. 执行:
```bash
systemctl daemon-reload
systemctl enable lingniu-go-gateway lingniu-go-history-writer lingniu-go-stat-writer lingniu-go-realtime-api
systemctl restart lingniu-go-gateway lingniu-go-history-writer lingniu-go-stat-writer lingniu-go-realtime-api
```
## Verification
```bash
systemctl is-active lingniu-go-gateway lingniu-go-history-writer lingniu-go-stat-writer lingniu-go-realtime-api
ss -lntp | egrep ':(808|32960|20210)\b'
journalctl -u lingniu-go-gateway --since '5 minutes ago' --no-pager
curl -sS 'http://127.0.0.1:20210/api/history/raw-frames?protocol=JT808&limit=1'
curl -sS 'http://127.0.0.1:20210/api/history/raw-frames?protocol=GB32960&limit=1'
curl -sS 'http://127.0.0.1:20210/api/history/raw-frames?protocol=YUTONG_MQTT&limit=1'
```

View File

@@ -0,0 +1,13 @@
[Unit]
Description=Lingniu Go Capacity Health Check
Documentation=file:/opt/lingniu-go-native/current/docs/ops/go-service-observability.md
After=network-online.target
Wants=network-online.target
[Service]
Type=oneshot
WorkingDirectory=/opt/lingniu-go-native/current
ExecStart=/opt/lingniu-go-native/current/capacity-check
TimeoutStartSec=10
StandardOutput=journal
StandardError=journal

View File

@@ -0,0 +1,12 @@
[Unit]
Description=Run Lingniu Go Capacity Health Check every minute
[Timer]
OnBootSec=2min
OnUnitActiveSec=1min
AccuracySec=5s
Persistent=true
Unit=lingniu-go-capacity-check.service
[Install]
WantedBy=timers.target

View File

@@ -0,0 +1,15 @@
FEICHI_BASE_URL=http://mob.fsfeichi.com.cn:8000
FEICHI_AUTH_SECRET_FILE=/opt/lingniu-go-native/secrets/feichi-auth.json
FEICHI_TARGET_SECRET_FILE=/opt/lingniu-go-native/secrets/feichi-target.json
FEICHI_TARGET_ADDR=127.0.0.1:32960
FEICHI_STATE_FILE=/var/lib/lingniu-feichi-bridge/state.json
FEICHI_OCR_IMAGE=lingniu/feichi-captcha-ocr:1.0.0
FEICHI_LOGIN_MAX_ATTEMPTS=20
FEICHI_LOGIN_RETRY_SECONDS=1
FEICHI_POLL_INTERVAL_SECONDS=10
FEICHI_DISCOVERY_INTERVAL_SECONDS=300
FEICHI_FETCH_CONCURRENCY=4
FEICHI_SOURCE_STALE_SECONDS=120
FEICHI_STALE_REISSUE_ENABLED=true
FEICHI_BACKFILL_ENABLED=false
HEALTH_ADDR=127.0.0.1:20219

View File

@@ -0,0 +1,28 @@
[Unit]
Description=Lingniu Feichi HTTP to GB/T 32960 Bridge
Documentation=file:/opt/lingniu-go-native/current/docs/ops/feichi-bridge.md
After=network-online.target docker.service lingniu-go-gateway.service
Wants=network-online.target
Requires=docker.service
[Service]
Type=simple
WorkingDirectory=/opt/lingniu-go-native/current
EnvironmentFile=/opt/lingniu-go-native/env/base.env
EnvironmentFile=/opt/lingniu-go-native/env/feichi-bridge.env
Environment=GOMEMLIMIT=256MiB
ExecStart=/opt/lingniu-go-native/current/feichi-bridge
Restart=always
RestartSec=5
LimitNOFILE=65536
MemoryLimit=384M
KillSignal=SIGTERM
TimeoutStopSec=30
NoNewPrivileges=true
PrivateTmp=true
ProtectSystem=full
ProtectHome=true
ReadWriteDirectories=/var/lib/lingniu-feichi-bridge
[Install]
WantedBy=multi-user.target

View File

@@ -0,0 +1,21 @@
[Unit]
Description=Lingniu Go RAW Fields Projector
After=network-online.target
Wants=network-online.target
[Service]
Type=simple
WorkingDirectory=/opt/lingniu-go-native/current
EnvironmentFile=/opt/lingniu-go-native/env/base.env
EnvironmentFile=/opt/lingniu-go-native/env/fields-projector.env
Environment=GOMEMLIMIT=384MiB
ExecStart=/opt/lingniu-go-native/current/fields-projector
Restart=always
RestartSec=3
LimitNOFILE=1048576
MemoryLimit=512M
KillSignal=SIGTERM
TimeoutStopSec=30
[Install]
WantedBy=multi-user.target

View File

@@ -1,18 +0,0 @@
[Unit]
Description=Lingniu Go Vehicle Gateway
After=network-online.target
Wants=network-online.target
[Service]
Type=simple
WorkingDirectory=/opt/lingniu-go-native/current
EnvironmentFile=/opt/lingniu-go-native/env/gateway.env
ExecStart=/opt/lingniu-go-native/current/gateway
Restart=always
RestartSec=3
LimitNOFILE=1048576
KillSignal=SIGTERM
TimeoutStopSec=30
[Install]
WantedBy=multi-user.target

View File

@@ -1,18 +0,0 @@
[Unit]
Description=Lingniu Go History Writer
After=network-online.target
Wants=network-online.target
[Service]
Type=simple
WorkingDirectory=/opt/lingniu-go-native/current
EnvironmentFile=/opt/lingniu-go-native/env/history-writer.env
ExecStart=/opt/lingniu-go-native/current/history-writer
Restart=always
RestartSec=3
LimitNOFILE=1048576
KillSignal=SIGTERM
TimeoutStopSec=30
[Install]
WantedBy=multi-user.target

View File

@@ -1,18 +0,0 @@
[Unit]
Description=Lingniu Go Realtime API
After=network-online.target
Wants=network-online.target
[Service]
Type=simple
WorkingDirectory=/opt/lingniu-go-native/current
EnvironmentFile=/opt/lingniu-go-native/env/realtime-api.env
ExecStart=/opt/lingniu-go-native/current/realtime-api
Restart=always
RestartSec=3
LimitNOFILE=1048576
KillSignal=SIGTERM
TimeoutStopSec=30
[Install]
WantedBy=multi-user.target

View File

@@ -1,18 +0,0 @@
[Unit]
Description=Lingniu Go Stat Writer
After=network-online.target
Wants=network-online.target
[Service]
Type=simple
WorkingDirectory=/opt/lingniu-go-native/current
EnvironmentFile=/opt/lingniu-go-native/env/stat-writer.env
ExecStart=/opt/lingniu-go-native/current/stat-writer
Restart=always
RestartSec=3
LimitNOFILE=1048576
KillSignal=SIGTERM
TimeoutStopSec=30
[Install]
WantedBy=multi-user.target

View File

@@ -1,86 +0,0 @@
import java.time.Duration;
import java.util.ArrayList;
import java.util.List;
import java.util.Map;
import java.util.Properties;
import java.util.Set;
import java.util.concurrent.ExecutionException;
import java.util.stream.Collectors;
import org.apache.kafka.clients.admin.AdminClient;
import org.apache.kafka.clients.admin.AdminClientConfig;
import org.apache.kafka.clients.admin.NewPartitions;
import org.apache.kafka.clients.admin.NewTopic;
import org.apache.kafka.clients.admin.TopicDescription;
public final class EnsureKafkaTopics {
private static final short REPLICATION_FACTOR = 1;
private record TopicSpec(String name, int partitions) {}
private EnsureKafkaTopics() {}
public static void main(String[] args) throws Exception {
String brokers = args.length > 0 ? args[0] : "172.17.111.56:9092";
List<TopicSpec> specs =
List.of(
topic("vehicle.event.gb32960.v1", 12),
topic("vehicle.raw.gb32960.v1", 12),
topic("vehicle.dlq.gb32960.v1", 3),
topic("vehicle.event.jt808.v1", 12),
topic("vehicle.raw.jt808.v1", 12),
topic("vehicle.dlq.jt808.v1", 3),
topic("vehicle.event.mqtt-yutong.v1", 12),
topic("vehicle.raw.mqtt-yutong.v1", 12),
topic("vehicle.dlq.mqtt-yutong.v1", 3));
Properties props = new Properties();
props.put(AdminClientConfig.BOOTSTRAP_SERVERS_CONFIG, brokers);
props.put(AdminClientConfig.REQUEST_TIMEOUT_MS_CONFIG, "15000");
props.put(AdminClientConfig.DEFAULT_API_TIMEOUT_MS_CONFIG, "30000");
try (AdminClient admin = AdminClient.create(props)) {
Set<String> existing = admin.listTopics().names().get();
List<NewTopic> missing =
specs.stream()
.filter(spec -> !existing.contains(spec.name()))
.map(spec -> new NewTopic(spec.name(), spec.partitions(), REPLICATION_FACTOR))
.toList();
if (!missing.isEmpty()) {
try {
admin.createTopics(missing).all().get();
} catch (ExecutionException e) {
System.out.println("createTopics result=" + e.getCause());
}
}
Map<String, TopicSpec> byName =
specs.stream().collect(Collectors.toMap(TopicSpec::name, spec -> spec));
Map<String, TopicDescription> descriptions =
admin.describeTopics(byName.keySet()).allTopicNames().get();
Map<String, NewPartitions> increases =
descriptions.entrySet().stream()
.filter(entry -> entry.getValue().partitions().size() < byName.get(entry.getKey()).partitions())
.collect(
Collectors.toMap(
Map.Entry::getKey,
entry -> NewPartitions.increaseTo(byName.get(entry.getKey()).partitions())));
if (!increases.isEmpty()) {
admin.createPartitions(increases).all().get();
Thread.sleep(Duration.ofSeconds(2).toMillis());
descriptions = admin.describeTopics(byName.keySet()).allTopicNames().get();
}
List<String> lines = new ArrayList<>();
descriptions.entrySet().stream()
.sorted(Map.Entry.comparingByKey())
.forEach(entry -> lines.add(entry.getKey() + " partitions=" + entry.getValue().partitions().size()));
System.out.println("bootstrap=" + brokers);
lines.forEach(System.out::println);
}
}
private static TopicSpec topic(String name, int partitions) {
return new TopicSpec(name, partitions);
}
}

View File

@@ -0,0 +1,101 @@
# Vehicle IoT Data Platform Principles
## Goal
Build a production vehicle data platform that can receive GB32960, JT808, and Yutong MQTT data reliably, keep raw evidence traceable, and expose only simple business tables for realtime, history, identity, and metrics.
## First Principles
1. Raw data is the source of truth.
2. Kafka is the durable replay log.
3. Redis is realtime cache, not permanent storage.
4. MySQL stores identity, light realtime business snapshots, and low-cardinality metrics.
5. TDengine stores time-series raw and location history.
6. Protocol-specific fields stay in raw parsed JSON unless they become a stable query or metric requirement.
7. Upper business tables must not duplicate full parsed payloads.
## Data Flow
```mermaid
flowchart LR
GB["GB32960 TCP"] --> GW["Go Gateway"]
JT["JT808 TCP"] --> GW
YM["Yutong MQTT"] --> GW
GW --> NATS["NATS JetStream ingress"]
NATS --> BR["NATS Kafka Bridge"]
BR --> KRAW["Kafka raw topics"]
GW -. fallback .-> KRAW
KRAW --> HW["history-writer"]
KRAW --> SW["stat-writer"]
KRAW --> RA["realtime-api projector"]
HW --> TDRAW["TDengine raw_frames"]
HW --> TDLOC["TDengine vehicle_locations"]
SW --> MYMET["MySQL vehicle_daily_mileage"]
RA --> REDIS["Redis realtime-raw"]
RA --> MYSNAP["MySQL vehicle_realtime_snapshot"]
RA --> MYLOC["MySQL vehicle_realtime_location"]
```
## Minimal Storage Contract
详细的落库边界见 [车辆数据最小落库合约](storage-minimal-contract.md),当前生产数据面见 [生产数据面清单](production-data-plane-inventory.md)。
| Store | Table or key | Purpose | Keep | Avoid |
| --- | --- | --- | --- | --- |
| TDengine | `raw_frames` | Replay and audit evidence | raw hex/text, full parsed JSON, parse status, protocol tags | business-only duplicated columns |
| TDengine | `vehicle_locations` | High-volume location history | time, vin, protocol, longitude, latitude, speed, direction, mileage | full parsed JSON |
| MySQL | `vehicle_realtime_snapshot` | Latest per-protocol vehicle heartbeat | protocol, vin, plate, event time, received time | phone, device, message sequence, parsed JSON |
| MySQL | `vehicle_realtime_location` | Latest business location cache | vin, plate, location, speed, mileage, SOC, event time | full raw payload and protocol internals |
| MySQL | `vehicle_daily_mileage` | Queryable daily mileage | vin, date, protocol, daily mileage, source mileage | temporary vehicle keys, generic metric key/value rows, per-frame raw details |
| MySQL | `vehicle_identity_binding` | Manual identity mapping | vin, plate, phone, oem | registration history, device id |
| MySQL | `jt808_registration` | JT808 registration and auth trace | phone, device id, plate, auth code, vin match state, first/latest seen | GB32960 or MQTT records |
| Redis | `vehicle:latest:{vin}` | Latest merged realtime state | cross-protocol latest fields only for VIN-bound vehicles | full parsed payloads, temporary identities, and historical data |
| Redis | `vehicle:realtime-raw:{protocol}:{vin}` | Latest full realtime protocol state | latest protocol parsed payload for VIN-bound vehicles | historical data |
## Event Envelope Rules
Every received frame should become one `FrameEnvelope`.
- `protocol`, `message_id`, `event_time_ms`, `received_at_ms`, and `parse_status` are mandatory.
- `vin` is preferred as the vehicle identity.
- JT808 can use `phone` as a temporary key before VIN binding is resolved.
- `raw_hex` or `raw_text` must be retained for replay and parser correction.
- `parsed` stores protocol-specific fields.
- `fields` stores only stable cross-protocol core fields: speed, total mileage, longitude, latitude, SOC.
## Optimization Order
1. Stabilize ingress: connection lifecycle, protocol parser correctness, bounded backpressure, durable publish.
2. Stabilize event log: NATS to Kafka bridge, topic names, partition keys, retry and replay.
3. Simplify storage: keep only the minimal tables above, remove duplicated payloads from business tables.
4. Stabilize realtime: Redis full latest raw state, MySQL lightweight snapshot and location. Realtime consumers start from latest when no committed offset exists; historical rebuilds must be explicit replay jobs, not accidental backlog scans.
5. Stabilize history: raw and location query pagination, clear time zone behavior.
6. Stabilize metrics: idempotent daily metrics from mileage differences and raw replay.
7. Add operations: health, readiness, metrics, lag, connection counts, deploy notes.
## Current Next Steps
1. Keep production inventory current whenever a service, topic, table, or Redis key family changes.
2. Add parser correctness tests before changing GB32960, JT808, or Yutong MQTT field extraction.
3. Add lag and write-failure alerts from the existing `/metrics` endpoints.
4. Isolate or delete legacy Kafka topics only after confirming no external consumers still depend on them.
5. Keep new business tables narrow by default. Add a column only when a query or index proves it is needed.
## Runtime Metrics Baseline
Runtime metrics are exposed as Prometheus text from `/metrics`. They are operational signals and must not create new business tables by default.
Core counters:
- `vehicle_gateway_frames_total`: received protocol frames by protocol and parse status.
- `vehicle_gateway_publish_total`: raw publish results by protocol. Unified publish metrics appear only when the explicit compatibility stream is enabled.
- `vehicle_realtime_kafka_messages_total`: realtime consumer messages by topic and status.
- `vehicle_realtime_updates_total`: Redis/MySQL realtime projector updates by topic and status.
- `vehicle_history_writes_total`: TDengine history writes by topic and status.
- `vehicle_stat_writes_total`: MySQL metric writes by topic and status.
- `vehicle_bridge_kafka_writes_total`: NATS to Kafka bridge writes by Kafka topic and status.
If a metric becomes a product requirement, derive a narrow metric table from Kafka replay instead of widening raw or realtime tables.

View File

@@ -0,0 +1,148 @@
# 生产数据面清单
审计时间2026-07-02
本文记录当前 ECS 上 Go 版本车辆接入链路的真实数据面用来约束后续重构新增表、topic、key 之前先对照这里,避免把已经删除的历史设计重新带回来。
## 运行服务
生产接入 ECS`115.29.187.205`
| 服务 | systemd 单元 | 端口 | 职责 |
| --- | --- | --- | --- |
| Gateway | `lingniu-go-gateway.service` | `0.0.0.0:32960``0.0.0.0:808``127.0.0.1:20211` | GB32960、JT808、宇通 MQTT 接入、协议解析、即时鉴权响应和内存身份解析 |
| NATS fast writer | `lingniu-go-nats-fast-writer.service` | `127.0.0.1:20215` | NATS raw 消费,写 Redis 实时投影;生产默认关闭 TDengine stage |
| NATS Kafka bridge | `lingniu-go-nats-kafka-bridge.service` | `127.0.0.1:20214` | NATS canonical raw 到 Kafka raw/fields 的可靠桥接和字段投影 |
| History writer | `lingniu-go-history-writer.service` | `127.0.0.1:20212` | 多 worker 消费 Kafka raw写 TDengine |
| Stat writer | `lingniu-go-stat-writer.service` | `127.0.0.1:20213` | 多 worker 消费 Kafka fields写每日里程 |
| Realtime writer | `lingniu-go-realtime-writer.service` | `127.0.0.1:20216` | 多 worker 消费 Kafka raw写 MySQL 实时表;生产关闭 Redis projector |
| Identity writer | `lingniu-go-identity-writer.service` | `127.0.0.1:20217` | 多 worker 消费 Kafka JT808 raw幂等写注册、鉴权和低频在线触达事实 |
| Realtime API | `lingniu-go-realtime-api.service` | `0.0.0.0:20200` | 读取 Redis/MySQL/TDengine 并提供查询 API不消费 Kafka |
## 总线
### NATS JetStream
NATS 部署在 Kafka ECS 内网 `172.17.111.56:4222`
| 项 | 当前值 |
| --- | --- |
| Stream | `VEHICLE_INGEST` |
| Subjects | Gateway 只新增三类 `vehicle.raw.go.*`;三类 `vehicle.fields.go.*` 暂留在 Stream 合同中兼容切换前消息,不再由 Gateway 直接写入 |
| Durable consumer | `vehicle-kafka-bridge` |
| 语义 | Gateway 每帧只发布一份 canonical rawbridge 复用其中已计算的 `parsed_fields` 生成 fieldsKafka raw 与 fields 都成功后才 ACK 同一条 NATS raw |
| 保留策略 | `NATS_STREAM_MAX_BYTES=21474836480``NATS_STREAM_MAX_AGE_HOURS=24``NATS_STREAM_ENSURE_TIMEOUT_SECONDS=60`NATS 仅做入口缓冲 |
协议 parser 和字段扁平化都只在 Gateway 接入线程执行一次。TDengine、Redis、MySQL、bridge 和统计链路只能读取 envelope 中预计算的 `parsed_fields`;实时帧缺失该字段时 RAW 仍保留,但实时投影会以 `skipped_missing_fields` 跳过,禁止在存储层重新解析或误标在线。
Gateway 到 JetStream 的生产接受边界是本地分段 WALcanonical RAW 先进入 1ms/256 条分组提交并完成 `fsync`,然后才异步发布 JetStream只有收到 PubAck 才把记录标记完成,提交失败或 ACK 超时会释放给有界重放。进程崩溃后从 WAL 重建待发布记录,使用稳定 `Nats-Msg-Id=kind:subject:event_id` 让崩溃窗口内的重复重放由 JetStream 去重。生产基线为 16MiB/5s 分段、100000 append queue、10000 PubAck inflight禁止和旧逐条 JSON spool 共用目录。Redis 快路径使用 `FAST_WRITER_WORKERS=16``FAST_WRITER_BATCH_SIZE=100``FAST_WRITER_FETCH_WAIT_MS=5`,仍在 Redis 成功后手工 ACK NATS。
### Kafka
当前 Go 链路正在使用的 topic
| Topic | 分区 | 写入方 | 主要消费方 |
| --- | --- | --- | --- |
| `vehicle.raw.go.gb32960.v1` | 12 | NATS Kafka bridge | history、realtime |
| `vehicle.raw.go.jt808.v1` | 12 | NATS Kafka bridge | history、realtime、identity |
| `vehicle.raw.go.yutong-mqtt.v1` | 12 | NATS Kafka bridge | history、realtime |
| `vehicle.fields.go.gb32960.v1` | 12 | NATS Kafka bridge 从 canonical raw 投影 | stat |
| `vehicle.fields.go.jt808.v1` | 12 | NATS Kafka bridge 从 canonical raw 投影 | stat |
| `vehicle.fields.go.yutong-mqtt.v1` | 12 | NATS Kafka bridge 从 canonical raw 投影 | stat |
| `vehicle.event.go.unified.v1` | 12 | 显式打开 `PUBLISH_UNIFIED_ENABLED=true` 后才写入 | 兼容旧统一事件消费;当前最小链路不依赖 |
审计时 Kafka 上仍存在旧 Java/Xinda/telemetry topic例如 `vehicle.raw.gb32960.v1``vehicle.raw.jt808.v1``vehicle.event.xinda.v1``vehicle.raw.telemetry-input.v1`。这些不属于当前 Go 最小链路,后续如果确认没有消费者依赖,应单独做 Kafka topic 下线清单,不在业务代码里继续引用。
history、stat、realtime-writer、identity-writer 四个消费服务会以 `vehicle_kafka_consumer_info{service,group,topic}` 暴露当前运行时 Kafka 订阅配置;容量巡检将实时 writer 采集为 `realtime`,将身份 writer 采集为 `identity`,将查询服务采集为 `realtime-api`,以此区分“消费线程实际挂载 topic”和“API 可用性”。
history、stat、realtime-writer、identity-writer 的 Kafka 批消费采用 failure-closed 语义存储失败后不抓取下一批history/realtime/stat 只提交各分区连续成功前缀并保留失败后缀identity 的 MySQL 批事务整体重试commit 失败时四个 writer 都只重试 commit不重复写存储。`vehicle_{history,realtime,stat}_retry_pending_messages``vehicle_identity_writer_retry_pending_messages` 表示当前卡在失败重试中的消息数,生产健康值必须为 `0`
history、realtime、stat 和 identity writer 默认都在一个进程内启动 3 个同 group Kafka consumer由 Kafka 分配 topic 分区。bridge 使用车辆键作为 Kafka key因此同一车辆、同一协议的消息仍在单一分区内有序处理不同分区分别并行写 TDengine、MySQL 实时投影、MySQL 统计和 JT808 身份事实。`HISTORY_WORKERS``REALTIME_WORKERS``STATS_WORKERS``IDENTITY_WRITER_WORKERS` 只能在 topic 分区数和后端连接容量允许的范围内调整,不能靠并发打乱单车状态、累计里程或注册鉴权顺序。
生产环境中 `nats-fast-writer` 只负责 Redis 快速当前态,`FAST_WRITER_TDENGINE_ENABLED=false`TDengine RAW/位置历史由 `history-writer` 通过 Kafka raw topic 单路写入,避免 NATS fast path 与 Kafka history path 双写同一证据层。
生产环境中 Redis 当前态只允许 `nats-fast-writer` 写入。`realtime-writer``REALTIME_REDIS_PROJECTOR_ENABLED=false`,只保留 MySQL 当前态投影;`realtime-api` 只负责查询。指标 `vehicle_realtime_redis_projector_enabled` 必须为 `0`,否则 `capacity-check` 判为 degraded避免 Kafka 慢链路覆盖 NATS 快链路的实时结果。
## TDengine
数据库:`lingniu_vehicle_ts`
| Stable | 主列 | Tags | 职责 |
| --- | --- | --- | --- |
| `raw_frames` | `ts``frame_id``event_id``message_id``event_time``received_at``raw_size_bytes``raw_hex``raw_text``parsed_json``parse_status``parse_error``source_endpoint` | `protocol``vehicle_key``vin``phone``device_id` | RAW 证据层,保存完整解析 JSON |
| `raw_frame_payload_chunks` | 分片 payload 列 | 协议和车辆标识 tags | 保存超长 raw/parsed payload |
| `vehicle_locations` | `ts``event_id``frame_id``received_at``longitude``latitude``altitude_m``speed_kmh``direction_deg``alarm_flag``status_flag``total_mileage_km``soc_percent` | `protocol``vin` | 高频历史位置、SOC 和总里程查询;只写 VIN 非空车辆。writer 启动时幂等执行 `ADD COLUMN soc_percent`,兼容已有 stable |
不应重新出现:
- `vehicle_mileage_points`
- `raw_frames.fields_json`
## MySQL
数据库:`lingniu_vehicle_data`
| 表 | 核心字段 | 写入方 | 职责 |
| --- | --- | --- | --- |
| `vehicle` | `vin``plate``oem``enabled` | 导入/人工维护Gateway 只读 | 车辆主实体,承载 VIN 到默认车牌/厂商的轻量索引 |
| `vehicle_identifier` | `protocol``source_code``identifier_type``identifier_value``vin``plate``oem` | 映射文件导入/人工维护Gateway 只读 | 多平台、多标识反查 VIN当前主要承载 JT808 手机号/车牌映射 |
| `vehicle_identity_binding` | `vin``plate``phone``oem` | 导入/人工维护Gateway 兼容只读 | 旧版 VIN/车牌/手机号事实表;新映射导入会用它反查 VIN运行期解析优先使用 `vehicle_identifier` |
| `jt808_registration` | `phone``device_id``plate``vin``manufacturer``auth_token``source_endpoint`、首次/最新注册鉴权时间 | Identity writer | JT808 注册、鉴权和 VIN 匹配状态;可从 Kafka JT808 raw 重放重建 |
| `vehicle_realtime_snapshot` | `protocol``vin``plate``platform_name``peer``parsed_json``event_time``received_at``event_id``access_first_seen_at/access_previous_received_at/access_latest_received_at/access_report_interval_ms/access_sample_count/access_latest_event_id/access_first_seen_source` | Realtime writer | 每协议每 VIN 最新合并扁平字段快照;同一条原子 upsert 还维护与设备事件时间解耦的接收证据。重复 event ID、相同或回退接收时间不推进连续间隔旧行只做 `snapshot_backfill` 上线基线,不宣称历史首次接入 |
| `vehicle_realtime_location` | `protocol``vin``plate`、经纬度、速度、总里程、SOC、事件时间 | Realtime writer | 每协议每 VIN 最新位置业务缓存;由 Kafka raw 写入 MySQL不参与 Redis 当前态 |
| `vehicle_data_source` | `protocol``source_ip``source_code``platform_name``trust_priority``enabled` | Stat writer / 映射导入 / 人工维护 | 车辆数据来源管理;`source_code` 是机器可用的稳定来源编码,`platform_name` 是展示和人工维护名称 |
| `vehicle_daily_mileage_source` | `vin``stat_date``protocol``source_key``source_ip`、首末总里程、样本数、质量状态、是否选中 | Stat writer | 每协议、每来源的每日里程事实层;高频更新,保留多源候选 |
| `vehicle_daily_mileage` | `vin``stat_date``protocol``source_id``daily_mileage_km``latest_total_mileage_km` | Stat writer | 对外查询的每日里程结果层;由 `vehicle_daily_mileage_source` 按来源优先级/质量选举投影 |
身份表使用业务主键:`vehicle``vin` 为主键,`vehicle_identifier``(protocol, source_code, identifier_type, identifier_value)` 为主键,`vehicle_identity_binding``vin` 为主键,`jt808_registration``phone` 为主键;这些核心身份表不保留代理自增主键和 `created_at``vehicle_identity_binding` 是外部维护的兼容事实表,服务运行时不回写。
Gateway 默认每 60 秒把 `vehicle_identity_binding``vehicle_identifier``jt808_registration``vehicle_data_source` 原子刷新为本机只读身份快照。808 每帧只查内存,注册帧仍会立即更新当前 Gateway 的 phone 会话并返回鉴权码;持久化由 `identity-writer` 消费 Kafka JT808 raw 完成Gateway 必须设置 `JT808_REGISTRATION_GATEWAY_WRITES_ENABLED=false`。注册、鉴权帧不节流,普通位置帧默认每个 phone 每 10 分钟触达一次MySQL 写失败时不提交 Kafka offset恢复后继续重放。快照刷新失败继续使用上一版MySQL 不可用不会阻断 TCP 接入。
NATS→Kafka Bridge 只消费 canonical RAW并从 RAW 中已有的 `parsed_fields` 派生 fields envelope不重新解析协议。单批次按 Kafka topic 分组并发写入,默认 `BRIDGE_KAFKA_WRITE_CONCURRENCY=6`;同一个 RAW 对应的 RAW/fields topic 全部成功后才 ACK NATS失败时保留源消息重放。因此并行化只缩短独立 topic 的网络确认等待,不改变 RAW 证据和统计字段的一致性边界。
日里程事实表 `vehicle_daily_mileage_source` 使用 `(vin, stat_date, protocol, source_key)` 作为业务主键,保留首末总里程和样本数用于解释差值计算;对外结果表 `vehicle_daily_mileage` 使用 `(vin, stat_date, protocol)` 作为业务主键,不保留自增 `id``created_at` 和候选来源明细。
Stat writer 每条有效里程样本都会更新 `vehicle_daily_mileage_source`,但 `vehicle_daily_mileage` 的选举投影默认按 `STATS_PROJECT_INTERVAL_SECONDS=15` 节流;新车辆/新协议/新日期/新来源会立即投影。这样保留每日里程事实的实时性,同时避免高频帧对 MySQL 结果表做多 SQL 放大。
每日里程只采用同来源累计里程边界差值:`当日最新累计总里程 - 最近历史日最后累计总里程`。优先取前一自然日;前一日没有数据时继续向更早日期查找,历史完全为空才以当天第一条为基线。不同协议、平台和终端来源分别计算候选,再由来源质量和优先级选举最终结果,不混用两个来源的累计总里程。
`vehicle_data_source``(protocol, source_ip)` 唯一识别来源。直连终端可能产生很多来源 IP这些记录可以保留但默认不要求平台名转发平台来源通过 `vehicle_identifier.source_code``jt808_registration.source_endpoint` 推断后补充 `source_code/platform_name`,后续统计优先使用 `source_id/trust_priority/enabled` 做可信源选择。
统计分日优先使用协议事件时间;如果事件时间明显晚于接收时间超过 10 分钟则认为设备时间异常使用接收时间归属统计日。RAW 历史仍保留原始事件时间,便于追溯。
两张实时当前态表均使用 `(protocol, vin)` 作为业务主键,不保留自增 `id``created_at`;最新更新时间使用 `updated_at`
不应重新出现:
- `vehicle_daily_metric`
- `vehicle_identity_binding_registration`
- `vehicle_identity_bindings`
## Redis
Redis 使用 DB 50定位为实时缓存不作为历史事实来源。
| Key 族 | 用途 |
| --- | --- |
| `vehicle:latest:{vin}` | 跨协议合并后的最新核心字段快照,不重复保存完整 parsed只写 VIN 非空车辆 |
| `vehicle:latest:{vin}:{protocol}` | 单协议最新轻量快照,保留核心 fields 和时间信息,不重复保存完整 parsed |
| `vehicle:realtime-raw:{protocol}:{vin}` | 单协议最新完整 parsed 状态,是实时完整协议字段的唯一 Redis 副本 |
| `vehicle:rt-kv:{protocol}:{vin}:values` | 单协议扁平化实时字段值,字段名遵循协议字段映射 |
| `vehicle:rt-kv:{protocol}:{vin}:types` | `values` 中每个字段的值类型 |
| `vehicle:rt-kv:{protocol}:{vin}:times` | `values` 中每个字段最后写入的归一事件时间;快路径用它防止乱序旧帧覆盖新字段 |
| `vehicle:rt-kv:{protocol}:{vin}:meta` | 单协议 KV 投影的最新事件、接收时间和字段映射版本 |
| `vehicle:online:{protocol}:{vin}` | 在线状态和 TTL |
| `vehicle:online-state:{protocol}:{vin}` | 在线状态的 Hash 副本,便于分页查询 |
| `vehicle:protocols:{vin}` | 当前车辆最近出现过的协议集合 |
| `vehicle:last_seen` | 最近活跃车辆排序集合 |
审计时活跃 key 族主要是 `vehicle:latest:*``vehicle:realtime-raw:*``vehicle:rt-kv:*``vehicle:online:*``vehicle:online-state:*``vehicle:protocols:*``vehicle:last_seen`。旧文档中的 `vehicle:realtime:*``vehicle:merged:*` 不是当前活跃 key 族。
## 后续优化约束
1. 接入层每帧只解析和扁平化一次并写一份 NATS canonical rawKafka 是持久回放层fields 必须由可重放消费者从 raw 中已有的 `parsed_fields` 投影,不能由 Gateway 双写或由存储层重算。
2. TDengine 只放高写入时序数据RAW 证据和位置历史。
3. MySQL 只放低基数业务状态:身份、实时轻量快照、每日里程。
4. Redis 只放当前态,所有 key 都必须允许 TTL 过期后从 Kafka/TDengine/MySQL 重建。
5. 协议新增字段默认进入 `raw_frames.parsed_json` 物理列中的扁平 `parsed_fields` JSON并同步进入 Redis KV只有稳定查询需求出现后才提升为 TDengine/MySQL 独立列。
6. Gateway 不直接写业务数据库JT808 即时会话留在内存,注册鉴权事实由 Kafka identity-writer 单路投影。

View File

@@ -0,0 +1,109 @@
# 车辆数据最小落库合约
## 目标
车辆接入系统只持久化能回答业务问题、能回放纠错、能支撑高频查询的数据。协议细节默认保存在 raw不向上层业务表扩散。
## 第一性原则
1. raw 是证据层,必须能证明收到过什么、解析成什么。
2. Kafka 是回放层,消费者失败后优先靠 Kafka lag 追平。
3. Redis 是当前态缓存,不是历史库。
4. TDengine 只承担高写入时间序列raw、位置历史。
5. MySQL 只承担低基数业务状态:身份映射、实时快照、日指标。
6. 同一个事实只落一个主表,其他表只存查询必要的投影。
7. 新字段先由 Gateway 一次扁平化后进入 `parsed_fields`TDengine 物理列仍名为 `parsed_json`,只有稳定查询需求出现后才提升为列。
## 目标表边界
| 存储 | 表/Key | 保留内容 | 不保留内容 |
| --- | --- | --- | --- |
| TDengine | `raw_frames` | raw hex/text、扁平 `parsed_fields` JSON物理列 `parsed_json`)、解析状态、协议标签、车辆标识标签 | 原始嵌套解析树、`fields_json` 等重复字段 |
| TDengine | `raw_frame_payload_chunks` | 超长 raw/parsed payload 分片 | 业务查询字段 |
| TDengine | `vehicle_locations` | 时间、VIN/协议标签、经纬度、速度、方向、SOC、总里程等位置核心字段 | 完整协议 JSON、注册鉴权信息、phone/device_id/vehicle_key 兜底标识 |
| MySQL | `vehicle_realtime_snapshot` | 每个协议+VIN 的最新事件态、合并扁平字段,以及独立的首次/前次/最新接收时间、连续间隔、样本数和证据来源 | phone、device、原始嵌套解析树、raw 报文、无限接收历史 |
| MySQL | `vehicle_realtime_location` | 每个协议+VIN 的最新位置核心字段 | raw、parsed JSON、消息头内部字段 |
| MySQL | `vehicle_daily_mileage` | 日期、VIN、协议、日里程、首末总里程、样本数 | 临时 vehicle key、泛化 metric key/value、自增 id、created_at、每帧细节、位置点列表 |
| MySQL | `vehicle_identity_binding` | 人工维护或导入的 VIN、车牌、phone、oem 映射 | 注册历史、协议状态、device_id、自增 id、created_at |
| MySQL | `jt808_registration` | JT808 phone 主键下的注册、鉴权、VIN 匹配状态、来源端点 | GB32960/MQTT 注册信息、位置历史、created_at |
| Redis | `vehicle:realtime-raw:{protocol}:{vin}` | 每协议每 VIN 最新完整 parsed 状态 | 无 VIN 临时身份、历史数据、统计结果 |
| Redis | `vehicle:rt-kv:{protocol}:{vin}:values/types/times/meta` | 每协议每 VIN 最新扁平实时字段、类型、字段级事件时间和投影元信息 | 历史字段值、统计结果、无 VIN 临时身份 |
## 当前应收敛的重复点
1. `vehicle_mileage_points``vehicle_locations.total_mileage_km` 重复。
- 状态Go 写入链路已停止创建和写入 `vehicle_mileage_points`
- 查询:不再暴露单独 `/api/history/mileage-points`,里程点直接从 `/api/history/locations``total_mileage_km` 获取。
- 生产ECS TDengine 历史库已在上线前重建,`vehicle_mileage_points` 不再存在。
2. `raw_frames.fields_json``raw_frames.parsed_json``vehicle_locations` 重复。
- 状态:新建 raw schema 和 raw 写入已停止使用 `fields_json`
- 兼容raw 查询只选择公共列,不依赖 `fields_json`;旧表保留该列也不影响查询。
- 生产ECS TDengine 历史库已在上线前重建,`raw_frames` 不再包含 `fields_json`
3. `vehicle_daily_mileage` 不再接收临时 `vehicle_key`
- 原因:日统计是正式业务指标,应只统计已经定位到 VIN 的车辆。
- 无 VIN 的 JT808 等数据保留在 raw/Redis/session 中用于排查和待绑定,不进入正式车辆指标。
- 状态schema 使用 `PRIMARY KEY (vin, stat_date, protocol)`,不保留自增 `id``created_at`;首末总里程和样本数保留为日里程计算状态。
- 索引:主键已覆盖按 VIN 查询,不再额外维护 `idx_vin`;只保留按日期、协议日期查询需要的索引。
- 查询:`/api/stats/daily-metrics` 默认不执行 `COUNT(*)``total` 表示本页返回条数;只有报表总页数等场景传 `includeTotal=true`
- 生产RDS 已在上线前删除泛化 `vehicle_daily_metric`,改为专用 `vehicle_daily_mileage`
4. `vehicle_locations` 不再接收临时身份兜底字段。
- 原因:位置历史是正式车辆时间序列,只按 `protocol + vin` 分表和查询。
- 无 VIN 的位置帧仍保留在 `raw_frames.parsed_json`,待 identity/binding 修复后可从 raw 重放补写。
- 查询:`/api/history/locations` 默认不执行大表 `COUNT(*)``total` 表示本页返回条数;只有需要精确总数时传 `includeTotal=true`
- 生产ECS TDengine `vehicle_locations` 已在上线前重建为 `protocol``vin` 两个 tag。
5. 历史 raw 查询默认不做精确总数。
- 原因:`raw_frames` 是最高写入量证据层,分页/排查通常只需要最新一页数据。
- 查询:`/api/history/raw-frames` 默认 `total` 为本页返回条数;只有导出前预估、后台管理需要总页数时才传 `includeTotal=true`
6. MySQL realtime 当前态表不保留代理主键和创建时间。
- 原因:`vehicle_realtime_snapshot``vehicle_realtime_location` 都是每协议每 VIN 一行的当前态投影,业务主键就是 `(protocol, vin)`
- 接入证据按接收时间单调推进,不受设备事件时间乱序影响;重复事件 ID 或回退接收时间不能增加样本数。`access_first_seen_source=snapshot_backfill` 仅表示部署时基线,只有 `live_writer` 新行能证明 writer 上线后的首次观测。
- 状态schema 使用 `PRIMARY KEY (protocol, vin)`,不保留自增 `id``created_at`,对外只暴露最新 `updated_at`
- 索引:不在 `vehicle_realtime_location` 维护经纬度组合索引;实时表只回答当前态,地理范围和轨迹类查询走 TDengine 历史位置。
- 查询:`/api/realtime/snapshots``/api/realtime/locations` 默认不执行 `COUNT(*)``total` 表示本页返回条数;只有需要精确总数时传 `includeTotal=true`
- 写入:车牌从 `vehicle_identity_binding` 按 VIN 主键直查,并使用 VIN 级短 TTL 内存缓存,避免实时帧每条都打 MySQL缓存不作为事实来源。
- `vehicle_realtime_snapshot.parsed_json` 使用与 Redis KV 同口径的扁平字段对象,例如 `gb32960.vehicle.soc_percent``jt808.location.total_mileage_km`
- 不写 MySQL KV 投影表;实时全量字段保留在 Redis KV历史全量字段走 TDengine raw fields。
- 生产:项目上线前可直接重建这两张 MySQL 表,实时数据会从 Kafka 新消息继续投影。
7. Redis realtime 只服务已定位 VIN 的正式车辆。
- 原因:在线状态和实时数据是业务当前态,不是身份排查队列。
- 无 VIN 的 808 等帧只进入 raw 证据层和注册/绑定排查链路,不写 `vehicle:latest:*``vehicle:online:*``vehicle:realtime-raw:*`
- 绑定补齐 VIN 后,后续新帧自然进入实时态;历史补偿需要从 raw/Kafka 回放。
8. identity 层只保留两张表。
- `vehicle_identity_binding`VIN 与 plate/phone/oem 的映射,供 808 等协议反查 VIN。该表只允许外部导入或人工维护服务运行时只读。
- `jt808_registration`808 注册、鉴权、最新活跃和 VIN 匹配状态。
- 状态:`vehicle_identity_binding` 使用 VIN 主键;`jt808_registration` 使用 phone 主键;两张表都不保留代理自增主键和 `created_at`
- 查询:`vehicle_identity_binding` 的 phone/plate 都是唯一键,接入侧按唯一键直查 VIN不做无意义排序。
- 写入链路gateway 对 binding 查 VIN 使用短 TTL 内存缓存,包含未命中缓存,避免无 VIN 高频位置帧每条都打 MySQL。
- 生产RDS 已在上线前删除旧 `vehicle_identity_binding_registration``vehicle_identity_bindings`gateway 启动会自动创建最小 schema。
## 字段提升规则
协议字段只在 Gateway 解析、映射和扁平化一次,进入系统后分三层。`vehicle.fields.*` 事件中的所有字段名必须位于对应协议命名空间:`gb32960.*``jt808.*``yutong_mqtt.*`;统计层不读取内部标准化核心字段作为兼容兜底。
1. `parsed_fields`默认入口保存全量扁平协议字段TDengine 继续使用兼容物理列名 `parsed_json`
2. 核心列只有跨协议稳定查询需要时才提升例如经纬度、速度、SOC、总里程。
3. 指标表:只有聚合口径稳定、产品需要分页/排序/报表时才持久化。
反例:
- 不因为某个协议字段存在就增加 MySQL 列。
- 不为了方便调试在 snapshot/location 放完整 JSON。
- 不为 telemetry field 配置服务提前复制一份字段表;字段配置服务应从 raw/Kafka 回放生成自己的结果。
## 删除或迁移顺序
1. 先写测试证明旧 API 能从新主表读到同等结果。
2. 改代码停止写重复表或重复列。
3. 部署后观察 Kafka lag、writer 成功计数、API 结果。
4. 查询生产库确认旧表不再新增。
5. 做一次备份或导出。
6. 删除旧表或旧列。
任何一步无法证明安全,就停在兼容状态,不直接删生产数据。

View File

@@ -0,0 +1,89 @@
# TDengine Batch Writer Design
更新时间2026-07-03
## Problem
`history-writer` 当前按 Kafka 消息逐条处理,并对每个 raw frame/location 调用 TDengine 写入。这个路径简单、可恢复,但在 10,000+ FPS 后会优先暴露三个瓶颈:
- 每帧一次 SQL/网络往返,吞吐上限低。
- Kafka offset 逐条提交commit 开销随帧率线性增长。
- burst 流量下无法用批量 flush 平滑 TDengine 写入压力。
## Target
按 topic/protocol/目标表分组批量写 TDengine按数量或时间触发 flush。
初始建议:
| 参数 | 初始值 |
| --- | --- |
| raw frame batch size | 200 |
| location batch size | 500 |
| flush interval | 100ms |
| operation timeout | 5s |
| max SQL payload | 8MB guardrail |
| Kafka commit | TDengine batch 成功后提交本批最大 offset |
## Write Semantics
1. Kafka consumer 拉取消息后解析一次 envelope。
2. envelope 已携带 `parsed_fields`TDengine 写入直接复用,不重新从 raw JSON 解析。
3. raw frame 和 location 可以分批写,但 offset 只能在本消息涉及的所有写入成功后提交。
4. TDengine 写入失败时不提交 offset让 Kafka 自动重放。
5. 依赖 `event_id` 保持幂等,避免失败重放造成重复业务记录。
## Non-goals
- 第一阶段不改变 raw frame 表结构。
- 第一阶段不合并不同 raw frame。
- 不丢弃 parse error frame。
- 不把 MySQL snapshot 或 Redis 当前态塞进 history-writer。
## Risks
- 批次越大,失败重试的重复写风险越高,必须依赖 `event_id` 或等价键。
- `parsed_json` 较大时容易让单条 SQL 超限,需要按 payload size 提前切批。
- batch flush 需要暴露 pending、flush duration、write error metrics否则问题只会表现为 Kafka lag。
## Metrics Required
| 指标 | 含义 |
| --- | --- |
| `vehicle_history_batch_flush_total{status}` | 批写成功/失败次数 |
| `vehicle_history_batch_rows_total{status}` | 批写行数 |
| `vehicle_history_batch_flush_duration_ms{status}` | 批写耗时 |
| `vehicle_history_kafka_lag` | 下游是否追得上 Kafka |
## Rollout
1. 保留当前逐条写实现作为 fallback。
2. 增加批写器单元测试,覆盖 count flush、interval flush、失败不 commit。
3. 在测试环境打开 batch writer。
4. 生产先用小批次 `100/100ms`,观察 TDengine latency 和 Kafka lag。
5. 逐步提升到 `200-500/100ms`
## Implementation Status
2026-07-03 已完成第一阶段实现:
- `history.Writer.AppendAllBatch` 按 raw child table 和 location child table 生成多行 `INSERT ... VALUES (...),(...)`
- oversized payload chunks 仍写入 `raw_frame_payload_chunks`,同样按 child table 批量写。
- `history-writer` 默认启动 `HISTORY_WORKERS=3` 个同组消费者,每个 worker 按 `HISTORY_BATCH_SIZE=200``HISTORY_BATCH_WAIT_MS=20` 收集其已分配分区的消息。
- TDengine batch 成功后才批量提交 Kafka messagesbatch 失败不提交 offset让 Kafka 保留可重放语义。
- 保留 `processHistoryMessage``AppendAll` 单条路径,便于回退和测试。
新增指标:
```text
vehicle_history_batch_flush_total{status}
vehicle_history_batch_rows_total{status}
vehicle_history_batch_flush_duration_ms{status}
```
生产 rollout 建议:
1. 先部署默认 `200/100ms`
2. 观察 `vehicle_history_batch_*``vehicle_history_kafka_lag`
3. 如果 TDengine latency 上升或 Kafka lag 不下降,将 `HISTORY_BATCH_SIZE` 降到 `100`
4. 如果稳定且有 backlog再逐步提升到 `500`

View File

@@ -0,0 +1,751 @@
<!doctype html>
<html lang="zh-CN">
<head>
<meta charset="utf-8" />
<meta name="viewport" content="width=device-width, initial-scale=1" />
<title>Lingniu Vehicle Ingest Go Version Data Flow</title>
<style>
:root {
--bg: #f6f7f9;
--ink: #17202a;
--muted: #5f6b7a;
--line: #ccd3dd;
--panel: #ffffff;
--blue: #1f6feb;
--green: #16803c;
--amber: #a15c00;
--red: #c23131;
--purple: #7c3aed;
--teal: #087f8c;
--shadow: 0 10px 30px rgba(25, 33, 45, .08);
}
* { box-sizing: border-box; }
body {
margin: 0;
color: var(--ink);
background: var(--bg);
font: 14px/1.55 -apple-system, BlinkMacSystemFont, "Segoe UI", Arial, "PingFang SC", "Microsoft YaHei", sans-serif;
}
header {
padding: 28px 34px 22px;
background: #111827;
color: white;
}
h1 {
margin: 0 0 8px;
font-size: 28px;
line-height: 1.2;
letter-spacing: 0;
}
.subtitle {
max-width: 1180px;
color: #d1d5db;
font-size: 15px;
}
main {
width: min(1480px, calc(100vw - 36px));
margin: 22px auto 48px;
}
h2 {
margin: 0 0 14px;
font-size: 20px;
line-height: 1.25;
}
h3 {
margin: 0 0 10px;
font-size: 15px;
}
.section {
margin: 18px 0;
padding: 20px;
background: var(--panel);
border: 1px solid #e4e8ef;
border-radius: 8px;
box-shadow: var(--shadow);
}
.legend {
display: flex;
flex-wrap: wrap;
gap: 10px;
margin-top: 16px;
}
.tag {
display: inline-flex;
align-items: center;
gap: 7px;
padding: 4px 8px;
border: 1px solid #d8dee8;
border-radius: 6px;
background: #fff;
color: var(--muted);
font-size: 12px;
}
.dot {
width: 10px;
height: 10px;
border-radius: 50%;
background: var(--blue);
}
.dot.source { background: var(--amber); }
.dot.gateway { background: var(--blue); }
.dot.bus { background: var(--purple); }
.dot.store { background: var(--green); }
.dot.api { background: var(--teal); }
.dot.warning { background: var(--red); }
.flow {
display: grid;
grid-template-columns: 1.15fr 1.35fr 1.15fr 1.2fr 1.4fr 1.1fr;
gap: 14px;
align-items: stretch;
overflow-x: auto;
padding-bottom: 8px;
}
.lane {
min-width: 190px;
display: flex;
flex-direction: column;
gap: 12px;
}
.lane-title {
padding: 8px 10px;
border-radius: 6px;
color: white;
font-weight: 700;
text-align: center;
background: #374151;
}
.node {
position: relative;
min-height: 88px;
padding: 13px 13px 12px;
border: 1px solid #d9e0ea;
border-left: 5px solid var(--blue);
border-radius: 8px;
background: #fff;
}
.node.source { border-left-color: var(--amber); }
.node.gateway { border-left-color: var(--blue); }
.node.bus { border-left-color: var(--purple); }
.node.store { border-left-color: var(--green); }
.node.api { border-left-color: var(--teal); }
.node.warning { border-left-color: var(--red); }
.node strong {
display: block;
margin-bottom: 5px;
font-size: 14px;
}
.node p {
margin: 0;
color: var(--muted);
font-size: 12px;
}
.arrow {
display: flex;
align-items: center;
justify-content: center;
height: 28px;
color: var(--muted);
font-size: 12px;
white-space: nowrap;
}
.arrow:before, .arrow:after {
content: "";
height: 1px;
background: var(--line);
flex: 1;
margin: 0 8px;
}
.arrow:after {
max-width: 16px;
height: 0;
border-top: 5px solid transparent;
border-bottom: 5px solid transparent;
border-left: 8px solid var(--line);
background: transparent;
margin-left: 0;
}
.grid-2 {
display: grid;
grid-template-columns: repeat(2, minmax(0, 1fr));
gap: 14px;
}
.grid-3 {
display: grid;
grid-template-columns: repeat(3, minmax(0, 1fr));
gap: 14px;
}
table {
width: 100%;
border-collapse: collapse;
overflow: hidden;
border: 1px solid #e1e6ef;
border-radius: 8px;
background: white;
}
th, td {
padding: 10px 11px;
border-bottom: 1px solid #edf1f6;
text-align: left;
vertical-align: top;
font-size: 13px;
}
th {
background: #f0f4f8;
color: #263241;
font-weight: 700;
}
tr:last-child td { border-bottom: 0; }
code {
padding: 1px 5px;
border-radius: 4px;
background: #eef2f7;
color: #1f2937;
font-family: ui-monospace, SFMono-Regular, Menlo, Consolas, monospace;
font-size: 12px;
}
.pill {
display: inline-block;
margin: 2px 4px 2px 0;
padding: 2px 7px;
border-radius: 999px;
background: #eef2ff;
color: #3730a3;
font-size: 12px;
white-space: nowrap;
}
.note {
padding: 11px 12px;
border-left: 4px solid var(--amber);
background: #fff7ed;
color: #61410b;
border-radius: 6px;
}
.ok {
padding: 11px 12px;
border-left: 4px solid var(--green);
background: #effaf1;
color: #164b25;
border-radius: 6px;
}
.mini-flow {
display: grid;
grid-template-columns: repeat(7, minmax(128px, 1fr));
gap: 8px;
overflow-x: auto;
padding-bottom: 8px;
}
.mini-step {
min-height: 82px;
padding: 10px;
border: 1px solid #dfe6f1;
border-radius: 8px;
background: #fff;
}
.mini-step b {
display: block;
margin-bottom: 4px;
color: #111827;
font-size: 13px;
}
.mini-step span {
color: var(--muted);
font-size: 12px;
}
.footer {
color: var(--muted);
font-size: 12px;
text-align: center;
padding: 20px;
}
@media (max-width: 980px) {
.grid-2, .grid-3 { grid-template-columns: 1fr; }
header { padding: 22px 18px; }
main { width: calc(100vw - 20px); }
.section { padding: 14px; }
}
</style>
</head>
<body>
<header>
<h1>Go 版本车辆数据处理与流转全图</h1>
<div class="subtitle">
当前 Go 架构以 gateway 做高性能协议接入NATS JetStream 做入口缓冲和解耦nats-kafka-bridge 保持现有 Kafka 下游兼容,最终进入 TDengine、Redis、MySQL并由 realtime-api 对外提供查询能力。
</div>
<div class="legend">
<span class="tag"><span class="dot source"></span>外部数据源</span>
<span class="tag"><span class="dot gateway"></span>Go 接入与解析</span>
<span class="tag"><span class="dot bus"></span>NATS / Kafka 解耦</span>
<span class="tag"><span class="dot store"></span>存储</span>
<span class="tag"><span class="dot api"></span>查询 API</span>
<span class="tag"><span class="dot warning"></span>可靠性边界</span>
</div>
</header>
<main>
<section class="section">
<h2>1. 总览图:从车辆平台到 API</h2>
<div class="flow">
<div class="lane">
<div class="lane-title">外部输入</div>
<div class="node source">
<strong>GB/T 32960 平台</strong>
<p>TCP 连接到 ECS <code>:32960</code>,包含登录、登出、实时信息、补发等帧。</p>
</div>
<div class="node source">
<strong>JT/T 808 平台</strong>
<p>TCP 连接到 ECS <code>:808</code>,包含注册、鉴权、位置上报 <code>0x0200</code> 等。</p>
</div>
<div class="node source">
<strong>宇通 MQTT</strong>
<p>gateway 作为 MQTT client 订阅 <code>/ytforward/shln/+</code></p>
</div>
</div>
<div class="lane">
<div class="lane-title">gateway 接入层</div>
<div class="node gateway">
<strong>TCPServer / MQTTClient</strong>
<p>读取连接、拆包、处理半包/粘包、按协议路由到解析器。</p>
</div>
<div class="node gateway">
<strong>协议解析器</strong>
<p><code>gb32960</code><code>jt808</code><code>yutongmqtt</code> 解析 RAW抽取核心 fields。</p>
</div>
<div class="node gateway">
<strong>身份解析</strong>
<p>通过 MySQL <code>vehicle_identity_binding</code> 将 phone/device/plate 尽量映射到 VIN。</p>
</div>
<div class="node gateway">
<strong>协议响应</strong>
<p>32960/808 需要应答时,在成功发布后写 ACK。</p>
</div>
</div>
<div class="lane">
<div class="lane-title">统一消息</div>
<div class="node gateway">
<strong>FrameEnvelope</strong>
<p>统一封装protocol、message_id、vin、phone、device_id、source_endpoint、raw、parsed、fields。</p>
</div>
<div class="node warning">
<strong>StableEventID</strong>
<p>基于协议、消息、车辆键、序号、时间、raw 生成稳定 ID用于去重和追踪。</p>
</div>
</div>
<div class="lane">
<div class="lane-title">入口解耦</div>
<div class="node bus">
<strong>NATS JetStream</strong>
<p>gateway 发布到 <code>VEHICLE_INGEST</code> stream文件存储24 小时保留。</p>
</div>
<div class="node bus">
<strong>Subjects</strong>
<p><code>vehicle.raw.go.gb32960.v1</code><br><code>vehicle.raw.go.jt808.v1</code><br><code>vehicle.raw.go.yutong-mqtt.v1</code></p>
</div>
<div class="node warning">
<strong>Async + Retry</strong>
<p>gateway 内部异步队列发布 NATS降低协议连接被 Kafka 慢写拖住的风险。</p>
</div>
</div>
<div class="lane">
<div class="lane-title">Kafka 兼容层</div>
<div class="node bus">
<strong>nats-kafka-bridge</strong>
<p>durable pull consumer 批量拉 NATS写 Kafka 成功后才 ACK NATS。</p>
</div>
<div class="node bus">
<strong>Kafka Topics</strong>
<p>与 NATS subject 同名:三个协议 RAW topic。<code>vehicle.event.go.unified.v1</code> 仅作为显式兼容开关保留。</p>
</div>
<div class="node warning">
<strong>失败语义</strong>
<p>Kafka 写失败不 ACKNATS 保留消息等待重试;未知 subject 不 ACK防止静默丢数据。</p>
</div>
</div>
<div class="lane">
<div class="lane-title">落库与查询</div>
<div class="node store">
<strong>history-writer</strong>
<p>消费 RAW topic写 TDengineraw_frames、payload_chunks、locations。</p>
</div>
<div class="node store">
<strong>stat-writer</strong>
<p>消费 RAW topic根据总里程差值写 MySQL 每日指标。</p>
</div>
<div class="node store">
<strong>realtime-api 消费器</strong>
<p>消费三个 RAW topic写 Redis 在线状态、实时合并快照、各协议 realtime-raw。</p>
</div>
<div class="node api">
<strong>HTTP API</strong>
<p>realtime-api 暴露实时、历史、统计查询接口,端口 <code>:20200</code></p>
</div>
</div>
</div>
</section>
<section class="section">
<h2>2. 单条报文的完整生命周期</h2>
<div class="mini-flow">
<div class="mini-step"><b>1. 到达</b><span>32960/808 TCP 报文或 MQTT 消息到 gateway。</span></div>
<div class="mini-step"><b>2. 拆包</b><span>TCP 按协议提取完整帧MQTT 直接按消息处理。</span></div>
<div class="mini-step"><b>3. 解析</b><span>生成 parsed 全量结构化数据和 fields 核心字段。</span></div>
<div class="mini-step"><b>4. 绑定 VIN</b><span>根据 VIN/phone/device_id/plate 生成 vehicle_key。</span></div>
<div class="mini-step"><b>5. 发 NATS</b><span>写协议 RAW subjectunified 只在兼容开关打开时写。</span></div>
<div class="mini-step"><b>6. 桥接 Kafka</b><span>bridge 写 Kafka 成功后 ACK NATS。</span></div>
<div class="mini-step"><b>7. 多路消费</b><span>历史、实时、统计各自消费 Kafka互不阻塞。</span></div>
</div>
</section>
<section class="section">
<h2>3. FrameEnvelopeGo 版本内部统一数据格式</h2>
<div class="grid-2">
<div>
<table>
<thead>
<tr><th>字段</th><th>含义</th><th>来源</th></tr>
</thead>
<tbody>
<tr><td><code>event_id</code></td><td>稳定事件 ID</td><td>已有则保留,否则按协议/消息/车辆/时间/raw 计算</td></tr>
<tr><td><code>protocol</code></td><td>协议</td><td><code>GB32960</code><code>JT808</code><code>YUTONG_MQTT</code></td></tr>
<tr><td><code>message_id</code></td><td>协议消息类型</td><td>如 808 <code>0x0200</code>32960 命令标识等</td></tr>
<tr><td><code>vin</code></td><td>VIN</td><td>协议原始字段或 MySQL binding 反查</td></tr>
<tr><td><code>phone</code></td><td>808 终端手机号</td><td>808 消息头 BCD 解析后去前导 0</td></tr>
<tr><td><code>device_id</code></td><td>设备标识</td><td>协议内设备号、MQTT 设备字段等</td></tr>
<tr><td><code>plate</code></td><td>车牌</td><td>注册帧或 binding 表</td></tr>
<tr><td><code>source_endpoint</code></td><td>来源地址</td><td>TCP remote ip:port 或 MQTT endpoint/topic</td></tr>
</tbody>
</table>
</div>
<div>
<table>
<thead>
<tr><th>字段</th><th>含义</th><th>去向</th></tr>
</thead>
<tbody>
<tr><td><code>raw_hex</code> / <code>raw_text</code></td><td>原始报文</td><td>TDengine <code>raw_frames</code>,超长进入 chunks</td></tr>
<tr><td><code>parsed</code></td><td>全量解析 JSON</td><td>TDengine <code>parsed_json</code>Redis <code>realtime-raw</code></td></tr>
<tr><td><code>fields</code></td><td>最小核心字段</td><td>TDengine 位置/里程点Redis 合并快照;统计计算</td></tr>
<tr><td><code>event_time_ms</code></td><td>车端事件时间</td><td>TDengine 主时间Redis field time</td></tr>
<tr><td><code>received_at_ms</code></td><td>平台接收时间</td><td>延迟分析、在线状态 last_seen</td></tr>
<tr><td><code>parse_status</code></td><td>解析状态</td><td><code>OK</code><code>PARTIAL</code><code>BAD_FRAME</code></td></tr>
<tr><td><code>parse_error</code></td><td>解析错误</td><td>RAW 查询排查使用</td></tr>
<tr><td><code>vehicle_key</code></td><td>车辆主键</td><td>优先 VIN否则 <code>PROTOCOL:phone/device</code></td></tr>
</tbody>
</table>
</div>
</div>
</section>
<section class="section">
<h2>4. 三个协议的处理差异</h2>
<table>
<thead>
<tr>
<th>协议</th>
<th>接入方式</th>
<th>身份主线</th>
<th>RAW 数据</th>
<th>核心 fields</th>
<th>特殊动作</th>
</tr>
</thead>
<tbody>
<tr>
<td><strong>GB32960</strong></td>
<td>gateway TCP <code>:32960</code></td>
<td>报文本身通常有 VIN平台账号用于登录鉴权</td>
<td>完整 parsed JSON 写 <code>raw_frames.parsed_json</code></td>
<td>速度、经纬度、SOC、累计里程等按帧类型抽取</td>
<td>登录/数据帧需要平台应答,成功发布后 ACK</td>
</tr>
<tr>
<td><strong>JT808</strong></td>
<td>gateway TCP <code>:808</code></td>
<td>消息头 phone 为主;注册/鉴权/位置帧结合 binding 找 VIN</td>
<td>完整 parsed JSON 写 <code>raw_frames.parsed_json</code></td>
<td>位置、速度、方向、状态、报警、GPS 总里程</td>
<td>注册/鉴权/通用应答;每日里程按总里程差值统计</td>
</tr>
<tr>
<td><strong>YUTONG_MQTT</strong></td>
<td>gateway MQTT client 订阅</td>
<td>MQTT payload 中 VIN/设备字段source endpoint 为 MQTT endpoint/topic</td>
<td>完整 parsed JSON 写 <code>raw_frames.parsed_json</code></td>
<td>速度、经纬度、总里程、SOC 等</td>
<td>不需要 TCP ACK实时多帧在 Redis 按协议合并</td>
</tr>
</tbody>
</table>
</section>
<section class="section">
<h2>5. NATS + Kafka 的职责分工</h2>
<div class="grid-3">
<div class="node bus">
<strong>NATS JetStream入口缓冲层</strong>
<p>部署在 Kafka ECSDocker Compose 管理;监听内网 <code>172.17.111.56:4222</code>,监控 <code>172.17.111.56:8222</code>。gateway 只需要把事件快速、可靠地写入 NATS。</p>
</div>
<div class="node bus">
<strong>nats-kafka-bridge可靠桥接层</strong>
<p>使用 durable pull consumer批量写 Kafka只有 Kafka 写入成功才 ACK NATS。Kafka 短暂异常时NATS 消息保留并重试。</p>
</div>
<div class="node bus">
<strong>Kafka业务消费层</strong>
<p>保持现有 topic 和消费者模型history/realtime/stat 不直接依赖 gateway也不直接依赖 NATS API。</p>
</div>
</div>
<div class="note" style="margin-top:14px;">
当前阶段没有让下游直接消费 NATS这是为了最小化切换风险入口换成 NATS 后,下游仍沿用 Kafka。后续如果要进一步实时化可以让实时缓存或控制命令直接使用 NATS subject。
</div>
</section>
<section class="section">
<h2>6. 存储模型:哪些数据进入哪里</h2>
<table>
<thead>
<tr>
<th>存储</th>
<th>表/Key</th>
<th>写入方</th>
<th>内容</th>
<th>用途</th>
</tr>
</thead>
<tbody>
<tr>
<td><strong>TDengine</strong></td>
<td><code>raw_frames</code></td>
<td>history-writer</td>
<td>原始报文、完整 parsed JSON、parse status、source endpoint</td>
<td>RAW 查询、追溯、排查解析问题</td>
</tr>
<tr>
<td><strong>TDengine</strong></td>
<td><code>raw_frame_payload_chunks</code></td>
<td>history-writer</td>
<td>超长 raw/parsed payload 分片</td>
<td>避免 BINARY 长度限制导致完整 JSON 丢失</td>
</tr>
<tr>
<td><strong>TDengine</strong></td>
<td><code>vehicle_locations</code></td>
<td>history-writer</td>
<td>经纬度、速度、方向、状态、报警、总里程等核心字段</td>
<td>高频历史位置分页查询</td>
</tr>
<tr>
<td><strong>Redis</strong></td>
<td><code>vehicle:latest:{vin}</code></td>
<td>realtime-api Kafka consumer</td>
<td>跨协议合并后的实时字段快照</td>
<td>查 VIN 是否在线、查 VIN 实时数据</td>
</tr>
<tr>
<td><strong>Redis</strong></td>
<td><code>vehicle:latest:{vin}:{protocol}</code></td>
<td>realtime-api Kafka consumer</td>
<td>单协议实时轻量快照</td>
<td>区分 32960/808/MQTT 的实时状态</td>
</tr>
<tr>
<td><strong>Redis</strong></td>
<td><code>vehicle:realtime-raw:{protocol}:{vin}</code></td>
<td>realtime-api Kafka consumer</td>
<td>单协议最新 parsed 全量字段</td>
<td>实时 RAW 字段查看,不走 TDengine 历史扫描</td>
</tr>
<tr>
<td><strong>Redis</strong></td>
<td><code>vehicle:online:{vin}</code><code>vehicle:last_seen</code></td>
<td>realtime-api Kafka consumer</td>
<td>在线状态、最后接收时间、协议列表</td>
<td>在线判断和近期车辆列表</td>
</tr>
<tr>
<td><strong>MySQL</strong></td>
<td><code>vehicle_identity_binding</code></td>
<td>人工/导入维护gateway 读取</td>
<td>phone/device/plate 到 VIN 的映射</td>
<td>808 等缺 VIN 协议反查正确 VIN</td>
</tr>
<tr>
<td><strong>MySQL</strong></td>
<td><code>jt808_registration</code></td>
<td>gateway</td>
<td>808 注册、鉴权、首次/最新上报、phone、device、plate、vin 解析结果</td>
<td>定位哪些手机号没有映射 VIN排查注册鉴权</td>
</tr>
<tr>
<td><strong>MySQL</strong></td>
<td><code>vehicle_daily_mileage</code></td>
<td>stat-writer</td>
<td>每日里程,按首末总里程差值计算,保留样本数</td>
<td>统计查询</td>
</tr>
</tbody>
</table>
</section>
<section class="section">
<h2>7. 查询入口</h2>
<div class="grid-2">
<div>
<h3>实时查询</h3>
<table>
<thead><tr><th>接口类型</th><th>数据源</th><th>说明</th></tr></thead>
<tbody>
<tr><td>VIN 是否在线</td><td>Redis online key</td><td>TTL 内有数据即在线,默认 TTL 600 秒。</td></tr>
<tr><td>VIN 实时数据</td><td>Redis latest snapshot</td><td>跨协议合并核心 fields完整协议字段通过 realtime-raw 查询。</td></tr>
<tr><td>单协议实时 RAW</td><td>Redis realtime-raw</td><td>查看某 VIN 在某协议下最新 parsed 全量字段。</td></tr>
</tbody>
</table>
</div>
<div>
<h3>历史查询</h3>
<table>
<thead><tr><th>接口类型</th><th>数据源</th><th>说明</th></tr></thead>
<tbody>
<tr><td>RAW 帧查询</td><td>TDengine raw_frames + chunks</td><td>按协议、VIN/phone、时间、消息类型分页。</td></tr>
<tr><td>位置/总里程历史</td><td>TDengine vehicle_locations</td><td>高频位置分页查询,包含 total_mileage_km避免重复里程点接口。</td></tr>
<tr><td>每日里程</td><td>MySQL vehicle_daily_mileage</td><td>按日期、协议查询首末总里程差值结果。</td></tr>
</tbody>
</table>
</div>
</div>
</section>
<section class="section">
<h2>8. 生产部署视图</h2>
<table>
<thead>
<tr><th>节点</th><th>组件</th><th>地址/端口</th><th>说明</th></tr>
</thead>
<tbody>
<tr>
<td>Gateway ECS</td>
<td><code>lingniu-go-gateway</code></td>
<td>公网 <code>115.29.187.205</code>;生产端口 <code>32960</code><code>808</code></td>
<td>承接外部平台真实连接,发布到 NATS。</td>
</tr>
<tr>
<td>Gateway ECS</td>
<td><code>lingniu-go-nats-kafka-bridge</code></td>
<td>连接 NATS <code>172.17.111.56:4222</code>Kafka <code>172.17.111.56:9092</code></td>
<td>当前部署在 gateway ECS负责 NATS 到 Kafka 桥接。</td>
</tr>
<tr>
<td>Gateway ECS</td>
<td><code>history-writer</code><code>stat-writer</code><code>realtime-api</code></td>
<td>API <code>115.29.187.205:20200</code></td>
<td>消费 Kafka写 TDengine/Redis/MySQL并提供查询。</td>
</tr>
<tr>
<td>Kafka ECS</td>
<td>Kafka + NATS</td>
<td>公网 <code>114.55.58.251</code>;内网 <code>172.17.111.56</code></td>
<td>NATS 使用 Docker Compose端口只绑定内网 <code>4222</code>/<code>8222</code></td>
</tr>
<tr>
<td>TDengine ECS</td>
<td>TDengine</td>
<td>内网 <code>172.17.111.57:6041</code></td>
<td>历史 RAW、位置时序存储。</td>
</tr>
<tr>
<td>云服务</td>
<td>MySQL / Redis</td>
<td>RDS 内网、Redis 内网</td>
<td>身份绑定、808 注册、每日指标、实时缓存。</td>
</tr>
</tbody>
</table>
</section>
<section class="section">
<h2>9. 关键可靠性设计</h2>
<div class="grid-3">
<div class="ok">
<strong>接入不直接等待 Kafka</strong><br>
gateway 先写 NATSKafka 慢写由 bridge 消化,降低 32960/808 连接积压风险。
</div>
<div class="ok">
<strong>NATS ACK 在 Kafka 成功之后</strong><br>
bridge 写 Kafka 失败时不 ACK消息仍在 JetStream恢复后继续拉取。
</div>
<div class="ok">
<strong>RAW 完整 JSON 保留</strong><br>
<code>parsed</code> 全量写 RAW核心表只存查询常用字段避免位置查询每次扫完整 JSON。
</div>
<div class="ok">
<strong>实时多协议合并</strong><br>
Redis latest snapshot 只保留统一实时核心字段;完整协议 parsed 只放 realtime-raw避免重复缓存大 JSON。
</div>
<div class="ok">
<strong>身份解析边界</strong><br>
实时缓存只写 VIN 非空车辆;无 VIN 帧保留在 RAW 和注册/绑定排查链路,补齐绑定后通过新帧或回放进入实时态。
</div>
<div class="ok">
<strong>数据分层查询</strong><br>
高频查询走 TDengine 核心字段表和 Redis完整追溯走 RAW JSON性能和完整性分开处理。
</div>
</div>
</section>
<section class="section">
<h2>10. 当前运行链路一句话</h2>
<p>
外部 32960/808/MQTT 数据进入 Go gateway 后,被解析成 <code>FrameEnvelope</code>,先写入 NATS JetStream
<code>nats-kafka-bridge</code> 把 NATS 消息可靠转写到 KafkaKafka 再分发给 <code>history-writer</code> 写 TDengine、
<code>stat-writer</code> 写 MySQL 每日指标、<code>realtime-api</code> 写 Redis 实时缓存并提供 HTTP 查询。
</p>
</section>
</main>
<div class="footer">
Generated for lingniu-vehicle-ingest Go architecture. Update this file when gateway subjects, storage tables, or deployment topology changes.
</div>
</body>
</html>

View File

@@ -1,749 +0,0 @@
<!doctype html>
<html lang="zh-CN">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>lingniu-vehicle-ingest 模块与数据流</title>
<style>
:root {
--bg: #f7f8fa;
--panel: #ffffff;
--ink: #172033;
--muted: #667085;
--line: #d8dee8;
--blue: #2563eb;
--green: #0f766e;
--amber: #b45309;
--red: #b42318;
--purple: #6d28d9;
}
* {
box-sizing: border-box;
}
body {
margin: 0;
background: var(--bg);
color: var(--ink);
font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", "PingFang SC", "Microsoft YaHei", sans-serif;
line-height: 1.55;
}
header {
padding: 32px 40px 24px;
background: #101828;
color: #ffffff;
}
header h1 {
margin: 0 0 10px;
font-size: 30px;
line-height: 1.2;
letter-spacing: 0;
}
header p {
margin: 0;
max-width: 980px;
color: #d0d5dd;
font-size: 15px;
}
main {
max-width: 1420px;
margin: 0 auto;
padding: 28px 28px 44px;
}
section {
margin-bottom: 24px;
padding: 24px;
background: var(--panel);
border: 1px solid var(--line);
border-radius: 8px;
box-shadow: 0 1px 2px rgba(16, 24, 40, 0.04);
}
h2 {
margin: 0 0 14px;
font-size: 21px;
letter-spacing: 0;
}
h3 {
margin: 22px 0 10px;
font-size: 17px;
}
p {
margin: 0 0 12px;
}
.summary-grid {
display: grid;
grid-template-columns: repeat(4, minmax(0, 1fr));
gap: 12px;
margin-top: 18px;
}
.summary-card {
padding: 14px 16px;
border: 1px solid var(--line);
border-left: 4px solid var(--blue);
border-radius: 6px;
background: #fbfcfe;
}
.summary-card strong {
display: block;
margin-bottom: 4px;
font-size: 14px;
}
.summary-card span {
color: var(--muted);
font-size: 13px;
}
.diagram {
overflow-x: auto;
padding: 12px;
border: 1px solid var(--line);
border-radius: 6px;
background: #ffffff;
}
.legend {
display: flex;
flex-wrap: wrap;
gap: 10px;
margin: 14px 0 0;
color: var(--muted);
font-size: 13px;
}
.legend span {
display: inline-flex;
align-items: center;
gap: 6px;
}
.dot {
width: 10px;
height: 10px;
border-radius: 50%;
display: inline-block;
}
.dot.entry { background: var(--blue); }
.dot.core { background: var(--green); }
.dot.sink { background: var(--amber); }
.dot.safety { background: var(--red); }
.dot.command { background: var(--purple); }
table {
width: 100%;
border-collapse: collapse;
margin-top: 12px;
font-size: 14px;
}
th,
td {
padding: 10px 12px;
border: 1px solid var(--line);
text-align: left;
vertical-align: top;
}
th {
background: #f2f4f7;
color: #344054;
white-space: nowrap;
}
td:first-child {
width: 190px;
font-family: ui-monospace, SFMono-Regular, Menlo, Monaco, Consolas, "Liberation Mono", monospace;
font-size: 13px;
color: #101828;
white-space: nowrap;
}
.tag {
display: inline-block;
margin: 0 6px 6px 0;
padding: 2px 7px;
border-radius: 999px;
background: #eef2ff;
color: #3730a3;
font-size: 12px;
white-space: nowrap;
}
.tag.red {
background: #fee4e2;
color: #912018;
}
.tag.green {
background: #ccfbef;
color: #134e48;
}
.tag.amber {
background: #fef0c7;
color: #93370d;
}
.callout {
margin-top: 14px;
padding: 14px 16px;
border-left: 4px solid var(--red);
border-radius: 6px;
background: #fff7f5;
}
code {
padding: 1px 4px;
border-radius: 4px;
background: #eef2f6;
font-family: ui-monospace, SFMono-Regular, Menlo, Monaco, Consolas, "Liberation Mono", monospace;
font-size: 0.95em;
}
@media (max-width: 900px) {
header {
padding: 24px 20px 20px;
}
main {
padding: 18px;
}
section {
padding: 18px;
}
.summary-grid {
grid-template-columns: repeat(2, minmax(0, 1fr));
}
table {
display: block;
overflow-x: auto;
white-space: nowrap;
}
}
@media (max-width: 560px) {
.summary-grid {
grid-template-columns: 1fr;
}
}
</style>
</head>
<body>
<header>
<h1>lingniu-vehicle-ingest 模块与数据流</h1>
<p>
本图基于当前 Maven 多模块项目整理,用于后续讨论协议接入、内部字段、氢能车辆运营统计、安全告警和下游消费边界。
当前架构将接入层、历史明细、Redis 热状态和日统计拆成独立模块,通过统一 Kafka Envelope 解耦。
</p>
</header>
<main>
<section>
<h2>一、系统定位</h2>
<p>
当前项目不是业务库写入服务。它的核心职责是把 GB/T 32960、JT/T 808、Yutong MQTT
等默认生产来源的车辆数据统一转换为 <code>VehicleEvent</code> 和全字段 <code>TelemetrySnapshot</code>
JT/T 1078、JSATL12 作为显式 profile 的可选能力保留,
再通过 Kafka、原始报文归档、TDengine 历史索引、Redis 热状态和独立统计模块交给下游业务。
</p>
<div class="summary-grid">
<div class="summary-card">
<strong>接入方式</strong>
<span>Netty TCP、MQTT、后续可扩展更多 Inbound Adapter信达 Push 源码已删除。</span>
</div>
<div class="summary-card">
<strong>统一模型</strong>
<span>所有协议归一到 <code>VehicleEvent</code> 和全字段内部 <code>TelemetrySnapshot</code></span>
</div>
<div class="summary-card">
<strong>输出方式</strong>
<span>Kafka Protobuf Envelope + 原始 bytes 冷存 + TDengine raw_frames/locations + Redis 热状态。</span>
</div>
<div class="summary-card">
<strong>业务边界</strong>
<span>日统计由 <code>vehicle-stat-service</code> 消费 Kafka 实现,告警工单、资产业务继续由下游消费。</span>
</div>
</div>
</section>
<section>
<h2>二、模块分层图</h2>
<div class="diagram">
<pre class="mermaid">
flowchart TB
subgraph external["外部来源"]
vehicle["车辆终端<br/>GB/T 32960 / JT808 / JT1078"]
mqttSource["车企或平台 MQTT"]
commandClient["业务系统 / 运维平台<br/>HTTP 下行命令"]
end
subgraph protocolModules["协议模块 modules/protocols"]
gb["protocol-gb32960<br/>32960 Netty 接入、鉴权、ACK、解析"]
jt808["protocol-jt808<br/>808 Netty 接入、会话、位置/批量位置、媒体、透传、下行"]
jt1078["protocol-jt1078<br/>1078 信令 + TCP/UDP RTP 媒体流归档"]
jsatl12["protocol-jsatl12<br/>主动安全附件流接入"]
end
subgraph inbound["入口适配 modules/inbound"]
mqtt["inbound-mqtt<br/>MQTT endpoint 生命周期、PEM TLS"]
mqttProfile["MqttProfileRegistry<br/>按 endpoint profile 解析/映射"]
end
subgraph protocolBase["核心与公共能力 modules/core"]
codec["ingest-codec-common<br/>BCD / CRC / BCC / bit 工具"]
session["session-core<br/>设备会话、命令分发接口、会话存储"]
identity["vehicle-identity<br/>跨协议身份解析、MySQL 绑定表"]
obs["observability<br/>metrics / health"]
api["ingest-api<br/>ProtocolId / RawFrame / VehicleEvent / 注解 SPI"]
registry["HandlerRegistry<br/>按协议、命令、infoType 路由"]
dispatcher["Dispatcher<br/>RawFrame 到 Handler 到 EventBus"]
bus["DisruptorEventBus<br/>高吞吐事件发布"]
end
subgraph sink["输出与明细存储 modules/sinks"]
kafka["sink-kafka<br/>VehicleEvent 到 Protobuf Envelope 到 Kafka"]
archive["sink-archive<br/>RawArchive 到本地文件系统冷存"]
tdengineStore["tdengine-history-store<br/>TDengine raw_frames + locations"]
end
subgraph services["消费服务 modules/services"]
history["event-history-service<br/>Kafka event/raw 到 TDengine 历史查询"]
state["vehicle-state-service<br/>可选 Redis 热状态<br/>optional-latest-state"]
stat["vehicle-stat-service<br/>Kafka 全字段事件到指标表"]
end
subgraph apps["应用入口 modules/apps"]
gbApp["gb32960-ingest-app<br/>GB32960 TCP 接入"]
jtApp["jt808-ingest-app<br/>JT808 TCP 接入"]
yutongApp["yutong-mqtt-app<br/>宇通 MQTT 接入"]
historyApp["vehicle-history-app<br/>历史查询与 TDengine 写入"]
analyticsApp["vehicle-analytics-app<br/>808 日里程指标消费"]
gateway["command-gateway<br/>可选 HTTP 设备命令入口"]
end
vehicle --> gbApp
vehicle --> jtApp
vehicle --> jt1078
vehicle --> jsatl12
mqttSource --> yutongApp
gbApp --> gb
jtApp --> jt808
yutongApp --> mqtt
mqtt --> mqttProfile
mqttProfile --> api
commandClient --> gateway
gb --> codec
jt808 --> codec
jt1078 --> codec
jsatl12 --> codec
gb --> api
jt808 --> api
jsatl12 --> archive
jsatl12 --> jt808
api --> registry
registry --> dispatcher
dispatcher --> bus
bus --> kafka
bus --> archive
kafka --> historyApp
kafka --> analyticsApp
historyApp --> history
analyticsApp --> stat
kafka -.可选热状态.-> state
history --> tdengineStore
gateway --> session
session --> jt808
session --> gb
identity --> jt808
identity --> mqttProfile
obs -.监控.-> dispatcher
obs -.监控.-> bus
obs -.监控.-> kafka
classDef entry fill:#eff6ff,stroke:#2563eb,color:#1e3a8a;
classDef core fill:#ecfdf3,stroke:#0f766e,color:#134e48;
classDef sink fill:#fffbeb,stroke:#b45309,color:#7c2d12;
classDef command fill:#f5f3ff,stroke:#6d28d9,color:#4c1d95;
classDef support fill:#f8fafc,stroke:#64748b,color:#334155;
class gb,jt808,jt1078,jsatl12,mqtt,mqttProfile,gbApp,jtApp,yutongApp,historyApp,analyticsApp entry;
class api,registry,dispatcher,bus,identity core;
class kafka,archive,tdengineStore sink;
class gateway,session command;
class codec,obs support;
</pre>
</div>
<div class="legend">
<span><i class="dot entry"></i>入口/协议模块</span>
<span><i class="dot core"></i>核心管线</span>
<span><i class="dot sink"></i>输出层</span>
<span><i class="dot command"></i>下行命令/会话</span>
</div>
</section>
<section>
<h2>三、上行数据流转</h2>
<div class="diagram">
<pre class="mermaid">
sequenceDiagram
autonumber
participant Device as 车辆/平台
participant Inbound as Inbound Adapter<br/>Netty/MQTT
participant Decoder as FrameDecoder<br/>MessageDecoder
participant Dispatcher as Dispatcher
participant Handler as Protocol Handler<br/>Mapper
participant EventBus as DisruptorEventBus
participant Kafka as sink-kafka<br/>Kafka Envelope
participant Archive as sink-archive<br/>ArchiveStore
participant History as vehicle-history-app
participant TDengine as tdengine-history-store<br/>raw_frames + locations
participant Consumer as 下游业务消费者
Device->>Inbound: 原始报文 bytes / MQTT message
Inbound->>Decoder: 粘包拆帧、校验、协议体解析
Inbound->>Inbound: MQTT 按 endpoint profile 选择厂商解析器
Decoder->>Dispatcher: RawFrame(protocol, command, infoType, payload, rawBytes)
Dispatcher->>EventBus: 先发布 RawArchive 事件
EventBus->>Archive: 保存原始 bytes生成可回放材料
Dispatcher->>Handler: 根据 protocol + command + infoType 路由
Handler->>Handler: 协议字段映射为内部字段
Dispatcher->>Handler: 绑定 rawArchiveKey / rawArchiveUri
Handler->>EventBus: VehicleEvent(Realtime / Location / Alarm / Login ...)
EventBus->>Kafka: 投递规整后的业务事件
Kafka->>Kafka: EnvelopeMapper 转 Protobuf
Kafka->>History: event/raw topickey=vin单车有序
History->>TDengine: 写 raw_frames、locations 和 raw parsed JSON
TDengine->>Consumer: 分页历史查询、原始帧追溯、位置查询
Kafka->>Consumer: Redis 热状态、指标统计、告警工单等下游消费
</pre>
</div>
<div class="callout">
<strong>关键边界:</strong>
协议字段只在协议模块内解释,出了 Mapper 以后都应该使用内部字段。统计每日里程、每日用电量、每日用氢量、储氢安全和氢泄露告警时,应消费全字段 TelemetrySnapshot不要直接依赖某个协议的原始字段名。
</div>
<div class="callout">
<strong>原始追溯:</strong>
Dispatcher 为所有带 rawBytes 的 RawFrame 生成统一的 <code>rawArchiveKey</code>,业务事件 metadata 使用 <code>rawArchiveUri=archive://...</code> 指向同一份原始 bytes。归档文件的真实本地路径只属于 ArchiveStore查询、导出和 Kafka Envelope 使用逻辑 URI 解耦存储实现。
</div>
</section>
<section>
<h2>四、32960 细化链路</h2>
<div class="diagram">
<pre class="mermaid">
flowchart LR
terminal["氢能车 TBOX<br/>GB/T 32960"] --> netty["Gb32960NettyServer<br/>TCP / TLS / Idle 检测"]
netty --> access["Gb32960AccessService<br/>VIN 白名单 / 平台登录鉴权"]
netty --> frame["Gb32960FrameDecoder<br/>拆帧、转义、校验"]
frame --> decoder["Gb32960MessageDecoder<br/>Header + Body 解析"]
decoder --> diag["Gb32960FrameDiagnostics<br/>首帧、Raw 块、异常诊断"]
decoder --> handler["Gb32960ChannelHandler<br/>ACK / NACK / dispatch"]
handler --> ack["Gb32960AckService<br/>登录应答、失败断开"]
handler --> dispatcher["Dispatcher"]
dispatcher --> rtHandler["Gb32960RealtimeHandler"]
rtHandler --> mapper["Gb32960EventMapper<br/>协议字段到内部字段"]
mapper --> realtime["RealtimePayload<br/>速度、里程、电量、氢量、压力、温度"]
mapper --> alarm["AlarmPayload<br/>安全分类、氢泄露、告警等级"]
mapper --> login["Login / Logout / Heartbeat"]
realtime --> bus["DisruptorEventBus"]
alarm --> bus
login --> bus
bus --> kafka["Kafka Envelope"]
bus --> archive["RawArchive 冷存"]
classDef safety fill:#fff1f3,stroke:#b42318,color:#7a271a;
classDef core fill:#ecfdf3,stroke:#0f766e,color:#134e48;
classDef entry fill:#eff6ff,stroke:#2563eb,color:#1e3a8a;
classDef sink fill:#fffbeb,stroke:#b45309,color:#7c2d12;
class terminal,netty,frame,decoder,handler entry;
class access,diag,ack,dispatcher,rtHandler,mapper,bus core;
class alarm safety;
class kafka,archive sink;
</pre>
</div>
</section>
<section>
<h2>五、模块职责表</h2>
<table>
<thead>
<tr>
<th>模块</th>
<th>职责</th>
<th>主要输入</th>
<th>主要输出</th>
<th>业务开发关注点</th>
</tr>
</thead>
<tbody>
<tr>
<td>modules/core/ingest-api</td>
<td>定义系统边界:协议 ID、RawFrame、IngestContext、VehicleEvent、Payload、Handler 注解、Sink SPIProtocolId 提供 UNKNOWN 隔离桶供消费侧兼容未来协议或脏数据不作为正式接入协议使用consumer 包定义 EnvelopeIngestor、EnvelopeConsumerProcessor 和 EnvelopeDeadLetterSink统一 Kafka 消费结果与死信边界。</td>
<td>无直接外部输入。</td>
<td>全项目共享的接口和内部事件模型。</td>
<td><span class="tag red">内部字段定义</span><span class="tag">新增事件类型</span><span class="tag">字段兼容性</span></td>
</tr>
<tr>
<td>modules/core/ingest-codec-common</td>
<td>公共编解码工具,承载 BCD、CRC、BCC、bit 操作等协议底层能力。</td>
<td>协议模块传入的 byte、bit、校验材料。</td>
<td>解析辅助结果。</td>
<td><span class="tag">协议基础能力</span></td>
</tr>
<tr>
<td>modules/core/ingest-core</td>
<td>核心管线。扫描 Handler按协议和命令路由执行拦截器发布 VehicleEvent 到 Disruptor。</td>
<td>RawFrame。</td>
<td>VehicleEvent、RawArchive。</td>
<td><span class="tag green">吞吐</span><span class="tag green">路由</span><span class="tag amber">原始可回放</span></td>
</tr>
<tr>
<td>modules/core/session-core</td>
<td>设备会话、命令下发抽象、SessionStore、CommandDispatcher 默认实现。SessionStore 仅保留 Redis 后端;协议进程本地只保存真实 Netty ChannelRedis 保存 <code>sessionId</code>、VIN 和 phone 三索引及 TTL协议模块只依赖 SPI不感知存储细节。</td>
<td>设备连接、命令请求。</td>
<td>会话状态、命令分发结果、Redis 会话索引。</td>
<td><span class="tag">在线状态</span><span class="tag">下行控制</span><span class="tag green">多实例会话索引</span></td>
</tr>
<tr>
<td>modules/core/vehicle-identity</td>
<td>跨协议车辆身份解析,维护 phone、deviceId、plate 到 VIN 的绑定关系;生产运行使用 MySQL 绑定表,避免重启丢失绑定并保证多实例一致。</td>
<td>协议模块传入的外部标识。</td>
<td>稳定 VIN、解析来源、是否已命中绑定、MySQL 绑定记录。</td>
<td><span class="tag red">统计主键</span><span class="tag">多协议归一</span><span class="tag green">持久化绑定</span></td>
</tr>
<tr>
<td>modules/core/observability</td>
<td>监控、指标、健康检查等横切能力。</td>
<td>核心管线和 Sink 运行状态。</td>
<td>Micrometer 指标、健康信息。</td>
<td><span class="tag green">可用性</span><span class="tag">延迟监控</span></td>
</tr>
<tr>
<td>modules/protocols/protocol-gb32960</td>
<td>GB/T 32960 接入、鉴权、ACK、报文解析、实时数据和告警映射。</td>
<td>32960 TCP 报文。</td>
<td>Realtime、Location、Alarm、Login、Logout、Heartbeat、RawFrame。</td>
<td><span class="tag red">氢泄露</span><span class="tag red">储罐安全</span><span class="tag">电量/氢量</span><span class="tag">里程</span></td>
</tr>
<tr>
<td>modules/protocols/protocol-jt808</td>
<td>JT/T 808 接入、注册/鉴权/心跳/注销、位置/位置附加项、批量位置、参数/属性、多媒体、透传、未知上行兜底透传、分帧边界异常和协议解析异常坏帧兜底、下行命令和超长下行分包能力0x0200 位置附加项 0x01 总里程按 0.1km 解码为内部字段 <code>total_mileage_km</code>,后续每日里程统计不依赖 808 原始字段0x0102 鉴权帧解析会保留完整 token并尽量提取 IMEI 和软件版本;注册和鉴权会话均使用共享身份解析后的内部 VIN避免 deviceId/IMEI 污染后续会话和下行边界;终端连接断开时同步解绑 Channel 并清理 <code>SessionStore</code> 会话,避免 command-gateway 误判离线车辆仍可下行;正常上行 RawArchive 和业务事件 metadata 均使用共享身份解析后的内部 vin并保留 phone、identityResolved、identitySource 便于冷存和业务事件按同一车辆查询;无法解析终端身份的坏帧 RawArchive 和 Passthrough 均按 vin=unknown、identityResolved=false、identitySource=UNKNOWN 标记0x0900 透传事件按原始消息 ID 归类,透传类型写入 passthroughType metadata。</td>
<td>808 TCP 报文。</td>
<td>Location、Login、Logout、Heartbeat、MediaMeta、Passthrough、未知上行 Raw 兜底、坏帧 Passthrough、会话状态、下行响应。</td>
<td><span class="tag">GPS 位置</span><span class="tag">在线状态</span><span class="tag">媒体证据</span><span class="tag">命令链路</span></td>
</tr>
<tr>
<td>modules/protocols/protocol-jt1078</td>
<td>JT/T 1078 音视频能力。信令在 JT808 mapper 存在时桥接复用 JT808 连接和包头0x1005 乘客流量保留原始 body同时结构化 channelId、startTime、endTime、passengerGetOn、passengerGetOff metadata 便于查询/导出0x1205 文件列表保留原始 body同时结构化 responseSerialNo、fileCount 摘要;下行信令编码由本模块提供,覆盖 0x1003 音视频属性查询、0x9101 实时预览、0x9102 实时控制、0x9201 历史回放、0x9202 回放控制、0x9205 资源列表查询、0x9206 文件上传和 0x9207 文件上传控制,命令发送仍由 command-gateway 经 CommandDispatcher 复用 JT808 在线通道TCP/UDP RTP 媒体流独立端口接入,生产默认端口按旧接收服务对齐为 <code>11078</code>,按 VIN/通道/时间分段写 ArchiveStoreKafka 只发送 MediaMeta 引用,事件 metadata 暴露内部 vin、sim、channelId、dataType、packetType、segment、sequence、segmentSizeBytes、archiveKey 和 archiveRefMediaMeta.sizeBytes 同步当前片段大小MediaMeta 按 archiveKey 去重,避免同一 SIM 运行中从 fallback 身份切换到内部 VIN 时漏发新归档引用,便于文件明细库追踪、展示和导出;超长/短包/坏魔数等 RTP 入口异常和媒体归档失败都会兜底产出 Passthrough 错误事件,坏 RTP 统一走 Dispatcher 产出 RawArchive 和 Passthrough统一标记 parseError 和 parseErrorMessage并保留内部 vin、原始 bytes、peer、长度、原因、RawArchive 引用、segment 和 archiveKey 便于排障;坏 RTP 头部可读时会提取 SIM 并通过共享身份服务映射 VIN。</td>
<td>1078 信令报文、TCP/UDP RTP 媒体流。</td>
<td>MediaMeta、乘客流量 Passthrough、文件列表 Passthrough、坏 RTP/归档失败 Passthrough、归档媒体分段。</td>
<td><span class="tag">视频扩展</span><span class="tag green">信令事件化</span><span class="tag amber">大流冷存</span></td>
</tr>
<tr>
<td>modules/protocols/protocol-jsatl12</td>
<td>苏标主动安全报警附件接入(仅 <code>optional-attachments</code> profile 显式构建,不属于默认生产面)。依赖 ArchiveStore 和 JT808 decoder端口按旧接收服务对齐为 <code>7612</code>;附件 DataPacket 按旧服务真实布局 <code>01cd + 50B文件名 + offset + length + data</code> 分帧解析,不再把文件名前 4 字节误判为帧长;附件数据写 ArchiveStore 后通过 Dispatcher 产出 MediaMeta 引用事件,事件 metadata 保留 fileName、fileOffset、declaredChunkSizeBytes、archiveKey 和 archiveRef便于附件明细查询、导出和补传诊断内置附件状态机处理 T1210 文件清单、DataPacket 分块区间和 T1212 上传完成消息,数据分块会按 T1210 文件清单中的文件名继承手机号并通过共享身份解析器得到内部 VIN入口 RawFrame、ArchivedChunk 和 MediaMeta 均保留 phone、vin、identityResolved、identitySource只有找不到文件归属时才标记 UNKNOWN按文件名合并无身份数据分块并返回 0x9212 完成/补传应答,补传应答包含缺失 offset/length 区间;附件归档失败会产出 JSATL12 Passthrough 错误事件并保留原始分块JT_MESSAGE 信令复用 JT808 decoder 后进入 Dispatcher桥接 RawFrame 使用共享身份解析后的内部 VIN并保留 phone、identityResolved、identitySource坏 JT 信令入口 RawFrame 同样标记 UNKNOWN 身份;超长/畸形附件帧、未闭合超长 JT 信令和坏 JT 信令都以 JSATL12 Passthrough 兜底,附件帧异常同时标记 frameError 与统一 parseError/parseErrorMessage保留原始字节、peer、长度和错误元数据帧长和 worker 线程支持部署配置。</td>
<td>主动安全附件流、JT808 信令帧。</td>
<td>归档文件、MediaMeta、JT808 统一事件、坏帧/坏信令/归档失败 Passthrough。</td>
<td><span class="tag amber">附件归档</span><span class="tag">报警证据</span><span class="tag">信令复用</span><span class="tag">错误隔离</span></td>
</tr>
<tr>
<td>modules/inbound/inbound-mqtt</td>
<td>MQTT 多 endpoint 生命周期管理,支持 PEM CA、客户端证书/私钥双向 TLS按 endpoint profile 路由到厂商解析/映射实现,默认提供宇通 profile生产默认接入口径参考旧接收服务<code>ssl://cpxlm.axxc.cn:38883</code>、topic <code>/ytforward/shln/+</code>、QoS 2、cleanSession=false、keepAlive=20s、connectTimeout=10s用户名和密码仍通过环境变量注入入站 RawFrame 和事件映射均接入统一车辆身份解析,解析器保留 externalVin、phone、mqttDeviceId、plateNo优先按设备号/终端 ID/IMEI 绑定解析内部 VIN未绑定且字段形态为 VIN 时再作为显式 VIN 兜底,冷存和事件 metadata 均保留这些外部标识、identityResolved、identitySource未知 profile、解析失败、profile 运行时异常、endpoint 初始化/连接/订阅失败均兜底为 Passthrough解析失败诊断和 profile 异常诊断都会同步写入 RawArchive metadata并在最终 Passthrough 事件保留 parseErrorMessage 或 operational reason无法解析设备身份的兜底事件和运行类归档都按 vin=unknown、identityResolved=false、identitySource=UNKNOWN 标记,保留原始 payload 或 operational 错误信息便于排障和归档追溯。</td>
<td>MQTT topic/message、endpoint profile、TLS PEM 配置。</td>
<td>统一 RawFrame、Realtime、Location、Passthrough。</td>
<td><span class="tag">车企平台接入</span><span class="tag green">profile 扩展</span><span class="tag green">身份映射</span><span class="tag green">双向 TLS</span><span class="tag green">错误隔离</span></td>
</tr>
<tr>
<td>modules/sinks/sink-kafka</td>
<td>Kafka Sink。把 VehicleEvent 转成 Protobuf Envelope按 VIN 分区投递;业务事件携带全字段 TelemetrySnapshot 和 rawArchiveUri显式 RawArchive Envelope 会填充 archive:// 逻辑 URI 与 size默认 Kafka Sink 仍不发送原始 bytes 本体;同时提供 KafkaEnvelopeConsumerFactory、KafkaEnvelopeConsumerRunner、KafkaEnvelopeConsumerWorker 和 KafkaEnvelopeDeadLetterSink统一服务侧 Kafka 消费启动、独立 group 绑定、死信发布和 topic 到 EnvelopeConsumerProcessor 的分发。</td>
<td>VehicleEvent。</td>
<td>Kafka Protobuf 消息、消费侧 DLQ 记录。</td>
<td><span class="tag green">单车有序</span><span class="tag">Schema 演进</span><span class="tag red">安全字段下发给消费者</span></td>
</tr>
<tr>
<td>modules/sinks/sink-archive</td>
<td>原始报文冷存,当前为本地文件系统实现;写入键与业务事件 metadata 中的 <code>rawArchiveKey</code> 保持一致。</td>
<td>RawArchive 事件、附件流。</td>
<td>可回放原始文件、<code>archive://...</code> 逻辑引用。</td>
<td><span class="tag amber">问题排查</span><span class="tag amber">合规留痕</span></td>
</tr>
<tr>
<td>modules/sinks/tdengine-history-store</td>
<td>生产历史库边界,负责 TDengine schema、raw_frames/location 行映射、批量写入和分页查询语句raw_frames 保留完整 payloadJson.parsed位置表保持轻量字段并通过 rawUri 关联原始帧。</td>
<td>Kafka Protobuf Envelope、Raw Envelope。</td>
<td>TDengine <code>raw_frames</code> 和位置表。</td>
<td><span class="tag green">生产历史</span><span class="tag green">分页查询</span><span class="tag amber">raw 追溯</span></td>
</tr>
<tr>
<td>modules/services/event-history-service</td>
<td>消费 32960、808、宇通 MQTT 的 Kafka event/raw topic生产默认写入 TDengine raw_frames 和位置表,并提供 raw 帧、位置历史等分页查询边界Raw 查询返回完整 parsed JSON位置历史只保留核心字段并通过 rawUri 关联原始帧;生产消费可使用 <code>tryIngest</code> 和自动装配的 <code>EnvelopeConsumerProcessor</code> 将坏 protobuf、缺快照和存储异常收敛为结构化结果并通过 <code>EnvelopeDeadLetterSink</code> 发布死信,避免阻塞 Kafka 分区。</td>
<td>Kafka Protobuf Envelope。</td>
<td>TDengine raw_frames、位置分页结果、raw archive 逻辑引用。</td>
<td><span class="tag green">历史明细</span><span class="tag green">分页查询</span><span class="tag amber">raw JSON</span></td>
</tr>
<tr>
<td>modules/services/vehicle-state-service</td>
<td>消费 Kafka 全字段事件,更新车辆最新状态、位置、安全和最后事件到 Redis仅通过 <code>optional-latest-state</code> profile 显式构建,不属于默认生产 reactor生产消费可使用 <code>tryIngest</code> 和自动装配的 <code>EnvelopeConsumerProcessor</code> 隔离坏 protobuf 和缺快照 Envelope并把失败记录交给死信出口。</td>
<td>Kafka Protobuf Envelope。</td>
<td><code>vehicle:state:{vin}</code><code>vehicle:location:{vin}</code><code>vehicle:safety:{vin}</code></td>
<td><span class="tag green">毫秒级热查询</span><span class="tag red">氢泄露安全</span></td>
</tr>
<tr>
<td>modules/services/vehicle-stat-service</td>
<td>消费 Kafka 全字段事件808 每日里程只使用 0x0200 附加项 0x01 上报的总里程做当日首末差值;生产消费可使用 <code>tryIngest</code> 和自动装配的 <code>EnvelopeConsumerProcessor</code> 隔离坏 protobuf缺少 telemetry_snapshot 的消息会标记为 SKIPPED 并进入死信出口。</td>
<td>Kafka Protobuf Envelope、808 <code>total_mileage_km</code></td>
<td><code>vehicle_stat_metric</code><code>metric_key=daily_mileage_km</code><code>calculation_method=JT808_TOTAL_MILEAGE_DIFF</code></td>
<td><span class="tag green">每日里程</span><span class="tag">指标表</span></td>
</tr>
<tr>
<td>modules/apps/command-gateway</td>
<td>可选 HTTP 到设备下行命令入口。支持会话查询、808 位置查询、参数查询/设置、终端控制、平台通用应答、终端属性查询 0x8107、区域删除 0x8601、人工报警确认 0x8203以及 1078 音视频属性查询、实时预览/控制、历史回放/控制、资源列表查询、文件上传/控制;通过 CommandDispatcher 复用 protocol-jt808 的在线通道、流水号和同步应答等待能力。</td>
<td>业务系统命令请求。</td>
<td>CommandDispatcher 调用、设备应答摘要。</td>
<td><span class="tag">远程控制</span><span class="tag">参数设置</span><span class="tag">属性查询</span><span class="tag">区域删除</span><span class="tag">报警确认</span><span class="tag">视频控制</span></td>
</tr>
<tr>
<td>modules/apps/gb32960-ingest-app</td>
<td>生产 GB32960 TCP 接入应用,只负责接收、鉴权、解析、冷存引用和 Kafka 投递。</td>
<td>GB/T 32960 TCP 32960。</td>
<td><code>vehicle.raw.gb32960.v1</code><code>vehicle.event.gb32960.v1</code></td>
<td><span class="tag green">生产接入</span><span class="tag">32960</span></td>
</tr>
<tr>
<td>modules/apps/jt808-ingest-app</td>
<td>生产 JT808 TCP 接入应用,只负责 808 注册/鉴权/位置等上行解析、冷存引用和 Kafka 投递。</td>
<td>JT/T 808 TCP 808。</td>
<td><code>vehicle.raw.jt808.v1</code><code>vehicle.event.jt808.v1</code></td>
<td><span class="tag green">生产接入</span><span class="tag">808</span></td>
</tr>
<tr>
<td>modules/apps/yutong-mqtt-app</td>
<td>生产宇通 MQTT 接入应用,按 endpoint profile 解析 MQTT 报文并投递 Kafka。</td>
<td>宇通 MQTT topic。</td>
<td><code>vehicle.raw.mqtt-yutong.v1</code><code>vehicle.event.mqtt-yutong.v1</code></td>
<td><span class="tag green">生产接入</span><span class="tag">MQTT</span></td>
</tr>
<tr>
<td>modules/apps/vehicle-history-app</td>
<td>生产历史应用,消费 32960、808、宇通 MQTT 的 raw/event Kafka Envelope写 TDengine 并提供历史查询。</td>
<td>Kafka Protobuf Envelope。</td>
<td>TDengine <code>raw_frames</code>、位置表和历史查询 API。</td>
<td><span class="tag green">历史查询</span><span class="tag amber">raw JSON</span></td>
</tr>
<tr>
<td>modules/apps/vehicle-analytics-app</td>
<td>生产指标应用,当前只消费 JT808 事件并用 GPS 总里程首末差值写每日里程指标。</td>
<td><code>vehicle.event.jt808.v1</code></td>
<td>MySQL <code>vehicle_stat_metric</code></td>
<td><span class="tag green">每日里程</span><span class="tag">指标表</span></td>
</tr>
</tbody>
</table>
</section>
<section>
<h2>六、氢能业务开发落点</h2>
<table>
<thead>
<tr>
<th>业务主题</th>
<th>当前建议落点</th>
<th>说明</th>
</tr>
</thead>
<tbody>
<tr>
<td>车辆状态、速度、位置</td>
<td><code>RealtimePayload</code><code>LocationPayload</code></td>
<td>32960 和 808 都可能提供位置。统计侧需要明确优先级,例如优先 32960缺失时用 808 补点。</td>
</tr>
<tr>
<td>每日里程</td>
<td>Kafka 下游统计服务</td>
<td>接入层只投递总里程和实时点位;日增里程建议下游按 VIN、自然日、事件时间聚合处理回补和乱序。</td>
</tr>
<tr>
<td>每日用电量</td>
<td>内部电池字段 + 下游统计</td>
<td>接入层保留 SOC、电压、电流等瞬时字段若有累计电耗字段可加入内部字段模型后统一投递。</td>
</tr>
<tr>
<td>每日用氢量</td>
<td><code>hydrogenRemainingKg</code> + 下游统计</td>
<td>优先使用累计氢耗字段;没有累计值时用氢余量差值估算,并在统计结果里标记估算口径。</td>
</tr>
<tr>
<td>储氢安全</td>
<td><code>RealtimePayload</code> 压力/温度 + <code>AlarmPayload.safetyCategory</code></td>
<td>高压、低压、温度、异常 bit 都应统一归入储罐安全域,下游可做趋势、阈值和连续异常判断。</td>
</tr>
<tr>
<td>氢气泄露</td>
<td><code>AlarmPayload.hydrogenLeakDetected</code></td>
<td>当前已作为高优先级安全事件。32960 中出现 <code>HYDROGEN_LEAK</code> 时强制映射为 <code>CRITICAL</code></td>
</tr>
<tr>
<td>原始报文追溯</td>
<td><code>RawArchive</code> + <code>rawArchiveUri</code> + <code>sink-archive</code></td>
<td>业务统计出现争议时,用事件 metadata 中的 <code>archive://...</code> 逻辑 URI 回查原始 bytes结合 eventId、traceId、VIN、时间复现解析链路。</td>
</tr>
<tr>
<td>明细展示和导出</td>
<td><code>vehicle-history-app</code> + <code>tdengine-history-store</code></td>
<td>生产查询通过 TDengine raw_frames、位置表和 rawUri 关联完成;不再维护文件型事件索引旁路。</td>
</tr>
</tbody>
</table>
<div class="callout">
<strong>后续开发建议:</strong>
当前 Kafka Protobuf 已加入全字段 <code>TelemetrySnapshot</code>。后续业务统计服务继续只依赖内部字段,不依赖 32960 原始字段名,这样接入 808、MQTT、车企私有协议时不会重写统计口径。
</div>
</section>
</main>
<script type="module">
import mermaid from "https://cdn.jsdelivr.net/npm/mermaid@10/dist/mermaid.esm.min.mjs";
mermaid.initialize({
startOnLoad: true,
theme: "base",
securityLevel: "strict",
flowchart: {
htmlLabels: true,
curve: "basis"
},
themeVariables: {
fontFamily: "-apple-system, BlinkMacSystemFont, Segoe UI, PingFang SC, Microsoft YaHei, sans-serif",
primaryColor: "#eff6ff",
primaryTextColor: "#172033",
primaryBorderColor: "#2563eb",
lineColor: "#667085",
secondaryColor: "#ecfdf3",
tertiaryColor: "#fffbeb"
}
});
</script>
</body>
</html>

View File

@@ -1,227 +0,0 @@
# 当前 ECS Go 原生部署说明
更新时间2026-07-02 01:30 CST
本文档记录当前 `lingniu-vehicle-ingest` 的生产运行面。当前 goal 的接入链路已经切到 Go 原生 systemd 部署,不再以 Docker/Portainer 作为生产运行方式。密钥只维护在 ECS 环境文件或受控凭据中,不写入 Git。
## 部署范围
| systemd 服务 | 二进制 | 说明 | 对外端口 |
| --- | --- | --- | --- |
| `lingniu-go-gateway.service` | `gateway` | GB32960 TCP、JT808 TCP、宇通 MQTT 接入,解析后写 Kafka RAW 和统一事件 | TCP `32960`、TCP `808` |
| `lingniu-go-history-writer.service` | `history-writer` | 消费 Kafka RAW写 TDengine RAW、位置点、里程点核心表 | 无 HTTP |
| `lingniu-go-stat-writer.service` | `stat-writer` | 消费 Kafka RAW按总里程差值法写 MySQL 每日指标 | 无 HTTP |
| `lingniu-go-realtime-api.service` | `realtime-api` | 消费统一事件写 Redis并提供实时、历史、统计查询 API | HTTP `20210` |
信达 Push 已废弃,不参与当前 Go 生产链路。旧 Java/Docker 服务不应占用 `808``32960``20210`
## 应用 ECS
| 项 | 值 |
| --- | --- |
| 公网 IP | `115.29.187.205` |
| 登录用户 | `root` |
| 部署目录 | `/opt/lingniu-go-native` |
| 当前 release | `/opt/lingniu-go-native/current` |
| 环境文件目录 | `/opt/lingniu-go-native/env` |
| systemd unit 目录 | `/etc/systemd/system` |
| 对外服务 | `32960``808``20210` |
运行目录结构:
```text
/opt/lingniu-go-native/
current -> /opt/lingniu-go-native/releases/<git-short-sha>
releases/<git-short-sha>/
gateway
history-writer
stat-writer
realtime-api
env/
gateway.env
history-writer.env
stat-writer.env
realtime-api.env
spool/gateway/
```
## 中间件
| 组件 | 内网地址 | 公网地址 | 用途 |
| --- | --- | --- | --- |
| Kafka | `172.17.111.56:9092` | `114.55.58.251:9092` | RAW 和统一事件消息总线 |
| TDengine | `172.17.111.57:6041` | `115.29.185.82:6041` | RAW、位置、里程点时序热存储 |
| MySQL RDS | `rm-bp179zbv481rnw3e2.mysql.rds.aliyuncs.com:3306` | `rm-bp179zbv481rnw3e2no.mysql.rds.aliyuncs.com:3306` | 身份映射、JT808 注册、每日指标 |
| Redis RDS | `r-bp1u741kij7e51i481.redis.rds.aliyuncs.com:6379` | 无 | 准实时车辆状态缓存 |
生产应用之间访问中间件优先使用内网地址。RDS 白名单需要允许应用 ECS 私网访问。
## Kafka Topic
| Topic | 生产者 | 消费者 | 内容 |
| --- | --- | --- | --- |
| `vehicle.raw.go.gb32960.v1` | `gateway` | `history-writer``stat-writer` | GB32960 完整 RAW 记录 |
| `vehicle.raw.go.jt808.v1` | `gateway` | `history-writer``stat-writer` | JT808 完整 RAW 记录 |
| `vehicle.raw.go.yutong-mqtt.v1` | `gateway` | `history-writer` | 宇通 MQTT 完整 RAW 记录 |
| `vehicle.event.go.unified.v1` | `gateway` | `realtime-api` | 三类协议统一事件,用于 Redis 实时状态 |
`gateway` 配置了 Kafka retry生产 env 中启用 `KAFKA_SPOOL_DIR` 时,本地 spool 可在 Kafka 短暂不可用后回放。
Kafka 消费组:
| Consumer Group | Topic | 当前验证 |
| --- | --- | --- |
| `go-history-writer` | 三个 `vehicle.raw.go.*` topic | 2026-07-02 验证 lag 为 `0` |
| `go-stat-writer` | `vehicle.raw.go.gb32960.v1``vehicle.raw.go.jt808.v1` | 2026-07-02 验证 lag 为 `0` |
| `go-realtime-api` | `vehicle.event.go.unified.v1` | 2026-07-02 验证 lag 为 `0` |
可重复 smoke
```bash
python3 tools/go_kafka_prod_smoke.py --host 114.55.58.251 --user root --max-lag 100
```
该脚本通过 SSH 登录 Kafka ECS检查 Go 生产 topic 是否存在,并校验 `go-history-writer``go-stat-writer``go-realtime-api` 的消费组 lag。运行环境需要已配置 SSH 免密或已建立可用的 SSH 认证方式。
## TDengine 数据层
默认数据库:`lingniu_vehicle_ts`
| 表 | 类型 | 说明 |
| --- | --- | --- |
| `raw_frames` | stable | 完整 RAW 帧,包含 `raw_hex``parsed_json``fields_json``source_endpoint` 和协议/车辆 tags |
| `vehicle_locations` | stable | 最小化位置历史,保留 `frame_id` 回链 RAW |
| `vehicle_mileage_points` | stable | 最小化总里程点,供历史查询和统计复核 |
设计原则:
- 完整结构化 payload 只落在 `raw_frames.parsed_json` / `fields_json`
- `vehicle_locations``vehicle_mileage_points` 只保留核心查询字段,不重复保存完整 JSON。
- `telemetry_fields` 不由当前服务维护,字段配置解析由独立子服务处理。
- API 传入东八区时间时,服务会转换为 TDengine 当前存储使用的 UTC 字面量再查询。
## MySQL 数据层
默认业务库:`lingniu_vehicle_data`
| 表 | 维护方 | 说明 |
| --- | --- | --- |
| `vehicle_identity_binding` | 人工/外部主数据 | VIN 映射主表,用 `phone``device_id``plate` 降级定位 VIN |
| `jt808_registration` | `gateway` 自动写入 | 仅 JT808 注册和鉴权记录,以 `phone` 作为主键,记录设备、车牌、厂家、鉴权、来源端点 |
| `vehicle_daily_metric` | `stat-writer` 自动写入 | 每日指标表,保存 `daily_mileage_km``daily_total_mileage_km` |
每日里程算法:
```text
daily_mileage_km = 当日最大 total_mileage_km - 当日最小 total_mileage_km
daily_total_mileage_km = 当日最大 total_mileage_km
```
统计按 `Asia/Shanghai` 自然日归档,当前支持 GB32960、JT808、宇通 MQTT 中能解析出 `total_mileage_km` 的数据。JT808 的总里程来自位置附加信息 `0x01`,单位按协议转换为 km宇通 MQTT 的 `TOTAL_MILEAGE` 原始单位为米,入库前转换为 km。
## Redis 实时层
`realtime-api` 消费 `vehicle.event.go.unified.v1`,按 VIN/vehicle key 维护准实时快照。主要能力:
| API | 说明 |
| --- | --- |
| `/api/realtime/vehicles/{vin}` | 查询车辆合并实时快照 |
| `/api/realtime/vehicles/{vin}/online` | 查询车辆是否在线 |
| `/api/realtime/vehicles/{vin}/protocols/{protocol}` | 查询指定协议的实时快照 |
Redis 只作为实时缓存;历史和统计以 TDengine/MySQL 为准。
## API 入口
公网基础地址:
```text
http://115.29.187.205:20210
```
常用查询:
```bash
curl -sS 'http://115.29.187.205:20210/api/history/raw-frames?protocol=GB32960&dateFrom=2026-07-02T00:00:00%2B08:00&dateTo=2026-07-03T00:00:00%2B08:00&limit=1'
curl -sS 'http://115.29.187.205:20210/api/history/raw-frames?protocol=JT808&dateFrom=2026-07-02T00:00:00%2B08:00&dateTo=2026-07-03T00:00:00%2B08:00&limit=1'
curl -sS 'http://115.29.187.205:20210/api/history/raw-frames?protocol=YUTONG_MQTT&dateFrom=2026-07-02T00:00:00%2B08:00&dateTo=2026-07-03T00:00:00%2B08:00&limit=1'
curl -sS 'http://115.29.187.205:20210/api/history/locations?protocol=JT808&dateFrom=2026-07-02T00:00:00%2B08:00&dateTo=2026-07-03T00:00:00%2B08:00&limit=1'
curl -sS 'http://115.29.187.205:20210/api/history/mileage-points?protocol=JT808&dateFrom=2026-07-02T00:00:00%2B08:00&dateTo=2026-07-03T00:00:00%2B08:00&limit=1'
curl -sS 'http://115.29.187.205:20210/api/stats/daily-metrics?dateFrom=2026-07-02&dateTo=2026-07-02&limit=20'
```
分页参数统一使用 `limit``offset`
部署后推荐直接运行 Go 原生生产 smoke
```bash
python3 tools/go_prod_acceptance.py --date 2026-07-02
```
该聚合脚本会串行执行 systemd/端口检查、Kafka topic/consumer lag 检查、HTTP 数据链路检查,任一子检查失败时整体退出码为非 0。需要 SSH 可登录应用 ECS 和 Kafka ECS。
也可以单独运行子检查:
```bash
python3 tools/go_systemd_prod_smoke.py --host 115.29.187.205 --user root
python3 tools/go_native_prod_smoke.py --date 2026-07-02 --timeout 8
```
`go_systemd_prod_smoke.py` 通过 SSH 检查四个 Go systemd 服务是否 `active``enabled`,并确认 `808``32960``gateway` 监听、`20210``realtime-api` 监听,同时要求 `/opt/lingniu-go-native/current` 指向有效 release、四个生产二进制均可执行、`/opt/lingniu-go-native/spool/gateway` 没有待回放文件,避免旧 Java/Docker 进程占用生产端口、部署目录错乱或 Kafka spool 静默堆积。运行环境需要已配置 SSH 免密或已建立可用的 SSH 认证方式。
`go_native_prod_smoke.py` 通过 `20210` HTTP API 验证 GB32960、JT808、宇通 MQTT 的 RAW 查询及结构化 `parsed_json`,三类协议的位置和里程点查询及 `frame_id` RAW 回链GB32960/JT808 的 `daily_mileage_km``daily_total_mileage_km` 统计公式,以及三类协议的 Redis realtime snapshot、online、protocol 查询。实时 snapshot/protocol 查询会校验 `updated_at_ms` 新鲜度;默认检查今天东八区数据时,还会要求三类 RAW 最新样本不超过 15 分钟,避免旧数据误判为接收正常;任一检查不达标时退出码为非 0。
## 部署命令
推荐使用仓库脚本发布完整 Go 原生 release。脚本会在本机/CI 构建 Linux amd64 四个二进制,打包上传到应用 ECS切换 `/opt/lingniu-go-native/current`,逐个重启四个 systemd 服务,等待 gateway Kafka spool 清空,并在发布后运行聚合生产验收:
```bash
python3 tools/go_native_deploy.py --date 2026-07-02
```
脚本不会保存 SSH 或数据库密钥;认证仍使用当前操作环境可用的 SSH 凭据。需要跳过发布后验收时可显式追加 `--skip-acceptance`,但生产发布默认应保留验收。发布后 spool 默认最多等待 `900` 秒,可用 `--spool-drain-timeout``--spool-poll-interval` 调整。
脚本执行的发布动作等价于:
```bash
CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags='-s -w' ./cmd/gateway
CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags='-s -w' ./cmd/history-writer
CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags='-s -w' ./cmd/stat-writer
CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags='-s -w' ./cmd/realtime-api
scp release.tar.gz root@115.29.187.205:/tmp/
ln -sfn /opt/lingniu-go-native/releases/<git-short-sha> /opt/lingniu-go-native/current
systemctl daemon-reload
systemctl restart lingniu-go-gateway lingniu-go-history-writer lingniu-go-stat-writer lingniu-go-realtime-api
```
只更新查询 API 时,可以仅替换 `/opt/lingniu-go-native/current/realtime-api` 并重启:
```bash
systemctl restart lingniu-go-realtime-api.service
```
## 运行检查
```bash
python3 tools/go_systemd_prod_smoke.py --host 115.29.187.205 --user root
systemctl is-active lingniu-go-gateway lingniu-go-history-writer lingniu-go-stat-writer lingniu-go-realtime-api
ss -lntp | egrep ':(808|32960|20210)\b'
journalctl -u lingniu-go-gateway --since '5 minutes ago' --no-pager
journalctl -u lingniu-go-history-writer --since '5 minutes ago' --no-pager
journalctl -u lingniu-go-stat-writer --since '5 minutes ago' --no-pager
journalctl -u lingniu-go-realtime-api --since '5 minutes ago' --no-pager
```
2026-07-02 01:30 CST 已验证:
- 四个 Go systemd 服务均为 `active``enabled`
- `/opt/lingniu-go-native/current` 指向有效 release四个生产二进制均存在且可执行。
- `gateway` 正在监听 `32960``808`
- `realtime-api` 正在监听 `20210`
- gateway Kafka spool 当前无待回放文件。
- GB32960、JT808、YUTONG_MQTT 的东八区 RAW 查询均可命中生产数据,且最新 RAW 样本包含结构化 `parsed_json`
- GB32960、JT808、YUTONG_MQTT 的最新 RAW 样本均在 15 分钟内。
- GB32960、JT808、YUTONG_MQTT 的 `vehicle_locations``vehicle_mileage_points` 查询均可命中生产数据,且样本包含 `frame_id` 回链 RAW。
- `vehicle_daily_metric` 可查到 `daily_mileage_km``daily_total_mileage_km`,且公式满足 `daily_mileage_km = latest_total_mileage_km - first_total_mileage_km``daily_total_mileage_km = latest_total_mileage_km`
- Redis realtime 可查到 GB32960、JT808、YUTONG_MQTT 的在线状态和协议快照,且 snapshot/protocol 快照最近刷新。

View File

@@ -1,273 +0,0 @@
# GB32960 Split Service Runbook
This runbook covers the default production runtimes: GB32960 ingest, JT808 ingest, Yutong MQTT ingest, history, and analytics. Kafka is the durability boundary between protocol ingest services and downstream business services.
## Service Roles
| Service | Module | Role | Production ports |
| --- | --- | --- | --- |
| GB32960 ingest | `:gb32960-ingest-app` | Accept GB32960 TCP connections, decode/authenticate frames, produce raw and normalized records to Kafka, and send protocol ACKs after required Kafka production succeeds. | TCP `32960`, HTTP `20100` |
| JT808 ingest | `:jt808-ingest-app` | Accept JT/T 808 TCP connections on the production receive port, parse raw frames, archive frame bytes, and produce raw/event envelopes to Kafka. | TCP `808`, HTTP `20400` |
| Yutong MQTT ingest | `:yutong-mqtt-app` | Connect to Yutong MQTT, parse subscribed messages, archive payloads, and produce raw/event envelopes to Kafka. | HTTP `20500` |
| Vehicle history | `:vehicle-history-app` | Consume GB32960/JT808/Yutong MQTT raw/event Kafka records, store compact history rows plus TDengine `raw_frames` `parsedJson`/`metadataJson`/`rawUri`, and expose TDengine-backed history APIs. | HTTP `20200` |
| Vehicle analytics | `:vehicle-analytics-app` | Consume JT808 event Kafka records, calculate daily mileage from reported GPS total mileage, and write metrics to `vehicle_stat_metric`. | HTTP `20310` |
## Kafka Contract
| Topic | Producer | Consumers | Purpose |
| --- | --- | --- | --- |
| `vehicle.raw.gb32960.v1` | `gb32960-ingest-app` | `vehicle-history-app` | Raw frame records. History writes the parsed JSON, metadata JSON, and raw URI to TDengine `raw_frames`. |
| `vehicle.event.gb32960.v1` | `gb32960-ingest-app` | `vehicle-history-app` | Normalized vehicle events keyed by VIN when available. |
| `vehicle.dlq.gb32960.v1` | Kafka producer/consumer error paths | Operators/replay tooling | Dead-letter records for failed production or consumer processing. |
| `vehicle.raw.jt808.v1` | `jt808-ingest-app` | `vehicle-history-app` | JT808 raw frame records for TDengine `raw_frames` and raw-frame troubleshooting APIs. |
| `vehicle.event.jt808.v1` | `jt808-ingest-app` | `vehicle-history-app`, `vehicle-analytics-app` | JT808 parsed location/session/alarm events keyed by phone or mapped VIN; analytics uses location total mileage for daily mileage metrics. |
| `vehicle.dlq.jt808.v1` | Kafka producer/consumer error paths | Operators/replay tooling | Dead-letter records for failed JT808 production or consumer processing. |
| `vehicle.raw.mqtt-yutong.v1` | `yutong-mqtt-app` | `vehicle-history-app` | Yutong MQTT raw payload records for TDengine `raw_frames` and troubleshooting APIs. |
| `vehicle.event.mqtt-yutong.v1` | `yutong-mqtt-app` | `vehicle-history-app` | Yutong MQTT parsed telemetry events keyed by mapped VIN or device identity. |
| `vehicle.dlq.mqtt-yutong.v1` | Kafka producer/consumer error paths | Operators/replay tooling | Dead-letter records for failed Yutong MQTT production or consumer processing. |
| `vehicle.dlq.history.v1` | History app consumer error paths | Operators/replay tooling | Dead-letter records for failed history storage processing. |
Default consumer groups:
- History: `vehicle-history`
- Analytics statistics: `vehicle-stat`
## ACK Semantics
GB32960 success ACKs are sent only after the ingest service completes the required Kafka production boundary for the accepted frame. If required Kafka production fails, the ingest service must not claim a successful ACK for that frame.
History and analytics failures do not block GB32960 ACKs. They are downstream Kafka consumer failures and should be handled through retry, DLQ inspection, offset replay, and service-specific recovery.
Authorization failures are rejected before the Kafka durability boundary and can be ACKed/rejected immediately according to protocol handling.
## Build Prerequisites
- Java 25 and Maven available on PATH.
- Kafka CLI tools such as `kafka-topics` and optionally `kafka-console-consumer` available on the ECS/operator host.
- Session state requires Redis. GB32960/JT808 ingest services keep live Netty
channels in-process, but session indexes and TTL metadata use Redis only.
- Redis is configured through Portainer environment variables such as `REDIS_HOST` and `REDIS_DATABASE=50`; there is no memory session-store fallback.
If your shell needs an explicit JDK:
```bash
export JAVA_HOME=/opt/homebrew/opt/openjdk@25/libexec/openjdk.jdk/Contents/Home
export PATH="$JAVA_HOME/bin:/opt/homebrew/bin:$PATH"
```
## Build
From the repository root:
```bash
mvn -pl :gb32960-ingest-app,:jt808-ingest-app,:yutong-mqtt-app,:vehicle-history-app,:vehicle-analytics-app -am package -Dmaven.test.skip=true
```
Expected result: `BUILD SUCCESS` and these runnable jars:
- `modules/apps/gb32960-ingest-app/target/gb32960-ingest-app.jar`
- `modules/apps/jt808-ingest-app/target/jt808-ingest-app.jar`
- `modules/apps/yutong-mqtt-app/target/yutong-mqtt-app.jar`
- `modules/apps/vehicle-history-app/target/vehicle-history-app.jar`
- `modules/apps/vehicle-analytics-app/target/vehicle-analytics-app.jar`
## Create Topics
Run these only on the ECS/operator host after Kafka is reachable. The default bootstrap address is the ECS private Kafka address used by the production stack:
```bash
KAFKA_BOOTSTRAP="${KAFKA_BOOTSTRAP:-172.17.111.56:9092}"
kafka-topics --bootstrap-server "$KAFKA_BOOTSTRAP" --create --if-not-exists --topic vehicle.raw.gb32960.v1 --partitions 12 --replication-factor 1
kafka-topics --bootstrap-server "$KAFKA_BOOTSTRAP" --create --if-not-exists --topic vehicle.event.gb32960.v1 --partitions 12 --replication-factor 1
kafka-topics --bootstrap-server "$KAFKA_BOOTSTRAP" --create --if-not-exists --topic vehicle.dlq.gb32960.v1 --partitions 3 --replication-factor 1
kafka-topics --bootstrap-server "$KAFKA_BOOTSTRAP" --create --if-not-exists --topic vehicle.raw.jt808.v1 --partitions 12 --replication-factor 1
kafka-topics --bootstrap-server "$KAFKA_BOOTSTRAP" --create --if-not-exists --topic vehicle.event.jt808.v1 --partitions 12 --replication-factor 1
kafka-topics --bootstrap-server "$KAFKA_BOOTSTRAP" --create --if-not-exists --topic vehicle.dlq.jt808.v1 --partitions 3 --replication-factor 1
kafka-topics --bootstrap-server "$KAFKA_BOOTSTRAP" --create --if-not-exists --topic vehicle.raw.mqtt-yutong.v1 --partitions 12 --replication-factor 1
kafka-topics --bootstrap-server "$KAFKA_BOOTSTRAP" --create --if-not-exists --topic vehicle.event.mqtt-yutong.v1 --partitions 12 --replication-factor 1
kafka-topics --bootstrap-server "$KAFKA_BOOTSTRAP" --create --if-not-exists --topic vehicle.dlq.mqtt-yutong.v1 --partitions 3 --replication-factor 1
kafka-topics --bootstrap-server "$KAFKA_BOOTSTRAP" --create --if-not-exists --topic vehicle.dlq.history.v1 --partitions 3 --replication-factor 1
```
Optional sanity check:
```bash
kafka-topics --bootstrap-server "$KAFKA_BOOTSTRAP" --list | grep -E 'vehicle\.(raw|event|dlq)\.(gb32960|jt808|mqtt-yutong|history)\.v1'
```
## Production Runtime
Production services run on ECS through Portainer/Docker. Do not start GB32960, JT808, Yutong MQTT, history, or analytics services on local developer machines; local machines are for code build, unit tests, and source inspection only.
Use `deploy/portainer/docker-compose.yml` as the production runtime manifest. The default stack is:
- `gb32960-ingest-app`
- `jt808-ingest-app`
- `yutong-mqtt-app`
- `vehicle-history-app`
- `vehicle-analytics-app`
`vehicle-history-app` intentionally has no raw archive volume. The history hot path is Kafka to TDengine: raw envelopes are stored in TDengine `raw_frames` with `parsedJson`, `metadataJson`, and `rawUri`. Ingest services may still write original `.bin` files as cold backup, but history API queries must not depend on a shared local archive mount.
Keep `TDENGINE_HISTORY_ENABLED=true` in the `vehicle-history-app` environment so history reads and writes use TDengine in production.
`vehicle-analytics-app` is intentionally a metric runtime only. Latest-state APIs belong to `vehicle-state-service`, which stays outside the default reactor and should be built with `-Poptional-latest-state` and deployed as a separate consumer if that surface is needed.
## Health Verification
Run these from the ECS host or inside the matching container:
```bash
curl -sS http://127.0.0.1:20100/actuator/health
curl -sS http://127.0.0.1:20100/actuator/health/liveness
curl -sS http://127.0.0.1:20100/actuator/health/readiness
curl -sS http://127.0.0.1:20400/actuator/health
curl -sS http://127.0.0.1:20400/actuator/health/liveness
curl -sS http://127.0.0.1:20400/actuator/health/readiness
curl -sS http://127.0.0.1:20500/actuator/health
curl -sS http://127.0.0.1:20500/actuator/health/liveness
curl -sS http://127.0.0.1:20500/actuator/health/readiness
curl -sS http://127.0.0.1:20200/actuator/health
curl -sS http://127.0.0.1:20200/actuator/health/liveness
curl -sS http://127.0.0.1:20200/actuator/health/readiness
curl -sS http://127.0.0.1:20310/actuator/health
curl -sS http://127.0.0.1:20310/actuator/health/liveness
curl -sS http://127.0.0.1:20310/actuator/health/readiness
```
Expected: each aggregate, liveness, and readiness endpoint returns `{"status":"UP"}` or an equivalent Spring Boot health JSON with top-level status `UP`.
## End-to-End Verification
Use a known-valid GB32960 fixture, for example one of the hex samples under `modules/protocols/protocol-gb32960/src/test/resources/samples/gb32960/`.
Example send command:
```bash
xxd -r -p modules/protocols/protocol-gb32960/src/test/resources/samples/gb32960/realtime_001.hex | nc 127.0.0.1 32960
```
To capture and inspect the binary ACK, write the response to a file:
```bash
rm -f /tmp/gb32960-ack.bin
xxd -r -p modules/protocols/protocol-gb32960/src/test/resources/samples/gb32960/realtime_001.hex \
| nc -w 3 127.0.0.1 32960 > /tmp/gb32960-ack.bin
xxd -p /tmp/gb32960-ack.bin
```
Expected: the ACK file is non-empty and the decoded bytes begin with the GB32960 frame marker `2323`. Treat the ACK as verified only after checking the response bytes from your run.
Verify only what was actually run in your environment:
```bash
kafka-console-consumer --bootstrap-server "$KAFKA_BOOTSTRAP" --topic vehicle.raw.gb32960.v1 --from-beginning --max-messages 1 --timeout-ms 10000
kafka-console-consumer --bootstrap-server "$KAFKA_BOOTSTRAP" --topic vehicle.event.gb32960.v1 --from-beginning --max-messages 1 --timeout-ms 10000
kafka-console-consumer --bootstrap-server "$KAFKA_BOOTSTRAP" --topic vehicle.raw.mqtt-yutong.v1 --from-beginning --max-messages 1 --timeout-ms 10000
kafka-console-consumer --bootstrap-server "$KAFKA_BOOTSTRAP" --topic vehicle.event.mqtt-yutong.v1 --from-beginning --max-messages 1 --timeout-ms 10000
```
Use TDengine CLI or a JDBC client to verify history writes:
```sql
USE vehicle_ts;
SELECT COUNT(*) FROM raw_frames WHERE protocol = 'GB32960';
SELECT COUNT(*) FROM vehicle_locations WHERE protocol = 'GB32960';
SELECT COUNT(*) FROM raw_frames WHERE protocol = 'JT808';
SELECT COUNT(*) FROM jt808_locations WHERE protocol = 'JT808';
SELECT COUNT(*) FROM raw_frames WHERE protocol = 'MQTT_YUTONG';
SELECT COUNT(*) FROM vehicle_locations WHERE protocol = 'MQTT_YUTONG';
```
Useful HTTP query checks after events are consumed. For `realtime_001.hex`, the fixture VIN is `LTEST000000000001`; replace `<date-from>`, `<date-to>`, and `<stat-date>` with values that cover the consumed record's event time:
```bash
curl -sS 'http://127.0.0.1:20200/api/event-history/locations?protocol=GB32960&dateFrom=<date-from>&dateTo=<date-to>&vin=LTEST000000000001&limit=10'
curl -sS 'http://127.0.0.1:20200/api/event-history/raw-frames?protocol=GB32960&dateFrom=<date-from>&dateTo=<date-to>&vin=LTEST000000000001&limit=10'
curl -sS 'http://127.0.0.1:20200/api/event-history/gb32960/dictionary'
curl -sS 'http://127.0.0.1:20310/api/vehicle-stat/LTEST000000000001/daily?date=<stat-date>'
```
Expected E2E result when Kafka and all services are running:
- A GB32960 client receives a binary success ACK only after required Kafka production succeeds, and the captured ACK bytes are inspected.
- `vehicle.raw.gb32960.v1` receives a raw frame envelope.
- `vehicle.event.gb32960.v1` receives one or more normalized records.
- TDengine `raw_frames` receives GB32960/JT808 RAW rows with parsed JSON and metadata.
- TDengine `raw_frames` receives Yutong MQTT RAW rows with parsed JSON and metadata when MQTT payloads are consumed.
- TDengine location tables receive compact GB32960/JT808/Yutong MQTT location rows when applicable telemetry is present.
- JT808 analytics writes daily mileage rows to MySQL `vehicle_stat_metric` after consuming applicable `vehicle.event.jt808.v1` records.
Do not expect `vehicle-history-app` to create raw `.bin` archive files from Kafka raw records in the current implementation. `RawArchiveEventSink` can write archive files only when it receives `VehicleEvent.RawArchive.rawBytes()` inside the same JVM; the Kafka envelope carries only `RawArchiveRef` metadata.
Do not mark any of these as verified unless the matching command was run and the output was inspected.
## Runtime Location Policy
Production services run on ECS through Portainer/Docker.
- Do not create local service runners for GB32960, JT808, Yutong MQTT, history, or analytics.
- Do not run the service jars directly on developer machines for production verification.
- Keep local work limited to source edits, builds, and automated tests.
For Portainer deployments, keep history without an archive volume: history consumes raw/event Kafka topics and writes TDengine history rows.
## Rollback Guidance
If the split deployment is unhealthy:
1. Stop `gb32960-ingest-app` first so new GB32960 ACKs are not issued against an unhealthy Kafka boundary.
2. Keep Kafka topics intact for replay unless storage or privacy policy requires deletion.
3. Restart or roll back `vehicle-history-app` and `vehicle-analytics-app` independently; their failures do not require protocol ACK rollback because they consume from Kafka offsets.
4. If Kafka itself is unavailable, stop `gb32960-ingest-app` or route GB32960 traffic to a previously verified release that preserves the Kafka durability boundary.
5. After rollback, compare Kafka consumer group lag and DLQ contents before resuming the split services.
Rollback commands depend on the deployment environment. For the default production stack, roll back the Portainer stack image version and let Docker restart the affected containers.
## Task 8 Verification Notes
Observed on 2026-06-23 in worktree `.worktrees/gb32960-service-split`:
- Package build should use the current active-app command shown in the Build section.
- Repository-local Kafka setup inspection found no Kafka script, no Docker Compose file, and no compose YAML within the searched repository paths.
- `nc -z -w 2 127.0.0.1 9092` exited `1`, so no local Kafka broker was reachable at `127.0.0.1:9092`.
- `kafka-topics`, `kafka-topics.sh`, `docker`, and `docker-compose` were not found on PATH.
- Kafka topic creation, service startup, health checks, Kafka record checks, archive checks, TDengine checks, stat output checks, and ACK observation were not run because the local Kafka prerequisite was absent.
## Latest Build Verification
Observed on 2026-06-23 in worktree `.worktrees/gb32960-service-split`:
- Targeted module tests: `mvn -pl :sink-kafka,:ingest-core,:protocol-gb32960,:event-history-service,:vehicle-state-service,:vehicle-stat-service test` ended with `BUILD SUCCESS`; Maven reported 55 protocol/sink/core tests, 30 event-history tests, 14 vehicle-state tests, and 19 vehicle-stat tests with 0 failures/errors/skips.
- Split app tests should cover the current active apps listed in the Service Roles section.
- Package verification should use the active-app package command from the Build section.
- Split app startup: not run in this verification pass because an ECS/Portainer runtime was outside that build-only check.
- Kafka raw records: not verified in that build-only check.
- Kafka event records: not verified in that build-only check.
- TDengine history rows: not verified; downstream services were not started without Kafka.
- Analytics output: not verified; downstream services were not started without Kafka.
- ACK behavior: not verified against a live broker in this pass. Unit tests cover the GB32960 ACK boundary and ordering, but an operator should still run the ACK capture command above on ECS or staging.
## Remote Kafka Live Verification
Observed on 2026-06-23 with Kafka `114.55.58.251:9092`:
- `gb32960-ingest-app` handled TCP `32960` and HTTP `20100`.
- Ingest app health check `curl -sS http://127.0.0.1:20100/actuator/health` returned `{"status":"UP"}`.
- Kafka AdminClient confirmed topics `vehicle.raw.gb32960.v1` and `vehicle.event.gb32960.v1` existed, and created `vehicle.dlq.gb32960.v1`.
- The app accepted the live Hyundai platform connection from `8.134.95.166`, authenticated platform login, and received real GB32960 realtime/report/login/logout frames.
- Probe consumer with a temporary group read records from `vehicle.event.gb32960.v1`, including VIN keys such as `LB9A32A2XR0LS1575` and `LNXNEGRR3SR319485`.
- Probe consumer with a temporary group read records from `vehicle.raw.gb32960.v1`, including VIN keys such as `LNXNEGRR0SR319444`, `LNXNEGRR2SR318201`, and `LB9A32A23R0LS1532`.
- Runtime logs showed successful GB32960 ACK flushes for platform login, vehicle login, and vehicle logout after the Kafka-backed split ingest service was running.
This verifies the live path `GB32960 TCP 32960 -> gb32960-ingest-app -> Kafka raw/event topics` against the remote Kafka broker. Downstream `vehicle-history-app` and `vehicle-analytics-app` were not started in this pass.
## 2026-06-29 Superseded Local Verification Record
Observed on 2026-06-29 in worktree `.worktrees/vehicle-ingest-redesign-phase1`. This record is retained only as historical evidence of parser/storage behavior; do not repeat it as a deployment procedure because production services now run on ECS through Portainer/Docker.
- External JT808 traffic on TCP `808` was parsed and written through Kafka into TDengine. Direct TDengine checks showed more than `6000` JT808 `raw_frames` and more than `1900` JT808 `vehicle_locations`; current production history does not maintain a `telemetry_fields` table.
- A GB32960 synthetic realtime frame sent to TCP `32960` received a binary ACK beginning with `2323`. The same VIN `LTEST202606290001` was queryable from TDengine-backed history APIs with speed, mileage, longitude, and latitude fields.
- History raw and event Kafka consumption used separate bindings so raw-frame writes remained current while location/field event ingestion continued.
The verified behavior was `GB32960/JT808 TCP -> protocol app -> Kafka raw/event topics -> vehicle-history-app -> TDengine raw_frames/location tables -> Swagger/API query`; the deployment method from that historical run is superseded.

View File

@@ -1,888 +0,0 @@
# Go Vehicle Gateway 旁路验证记录
验证时间2026-07-01 22:00-22:10 CST
## 部署版本
- Git commit`60ca42e`
- 镜像:`crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com/oneos/vehicle-gateway-go:go-60ca42e-20260701220622`
- 部署方式ECS `115.29.187.205` 上 Docker Compose 旁路部署
- Compose 目录:`/opt/lingniu-go`
## 旁路端口和 topic
为了不影响现有 Java 生产服务,本次没有占用生产端口和生产 topic。
| 项 | 值 |
|---|---|
| GB32960 旁路 TCP | `115.29.187.205:23296 -> container :32960` |
| JT808 旁路 TCP | `115.29.187.205:18080 -> container :808` |
| Realtime API | `115.29.187.205:20210` |
| GB32960 RAW topic | `vehicle.raw.go.gb32960.v1` |
| JT808 RAW topic | `vehicle.raw.go.jt808.v1` |
| Yutong MQTT RAW topic | `vehicle.raw.go.yutong-mqtt.v1` |
| Unified topic | `vehicle.event.go.unified.v1` |
## 容器状态
ECS 上四个 Go 容器均已启动:
```text
go-vehicle-gateway Up
go-history-writer Up
go-stat-writer Up
go-realtime-api Up
```
## Kafka 验证
`18080` 注入 JT808 样例帧后,`vehicle.raw.go.jt808.v1`
`vehicle.event.go.unified.v1` 均消费到统一 JSON envelope。
样例解析结果包含:
- `protocol=JT808`
- `message_id=0x0200`
- `phone=013307795425`
- `longitude=121.069881`
- `latitude=30.590151`
- `speed_kmh=23`
- `total_mileage_km=10241.2`
- 表 27 附加项 `0x01` 原始值 `0001900C`
## TDengine 验证
目标库:`lingniu_vehicle_ts`
```text
raw_frames count = 3
vehicle_locations count = 2
vehicle_mileage_points count = 2
```
最新 RAW 标识:
```text
GB32960 | LNBSCB3D4R1234567 | LNBSCB3D4R1234567 |
JT808 | JT808:013307795425 | | 013307795425
JT808 | JT808:013307795425 | | 013307795425
```
说明JT808 样例 phone 当前没有在 `vehicle_identity_binding` 中解析到 VIN因此
TDengine tag 使用 `vehicle_key=JT808:013307795425`
## Redis 验证
`23296` 注入带 VIN 的 GB32960 合成帧后Realtime API 可查询:
```text
GET /api/realtime/vehicles/LNBSCB3D4R1234567
```
返回字段包含:
- `vin=LNBSCB3D4R1234567`
- `protocols=["GB32960"]`
- `online=true`
- `speed_kmh=30`
- `total_mileage_km=10000`
- `soc_percent=85`
- `longitude=121`
- `latitude=30.56`
## MySQL 统计验证
`23296` 注入带 VIN 的 GB32960 合成帧后,`vehicle_daily_metric` 写入:
```text
LNBSCB3D4R1234567 | 2020-07-01 | GB32960 | daily_mileage_km | 0.000 | 10000.000 | 10000.000 | 1
LNBSCB3D4R1234567 | 2020-07-01 | GB32960 | daily_total_mileage_km | 10000.000 | 10000.000 | 10000.000 | 1
```
说明:本次 GB32960 合成帧的设备时间为测试值,因此统计日期为 `2020-07-01`
## 已修复问题
- TDengine WebSocket 普通 `INSERT` 使用参数占位符时,字符串未按预期加引号,导致语法错误。已在 `60ca42e` 中改为显式 SQL literal 转义。
- MySQL Go DSN 中 `charset=utf8mb4%2Cutf8` 会被 driver 传给 MySQL导致语法错误ECS 旁路 env 已改为 `charset=utf8mb4`
- TDengine ECS 当前可用 root 密码与早先记录不一致,本次旁路 env 已按实际可认证值修正。
## 未完成
- 未切换真实生产端口 `32960/808`
- 未将真实外部 32960/808 流量转发到 Go 旁路端口。
- 未开启宇通 MQTT 正式订阅。
- JT808 样例未解析到 VIN需维护 `vehicle_identity_binding` 后再验证 808 每日统计。
## 2026-07-01 22:22 真实归档帧复验
部署版本已更新:
- Git commit`5f4f4fd`
- 镜像:`crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com/oneos/vehicle-gateway-go:go-5f4f4fd-20260701222128`
本轮补充验证内容:
- JT808 支持 2019 版 versioned header真实帧 `0x0200` 可解析出 `phone=14894135060``longitude=117.178187``latitude=30.570155``speed_kmh=33.4``total_mileage_km=7868.9`
- GB32960 真实燃料电池帧 `VIN=LB9A32A21R0LS1707` 可解析 `0x01` 整车、`0x02` 驱动电机、`0x03` 燃料电池、`0x04` 发动机、`0x05` 位置、`0x06` 极值、`0x07` 报警、`0x08` 电压、`0x09` 温度;后续 `0x30` 广东燃料电池扩展暂以 unknown raw 保留。
- GB32960 真实帧字段包含 `fuel_cell_hydrogen_consumption_kg_per_100km=1.8``longitude=120.800326``latitude=31.634907``total_mileage_km=53490.9`
- Realtime API `GET /api/realtime/vehicles/LB9A32A21R0LS1707` 已返回上述经纬度和氢耗字段。
- TDengine `vehicle_locations``vehicle_mileage_points` 已写入真实 GB32960 位置和里程点:
```text
2020-07-01 08:09:22 | GB32960 | LB9A32A21R0LS1707 | lon=120.800326 | lat=31.634907 | mileage=53490.9
```
后续仍需补齐:
- JT808 真实 phone `14894135060` 仍未解析到 VIN需维护身份绑定后再验证 VIN 维度实时和统计。
## 2026-07-01 22:28 广东燃料电池扩展复验
部署版本已更新:
- Git commit`a15de91`
- 镜像:`crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com/oneos/vehicle-gateway-go:go-a15de91-20260701222743`
本轮补充验证内容:
- GB32960 真实燃料电池帧 `VIN=LB9A32A21R0LS1707` 的 vendor 尾部已从 `unknown_unit` 补为结构化块:
`0x30 gd_fc_stack``0x31 gd_fc_auxiliary``0x32 gd_fc_dcdc``0x33 gd_fc_air_conditioner`
`0x34 gd_fc_vehicle_info``0x80 gd_fc_demo_extension``0x83 gd_fc_vendor_tlv`
- Kafka `vehicle.raw.go.gb32960.v1` 最新消息包含上述 block并输出关键扁平字段
`gd_fc_stack_hydrogen_inlet_pressure_kpa=97``gd_fc_stack_cell_count=108`
`gd_fc_vehicle_hydrogen_mass_kg=4.3``gd_fc_demo_stack_temp_c=65`
- Realtime API `GET /api/realtime/vehicles/LB9A32A21R0LS1707` 已返回 vendor 扁平字段:
```text
gd_fc_dcdc_controller_temp_c=47
gd_fc_demo_stack_temp_c=65
gd_fc_stack_cell_count=108
gd_fc_stack_frame_cell_count=108
gd_fc_stack_hydrogen_inlet_pressure_kpa=97
gd_fc_stack_water_outlet_temp_c=65
gd_fc_vehicle_hydrogen_mass_kg=4.3
```
- TDengine `raw_frames` 中 GB32960 行数增加到 `4`,最新 RAW 标识:
```text
2026-07-01 14:28:24.869 | 172.20.0.1:47082 | GB32960 | LB9A32A21R0LS1707 | OK
```
后续仍需补齐:
- JT808 真实 phone `14894135060` 到 VIN 的绑定数据维护与统计复验。
- Go 旁路尚未切换到生产 `32960/808` 端口;仍运行在 `23296/18080`
## 2026-07-01 22:43 JT808 前导零 phone 身份解析复验
部署版本已更新:
- Git commit`6b55694`
- 镜像:`crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com/oneos/vehicle-gateway-go:go-6b55694-20260701224300`
本轮修复内容:
- JT808 2011 老版包头的终端手机号按 BCD[6] 解析会保留前导 `0`,例如包头为 `013079963379`
- `vehicle_identity_binding.phone` 中实际维护的是去前导零后的手机号 `13079963379`
- 身份解析候选键已调整为先查原始 phone再查去前导零 phone之后再按 `device_id``plate` 降级。
真实帧复验样本:
- 归档帧:`/opt/lingniu/vehicle-ingest/gb32960/archive/2026/07/01/JT808/unknown/1782903942845000.bin`
- 消息:`0x0200` 位置信息汇报
- 包头 phone`013079963379`
- 绑定表 phone`13079963379`
- 解析 VIN`LKLG7C4E3NA774736`
Kafka `vehicle.raw.go.jt808.v1` 最新消息确认:
```text
vin=LKLG7C4E3NA774736
phone=013079963379
parsed.header.phone_trim_zero=13079963379
parsed.identity.resolved=true
parsed.identity.source=phone
parsed.identity.value=13079963379
```
Realtime API 确认:
```text
GET /api/realtime/vehicles/LKLG7C4E3NA774736
longitude=119.557997
latitude=29.049092
speed_kmh=0
total_mileage_km=0
device_time=2026-07-01T19:05:40+08:00
```
TDengine 确认:
```text
raw_frames:
2026-07-01 14:49:10.913 | JT808 | LKLG7C4E3NA774736 | phone=013079963379 | message_id=512
vehicle_locations:
2026-07-01 11:05:40.000 | JT808 | LKLG7C4E3NA774736 | lon=119.557997 | lat=29.049092 | speed=0 | mileage=0
vehicle_mileage_points:
2026-07-01 11:05:40.000 | JT808 | LKLG7C4E3NA774736 | mileage=0
```
MySQL `vehicle_daily_metric` 确认:
```text
LKLG7C4E3NA774736 | JT808 | 2026-07-01 | daily_mileage_km | 0.000 | sample_count=1 | TOTAL_MILEAGE_DIFF
LKLG7C4E3NA774736 | JT808 | 2026-07-01 | daily_total_mileage_km | 0.000 | sample_count=1 | TOTAL_MILEAGE_DIFF
```
说明:该真实位置帧携带的表 27 附加项 `0x01` 总里程值为 `0`,因此本次只验证身份解析、链路落地和差值统计写入;要验证非零日里程,需要回放或等待同一 VIN 的非零总里程连续位置点。
## 2026-07-01 22:55 Kafka 发布重试部署复验
部署版本已更新:
- Git commit`7e7066d`
- 镜像:`crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com/oneos/vehicle-gateway-go:go-7e7066d-20260701225503`
本轮修复内容:
- gateway 的 Kafka 发布增加同步重试封装。
- 默认配置:
- `KAFKA_PUBLISH_ATTEMPTS=3`
- `KAFKA_PUBLISH_BACKOFF_MS=100`
- `KAFKA_PUBLISH_TIMEOUT_MS=3000`
- RAW 仍先于 unified event 写入RAW 写失败不会继续写 unified。
- 单元测试覆盖首次失败后成功、重试耗尽返回错误、上下文取消时停止重试。
部署后容器状态:
```text
go-vehicle-gateway | go-7e7066d-20260701225503 | Up
go-history-writer | go-7e7066d-20260701225503 | Up
go-stat-writer | go-7e7066d-20260701225503 | Up
go-realtime-api | go-7e7066d-20260701225503 | Up
```
真实 808 帧回放复验:
```text
source_endpoint=172.20.0.1:51966
protocol=JT808
message_id=0x0200
phone=013079963379
vin=LKLG7C4E3NA774736
parsed.identity.value=13079963379
```
Kafka `vehicle.raw.go.jt808.v1` 确认新消息写入partition 6 offset 从 `1` 增加到 `2`。最新消息包含:
```text
received_at_ms=1782917751903
source_endpoint=172.20.0.1:51966
vin=LKLG7C4E3NA774736
fields.longitude=119.557997
fields.latitude=29.049092
fields.total_mileage_km=0
```
TDengine `raw_frames` 确认新 RAW 落地:
```text
2026-07-01 14:55:51.903 | JT808 | LKLG7C4E3NA774736 | phone=013079963379 | source=172.20.0.1:51966
```
Realtime API 确认 Redis 已刷新:
```text
GET /api/realtime/vehicles/LKLG7C4E3NA774736
received_at_ms=1782917751903
source_endpoint=172.20.0.1:51966
longitude=119.557997
latitude=29.049092
total_mileage_km=0
```
说明:当前重试层只能覆盖 Kafka 短暂抖动;如果 Kafka 长时间不可用,仍需后续增加本地磁盘 spool/WAL 和恢复补发。
## 2026-07-01 23:02 切换 Go 到生产 32960/808 端口
本轮操作:
- 停止 ECS 上原有 Java Docker 容器:
- `gb32960-ingest-app`
- `jt808-ingest-app`
- `yutong-mqtt-app`
- `vehicle-history-app`
- `vehicle-analytics-app`
- `vehicle-state-app`
- `telemetry-field-parser-app`
- 保留 Go sidecar 和 Portainer agent。
- 修改 `/opt/lingniu-go/.env`
- `GO_GB32960_TCP_PORT=32960`
- `GO_JT808_TCP_PORT=808`
- Go gateway 使用镜像:
- `crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com/oneos/vehicle-gateway-go:go-243de5c-20260701230028`
- 开启 Kafka 本地 spool
- host`/opt/lingniu-go/spool/gateway`
- container`/data/spool/gateway`
端口验证:
```text
0.0.0.0:32960 -> go-vehicle-gateway:32960
0.0.0.0:808 -> go-vehicle-gateway:808
18080/23296 -> 已释放
```
启动日志确认:
```text
kafka durable spool enabled dir=/data/spool/gateway replay_interval_ms=1000
tcp listener started protocol=GB32960 addr=[::]:32960
tcp listener started protocol=JT808 addr=[::]:808
```
真实连接确认:
```text
GB32960 external remote: 8.134.95.166:50084
JT808 external remote: 115.231.168.135:13801
```
生产 `808` 回放复验:
```text
source_endpoint=172.20.0.1:54680
vin=LKLG7C4E3NA774736
phone=013079963379
longitude=119.557997
latitude=29.049092
total_mileage_km=0
```
TDengine 最新 RAW 复验显示生产 808 已有真实外部数据持续落地:
```text
2026-07-01 15:03:02.960 | JT808 | LKLG7C4E1NA774802 | 013079963291 | 222.66.200.68:28394
2026-07-01 15:03:02.114 | JT808 | | 14894135583 | 122.152.221.156:33871
2026-07-01 15:03:01.352 | JT808 | | 013307795519 | 115.231.168.135:13801
2026-07-01 15:03:01.330 | JT808 | LKLG7C4E3NA774736 | 013079963379 | 222.66.200.68:28393
```
Kafka spool 目录确认:
```text
/opt/lingniu-go/spool/gateway file count = 0
```
说明:生产 808 中仍有部分 phone 未解析出 VIN后续需要继续维护 `vehicle_identity_binding` 的 phone/device_id/plate 到 VIN 映射,或者增强 808 注册/鉴权帧中的 identity 回写。
## 2026-07-01 23:09 808 里程统计防 0 污染复验
部署版本已更新:
- Git commit`d347525`
- 镜像:`crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com/oneos/vehicle-gateway-go:go-d347525-20260701230845`
本轮修复内容:
- `stat-writer` 不再把 `total_mileage_km <= 0` 的采样写入 MySQL 每日指标。
- MySQL upsert 对历史 `first_total_mileage_km=0` / `latest_total_mileage_km=0` 做纠偏;后续首次有效总里程会替换掉历史 0。
- `realtime-api` 合并实时快照时,非正 `total_mileage_km` 不再覆盖已有正值;速度、位置、状态等其他字段仍按事件时间更新。
测试验证:
```text
go test ./...
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build ./cmd/gateway
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build ./cmd/history-writer
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build ./cmd/stat-writer
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build ./cmd/realtime-api
```
生产回放验证:
- 第一条 808 帧:`device_time=2026-07-01T23:09:35+08:00``total_mileage_km=12345.6`
- 第二条 808 帧:`device_time=2026-07-01T23:09:36+08:00``total_mileage_km=0`
- 两条帧使用同一 VIN`LKLG7C4E3NA774736`
Redis merged 快照确认:第二条 0 里程帧没有覆盖已有有效总里程。
```text
event_time_ms=1782918576000
source_endpoint=172.20.0.1:40946
speed_kmh=0
longitude=119.557997
latitude=29.049092
total_mileage_km=12345.6
field_times_ms.total_mileage_km=1782918575000
```
MySQL `vehicle_daily_metric` 确认0 里程采样未进入统计sample_count 只因有效总里程采样增加一次。
```text
LKLG7C4E3NA774736 | JT808 | 2026-07-01 | daily_mileage_km | 0.000 | sample_count=13 | first=12345.600 | latest=12345.600
LKLG7C4E3NA774736 | JT808 | 2026-07-01 | daily_total_mileage_km | 12345.600 | sample_count=13 | first=12345.600 | latest=12345.600
```
说明:生产 TDengine 中已经存在大量 808 非零 `total_mileage_km` 采样,但许多 phone 尚未映射到 VIN因此 MySQL 按 VIN 的统计只会覆盖已解析 VIN 的车辆。下一步应继续补齐 `vehicle_identity_binding`,让更多 808 车辆进入统计。
## 2026-07-01 23:23 808 注册身份自动维护复验
部署版本已更新:
- Git commit`7b44f97`
- 镜像:`crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com/oneos/vehicle-gateway-go:go-7b44f97-20260701232200`
本轮修复内容:
- 808 `0x0100` 注册帧解析 `province``city``manufacturer``device_type``device_id``plate_color``plate`
- 808 `0x0102` 鉴权帧解析 `auth_token`
- Go gateway 在 identity resolve 后写入 `jt808_registration`,普通上行也会刷新 `latest_seen_at`
- 注册帧通过车牌解析到 VIN 后,会尝试把 phone/device/plate 反写到 `vehicle_identity_binding`
- 注册帧车牌支持 GBK 解码,避免 `Incorrect string value` 写入错误。
验证命令:
```text
go test ./...
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build ./cmd/gateway
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build ./cmd/history-writer
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build ./cmd/stat-writer
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build ./cmd/realtime-api
```
ECS 运行状态:
```text
go-vehicle-gateway | go-7b44f97-20260701232200 | 0.0.0.0:808->808, 0.0.0.0:32960->32960
go-history-writer | go-7b44f97-20260701232200 | Up
go-stat-writer | go-7b44f97-20260701232200 | Up
go-realtime-api | go-7b44f97-20260701232200 | 0.0.0.0:20210->20210
```
生产日志确认:
- 新版本启动后真实 808、32960 连接持续进入。
- GBK 车牌注册帧不再出现 `Incorrect string value`
MySQL `jt808_registration` 确认:
```text
registration_rows=214
distinct_phone=214
vin_rows=74
plate_rows=74
device_rows=72
013079963379 | device_id=9963379 | plate=沪A06788F | vin=LKLG7C4E3NA774736 | latest_registered_at=2026-07-01 23:23:01
013079963321 | device_id=9963321 | plate=沪A59613F | vin=LKLG7C4EXNA774796 | latest_registered_at=2026-07-01 23:23:04
```
说明:`jt808_registration.phone` 保存 808 包头原始 phone`vehicle_identity_binding.phone` 中已有部分去前导零 phone。Go resolver 会同时查询原始 phone 和去前导零 phone。
## 2026-07-01 23:28 生产链路与指标清理复验
本轮复验目标:
- 确认 Go gateway 的 Kafka durable spool 是重启窗口残留,而不是持续发布失败。
- 确认 stat-writer / history-writer / realtime-api 的 Kafka consumer group 无 lag。
- 确认真实 808 车辆 registration、TDengine RAW/里程点、Redis 实时查询均在持续更新。
- 清理之前生产合成回放产生的 JT808 当日指标测试数据。
Kafka topic / consumer group
```text
vehicle.raw.go.jt808.v1 有持续 offset
vehicle.raw.go.gb32960.v1 有持续 offset
vehicle.event.go.unified.v1 有持续 offset
go-stat-writer LAG=0
go-history-writer LAG=0
go-realtime-api LAG=0
```
TDengine
```text
raw_frames:
JT808 2922
GB32960 27
vehicle_mileage_points:
JT808 2050
GB32960 2
```
Redis 实时查询样例:
```text
GET /api/realtime/vehicles/LKLG7C4E3NA774736
vin=LKLG7C4E3NA774736
source_endpoint=222.66.200.68:29646
plate=沪A06788F
longitude=119.556866
latitude=29.047911
total_mileage_km=12345.6
```
说明:该 Redis merged 快照保留了此前有效 `total_mileage_km=12345.6`,后续真实 808 位置帧上报的 `total_mileage_km=0` 不会覆盖该字段。TDengine 中该 VIN 后续真实里程点的 `total_mileage_km` 均为 0因此 stat-writer 按“只统计正总里程”的规则不再写 JT808 每日指标。
MySQL 指标清理:
```text
删除前:
LKLG7C4E2NA774775 | JT808 | 2026-07-01 | daily_mileage_km | first=0 | latest=0 | sample_count=7
LKLG7C4E2NA774775 | JT808 | 2026-07-01 | daily_total_mileage_km | first=0 | latest=0 | sample_count=7
LKLG7C4E3NA774736 | JT808 | 2026-07-01 | daily_mileage_km | first=12345.6 | latest=12345.6 | sample_count=13
LKLG7C4E3NA774736 | JT808 | 2026-07-01 | daily_total_mileage_km | first=12345.6 | latest=12345.6 | sample_count=13
删除后等待真实流量窗口:
JT808 2026-07-01 metric rows = 0
```
清理原因:
- `LKLG7C4E3NA774736` 的 12345.6 来自此前生产连通性合成回放,不是真实平台上报。
- `LKLG7C4E2NA774775` 的 0 指标来自旧逻辑污染,当前 stat-writer 已跳过非正总里程。
Kafka spool
```text
23:24 spool_count=904
23:25 spool_count=862
23:28 spool_count=722
new_spool_1min=0
```
说明spool 正在回放下降,且最近 1 分钟没有新增文件;生产 gateway 日志未见 `Incorrect string value`、Kafka publish error 或 replay error。
## 2026-07-01 23:34 每日指标查询 API 复验
部署版本已更新:
- Git commit`56f4811`
- 镜像:`crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com/oneos/vehicle-gateway-go:go-56f4811-20260701233311`
新增接口:
```text
GET /api/stats/daily-metrics
```
查询参数:
```text
vin 可选
protocol 可选GB32960 / JT808 / YUTONG_MQTT
metricKey 可选daily_mileage_km / daily_total_mileage_km
dateFrom 可选YYYY-MM-DD
dateTo 可选YYYY-MM-DD
limit 可选,默认 50最大 1000
offset 可选,默认 0
```
生产可访问样例:
```text
http://115.29.187.205:20210/api/stats/daily-metrics?vin=LB9A32A21R0LS1707&protocol=GB32960&dateFrom=2020-07-01&dateTo=2020-07-01&limit=10
```
返回确认:
```json
{
"items": [
{
"vin": "LB9A32A21R0LS1707",
"stat_date": "2020-07-01",
"protocol": "GB32960",
"metric_key": "daily_mileage_km",
"metric_value": 0,
"metric_unit": "km",
"first_total_mileage_km": 53490.9,
"latest_total_mileage_km": 53490.9,
"sample_count": 3,
"calculation_method": "TOTAL_MILEAGE_DIFF",
"created_at": "2026-07-01 22:14:13",
"updated_at": "2026-07-01 22:28:25"
},
{
"vin": "LB9A32A21R0LS1707",
"stat_date": "2020-07-01",
"protocol": "GB32960",
"metric_key": "daily_total_mileage_km",
"metric_value": 53490.9,
"metric_unit": "km",
"first_total_mileage_km": 53490.9,
"latest_total_mileage_km": 53490.9,
"sample_count": 3,
"calculation_method": "TOTAL_MILEAGE_DIFF",
"created_at": "2026-07-01 22:14:13",
"updated_at": "2026-07-01 22:28:25"
}
],
"limit": 10,
"offset": 0,
"total": 2
}
```
JT808 真实生产查询:
```text
http://115.29.187.205:20210/api/stats/daily-metrics?protocol=JT808&dateFrom=2026-07-01&dateTo=2026-07-01&limit=10
```
返回 `total=0`。原因是此前合成回放产生的测试指标已清理,当前真实 808 已绑定 VIN 的车辆仍上报 `total_mileage_km=0`stat-writer 按规则不写非正总里程指标。
实时接口回归:
```text
http://115.29.187.205:20210/api/realtime/vehicles/LKLG7C4E3NA774736/online
```
返回 `online=true`,说明新增 stats API 没有影响 Redis 实时查询。
发布后 spool 观察:
```text
发布后立即spool_count=1101new_spool_1min=102
等待 60 秒spool_count=1033new_spool_1min=0
```
说明:本次发布重启窗口产生的 spool 正在回放下降,且最近 1 分钟没有新增gateway/realtime-api 最近日志未见 error、failed、Kafka publish error。
## 2026-07-01 23:50 生产端口接管与 RAW 查询 API 复验
部署版本:
- Git commit`5937132`
- 镜像:`crpi-85r4m0ackrm3qpje.cn-shanghai.personal.cr.aliyuncs.com/oneos/vehicle-gateway-go:go-5937132-20260701234220`
生产端口状态:
```text
go-vehicle-gateway 0.0.0.0:808->808/tcp
go-vehicle-gateway 0.0.0.0:32960->32960/tcp
```
旧 Java 容器状态:
```text
gb32960-ingest-app restart=no status=exited
jt808-ingest-app restart=no status=exited
yutong-mqtt-app restart=no status=exited
vehicle-history-app restart=no status=exited
vehicle-analytics-app restart=no status=exited
vehicle-state-app restart=no status=exited
telemetry-field-parser-app restart=no status=exited
```
生产连接确认:
- JT808 端口 `808` 已收到外部连接:`222.66.200.68``115.231.168.135`
- GB32960 端口 `32960` 已收到外部连接:`8.134.95.166``117.160.0.65`
- ECS 当前没有 `8089` 监听。
RAW 查询 API
```text
GET /api/history/raw-frames
```
生产可访问样例:
```text
http://115.29.187.205:20210/api/history/raw-frames?protocol=JT808&vin=LKLG7C4E3NA774736&messageId=0x0200&limit=1
http://115.29.187.205:20210/api/history/raw-frames?protocol=GB32960&vin=LB9A32A21R0LS1707&limit=1
```
接口复验结果:
```text
JT808_RAW:
total=1
protocol=JT808
vin=LKLG7C4E3NA774736
phone=013079963379
message_id=512
message_id_hex=0x0200
parse_status=OK
raw_hex_len=94
parsed_json_len=651
fields_json_len=186
GB32960_RAW:
total=1
protocol=GB32960
vin=LB9A32A21R0LS1707
message_id=2
message_id_hex=0x0002
parse_status=OK
raw_hex_len=1548
parsed_json_len=3698
fields_json_len=563
```
同轮回归验证:
```text
GET /api/realtime/vehicles/LKLG7C4E3NA774736/online
online=true
protocols=["JT808"]
GET /api/stats/daily-metrics?vin=LB9A32A21R0LS1707&protocol=GB32960&dateFrom=2020-07-01&dateTo=2026-07-01&limit=1
total=1
protocol=GB32960
metric_key=daily_mileage_km
calculation_method=TOTAL_MILEAGE_DIFF
```
说明:
- RAW 查询已返回 `raw_hex``parsed_json``fields_json`,可以用于查看完整解析结果。
- 日统计和实时查询在生产端口接管后仍可用。
## 2026-07-02 00:00 原生 systemd 部署与宇通 MQTT 复验
根据本次 goal 的新要求Go 接入链路后续不再使用 Docker 部署。本轮已中断 Docker 镜像构建路径,改为 ECS 原生 Linux 二进制 + systemd。
部署版本:
- Git commit`594ab2d`
- 部署目录:`/opt/lingniu-go-native`
- 当前 release`/opt/lingniu-go-native/releases/594ab2d`
- 当前指针:`/opt/lingniu-go-native/current -> /opt/lingniu-go-native/releases/594ab2d`
systemd 服务:
```text
lingniu-go-gateway.service active
lingniu-go-history-writer.service active
lingniu-go-stat-writer.service active
lingniu-go-realtime-api.service active
```
端口确认:
```text
[::]:32960 gateway
[::]:808 gateway
[::]:20210 realtime-api
```
Go Docker 容器已停止:
```text
go-vehicle-gateway Exited
go-history-writer Exited
go-stat-writer Exited
go-realtime-api Exited
```
宇通 MQTT
```text
yutong mqtt client started
mqtt subscribed broker=ssl://cpxlm.axxc.cn:38883 topic=/ytforward/shln/+ qos=1
```
说明MQTT broker 存在短连接 `EOF` 后自动重连现象,但订阅后已收到真实数据并落 RAW。
RAW 查询复验:
```text
GET /api/history/raw-frames?protocol=YUTONG_MQTT&limit=3
total=3
protocol=YUTONG_MQTT
vin=LMRKH9AC1R1004131
vehicle_key=LMRKH9AC1R1004131
parse_status=OK
parsed_json_len=1104
fields_json_len=116
```
同轮回归:
```text
GET /api/history/raw-frames?protocol=JT808&vin=LKLG7C4E3NA774736&messageId=0x0200&limit=1
total=1
parse_status=OK
GET /api/history/raw-frames?protocol=GB32960&vin=LB9A32A21R0LS1707&limit=1
total=1
parse_status=OK
GET /api/realtime/vehicles/LKLG7C4E3NA774736/online
online=true
protocols=["JT808"]
```
本轮落地文件:
- `deploy/systemd/README.md`
- `deploy/systemd/lingniu-go-gateway.service`
- `deploy/systemd/lingniu-go-history-writer.service`
- `deploy/systemd/lingniu-go-stat-writer.service`
- `deploy/systemd/lingniu-go-realtime-api.service`
## 2026-07-02 00:04 宇通 MQTT EOF 诊断
现象:
Go 原生部署后,`lingniu-go-gateway` 日志中出现多次:
```text
mqtt connection lost error=EOF
mqtt connection lost error=write: broken pipe
mqtt subscribed topic=/ytforward/shln/+ qos=1
```
10 分钟窗口统计:
```text
mqtt subscribed = 33
mqtt connection lost = 32
```
对照旧 Java `yutong-mqtt-app` 历史日志,停止前也存在同类行为:
```text
mqtt endpoint [yutong] received topic=/ytforward/shln/1 bytes=574
mqtt endpoint [yutong] connection lost
Caused by: java.io.EOFException
mqtt endpoint [yutong] re-subscribed topic=/ytforward/shln/+ qos=1
mqtt endpoint [yutong] received topic=/ytforward/shln/3 bytes=586
```
判断:
- EOF/短连接不是 Go TLS 支持或 systemd 原生部署新引入的问题。
- 旧 Java 和新 Go 都表现为订阅、收到数据、broker 断开、自动重连。
- 当前链路可持续收到宇通 MQTT 数据并写入 TDengine RAW。
复验:
```text
GET /api/history/raw-frames?protocol=YUTONG_MQTT&limit=1
total=1
protocol=YUTONG_MQTT
vin=LMRKH9ACXR1004094
parse_status=OK
parsed_json_len=1080
fields_json_len=115
```
运维建议:
- 单条 EOF 不作为不可用判断。
-`YUTONG_MQTT` RAW 入库增长、最近成功 `mqtt subscribed` 时间、服务 `systemctl is-active` 作为可用性判断。
- 如果 EOF 频率继续升高且 RAW 不再增长,再联系宇通侧确认 broker 长连接策略、同一 clientId 并发限制或网络出口策略。

View File

@@ -1,158 +0,0 @@
# Go Vehicle Gateway 验证手册
本文档用于验证 Go 重构运行面:
- `go-vehicle-gateway`GB32960 TCP、JT808 TCP、宇通 MQTT 接入。
- `go-history-writer`Kafka RAW 写 TDengine。
- `go-stat-writer`Kafka RAW 写 MySQL 每日指标。
- `go-realtime-api`Kafka unified event 写 Redis并提供实时查询 API。
## 本地构建验证
```bash
cd go/vehicle-gateway
go test ./...
go build ./cmd/gateway
go build ./cmd/history-writer
go build ./cmd/stat-writer
go build ./cmd/realtime-api
```
## 生产原生部署验证
```bash
python3 tools/go_prod_acceptance.py --date 2026-07-02
```
该命令会聚合执行:
- `go_systemd_prod_smoke.py`:应用 ECS systemd 服务 active/enabled、端口归属和 gateway spool
- `go_kafka_prod_smoke.py`Kafka topic 和消费组 lag
- `go_native_prod_smoke.py`HTTP API、RAW `parsed_json`、历史核心表 `frame_id` 回链、每日指标公式、Redis 实时新鲜度、TDengine/MySQL/Redis 数据链路
如需单独检查 systemd
```bash
python3 tools/go_systemd_prod_smoke.py --host 115.29.187.205 --user root
```
脚本通过 SSH 检查:
- `lingniu-go-gateway.service`
- `lingniu-go-history-writer.service`
- `lingniu-go-stat-writer.service`
- `lingniu-go-realtime-api.service`
- 上述服务均为 `active``enabled`
- `808``32960``gateway` 监听
- `20210``realtime-api` 监听
- `/opt/lingniu-go-native/spool/gateway` 无待回放文件
## 生产环境变量
当前 goal 的生产面使用 ECS 原生 systemd 部署,不再以 Docker/Portainer 作为生产运行方式。以下环境变量维护在 `/opt/lingniu-go-native/env/*.env`
```bash
KAFKA_BROKERS=172.17.111.56:9092
TDENGINE_DSN=root:<password>@ws(172.17.111.57:6041)/lingniu_vehicle_ts
MYSQL_DSN=lingniu_vehicle:<password>@tcp(rm-bp179zbv481rnw3e2.mysql.rds.aliyuncs.com:3306)/lingniu_vehicle_data?parseTime=true&charset=utf8mb4,utf8&loc=Asia%2FShanghai
IDENTITY_MYSQL_DSN=${MYSQL_DSN}
VEHICLE_IDENTITY_TABLE=vehicle_identity_binding
REDIS_ADDR=r-bp1u741kij7e51i481.redis.rds.aliyuncs.com:6379
REDIS_PASSWORD=<password>
REDIS_DB=50
```
宇通 MQTT 开启时再配置:
```bash
YUTONG_MQTT_ENABLED=true
YUTONG_MQTT_URI=<mqtt-uri>
YUTONG_MQTT_TOPICS=/ytforward/shln/+
YUTONG_MQTT_CLIENT_ID=lingniu-go-yutong-mqtt
YUTONG_MQTT_USERNAME=<username>
YUTONG_MQTT_PASSWORD=<password>
```
## ECS 进程检查
```bash
systemctl is-active lingniu-go-gateway lingniu-go-history-writer lingniu-go-stat-writer lingniu-go-realtime-api
ss -lntp | egrep ':(808|32960|20210)\b'
```
## Kafka 验证
优先使用 smoke 脚本做可重复检查:
```bash
python3 tools/go_kafka_prod_smoke.py --host 114.55.58.251 --user root --max-lag 100
```
脚本会检查 `vehicle.raw.go.gb32960.v1``vehicle.raw.go.jt808.v1``vehicle.raw.go.yutong-mqtt.v1``vehicle.event.go.unified.v1` 是否存在,并汇总 `go-history-writer``go-stat-writer``go-realtime-api` 的消费 lag。运行环境需要 SSH 免密或已建立可用的 SSH 认证方式。
手工核对命令:
```bash
kafka-console-consumer --bootstrap-server 172.17.111.56:9092 \
--topic vehicle.raw.go.jt808.v1 --max-messages 1 --timeout-ms 10000
kafka-console-consumer --bootstrap-server 172.17.111.56:9092 \
--topic vehicle.raw.go.gb32960.v1 --max-messages 1 --timeout-ms 10000
kafka-console-consumer --bootstrap-server 172.17.111.56:9092 \
--topic vehicle.raw.go.yutong-mqtt.v1 --max-messages 1 --timeout-ms 10000
kafka-console-consumer --bootstrap-server 172.17.111.56:9092 \
--topic vehicle.event.go.unified.v1 --max-messages 1 --timeout-ms 10000
kafka-consumer-groups --bootstrap-server 172.17.111.56:9092 \
--describe --group go-history-writer
kafka-consumer-groups --bootstrap-server 172.17.111.56:9092 \
--describe --group go-stat-writer
kafka-consumer-groups --bootstrap-server 172.17.111.56:9092 \
--describe --group go-realtime-api
```
## TDengine 验证
```sql
USE lingniu_vehicle_ts;
SHOW STABLES;
SELECT COUNT(*) FROM raw_frames;
SELECT COUNT(*) FROM vehicle_locations;
SELECT COUNT(*) FROM vehicle_mileage_points;
```
## MySQL 验证
```sql
SELECT protocol, metric_key, COUNT(*)
FROM vehicle_daily_metric
GROUP BY protocol, metric_key;
SELECT *
FROM vehicle_daily_metric
WHERE metric_key IN ('daily_mileage_km', 'daily_total_mileage_km')
ORDER BY updated_at DESC
LIMIT 20;
```
## Redis 实时查询
```bash
curl -sS 'http://115.29.187.205:20210/api/realtime/vehicles/<vin>'
curl -sS 'http://115.29.187.205:20210/api/realtime/vehicles/<vin>/online'
curl -sS 'http://115.29.187.205:20210/api/realtime/vehicles/<vin>/protocols/JT808'
```
## 验收证据
正式切换前至少记录:
- 一个真实 GB32960 VIN 的 RAW、location、mileage point、Redis snapshot。
- 一个真实 JT808 VIN/phone 的 RAW、location、mileage point、daily metric、Redis snapshot。
- 一个真实宇通 MQTT VIN 或 device_id 的 RAW、Redis snapshot。
- `go-history-writer``go-stat-writer` 重启后 Kafka offset 能继续推进。

View File

@@ -1,34 +0,0 @@
# JT808 Daily Mileage Streaming
## Scope
The analytics app can calculate daily mileage from JT808 telemetry only. It consumes `vehicle.event.jt808.v1` and saves the current daily result into the common `vehicle-stat` metric repository.
Only supported message backbone: Kafka.
## Storage
Runtime state: none outside `vehicle_stat_metric`. There is no separate JT808 daily-mileage table. The derived value is stored as one `daily_mileage_km` metric row in the common JDBC/MySQL table `vehicle_stat_metric`; the local-day minimum and maximum GPS total mileage values are kept on that same row as calculation source columns.
The JT808 daily-mileage value is calculated from the GPS total mileage reported in location additional information:
```text
daily_mileage_km = max_total_mileage_km - min_total_mileage_km
calculation_method = JT808_TOTAL_MILEAGE_DIFF
```
The first valid JT808 location point for a vehicle and local day stores both calculation endpoints on the metric row and writes `daily_mileage_km=0.0`. Later ordered or replayed points update the lower endpoint when the reported GPS total mileage is smaller, update the upper endpoint when it is larger, and rewrite the derived `daily_mileage_km` on the same row.
## Runtime Settings
Set these in Portainer or Nacos, without committing secrets:
```text
KAFKA_TOPIC_JT808_EVENT=vehicle.event.jt808.v1
VEHICLE_STAT_JT808_MILEAGE_ENABLED=true
MYSQL_JDBC_URL=<jdbc-url>
MYSQL_USERNAME=<user>
MYSQL_PASSWORD=<password>
```
Algorithm defaults use the Asia/Shanghai daily boundary. Restart recovery reads the same `daily_mileage_km` metric row.

View File

@@ -1,351 +0,0 @@
# Vehicle Ingest TDengine 链路验收手册
这份手册用于验收当前 GB32960/JT808/Yutong MQTT 接入重构链路:
1. 协议 App 接收 TCP 报文。
2. 协议 App 归档原始帧元数据,并把事件信封发布到 Kafka。
3. `vehicle-history-app` 消费 Kafka。
4. history 写入 raw、location 到 TDengine不写 telemetry_fields。
5. Swagger/API 通过 TDengine 做历史分页查询。
每一步都必须执行命令并检查输出后,才能标记为已验证。
## 必需运行参数
启动服务前先设置:
```bash
export KAFKA_BROKERS=114.55.58.251:9092
export TDENGINE_HISTORY_ENABLED=true
export TDENGINE_HISTORY_DATABASE=vehicle_ts
export TDENGINE_JDBC_URL='jdbc:TAOS-WS://<tdengine-host>:6041/vehicle_ts'
export TDENGINE_DRIVER_CLASS_NAME=com.taosdata.jdbc.ws.WebSocketDriver
export TDENGINE_USERNAME=root
export TDENGINE_PASSWORD='<tdengine-password>'
export TDENGINE_MIN_IDLE=0
export TDENGINE_MAX_POOL_SIZE=32
```
`TDENGINE_MIN_IDLE=0` 用于减少冷启动时 TDengine 连接重试日志。TDengine 地址稳定后,再按生产并发调大连接池。
高吞吐生产验收只验证 TDengine `raw_frames`、位置表和分页查询闭环。RAW 帧的 `parsedJson``metadataJson``rawUri` 直接进入 TDengineGB32960 snapshot/fields 接口基于 `raw_frames` 与当前解析器即时返回结构化结果;接入服务写出的原始 `.bin` 只作为冷备复核材料。逐字段趋势宽表由独立字段解析服务消费 Kafka RAW/事件后维护,不由 history-app 写入。
## 构建
```bash
mvn -pl :gb32960-ingest-app,:jt808-ingest-app,:yutong-mqtt-app,:vehicle-history-app,:vehicle-analytics-app -am package -Dmaven.test.skip=true
```
预期产物:
- `modules/apps/gb32960-ingest-app/target/gb32960-ingest-app.jar`
- `modules/apps/jt808-ingest-app/target/jt808-ingest-app.jar`
- `modules/apps/yutong-mqtt-app/target/yutong-mqtt-app.jar`
- `modules/apps/vehicle-history-app/target/vehicle-history-app.jar`
- `modules/apps/vehicle-analytics-app/target/vehicle-analytics-app.jar`
## ECS 服务状态验证
生产服务只在 ECS 上运行,通过 Portainer/Docker 管理。不要在本机用 `java -jar`、plist 或其它守护进程方式启动 GB32960、JT808、Yutong MQTT、history、analytics。
```bash
docker ps --format 'table {{.Names}}\t{{.Status}}\t{{.Ports}}' \
| egrep 'gb32960|jt808|yutong|vehicle-history|vehicle-analytics|NAMES'
ss -lntp | egrep ':(808|32960|20100|20200|20310|20400|20500)\b'
curl -sS http://127.0.0.1:20400/actuator/health
curl -sS http://127.0.0.1:20400/actuator/health/liveness
curl -sS http://127.0.0.1:20400/actuator/health/readiness
curl -sS http://127.0.0.1:20100/actuator/health
curl -sS http://127.0.0.1:20100/actuator/health/liveness
curl -sS http://127.0.0.1:20100/actuator/health/readiness
curl -sS http://127.0.0.1:20500/actuator/health
curl -sS http://127.0.0.1:20500/actuator/health/liveness
curl -sS http://127.0.0.1:20500/actuator/health/readiness
curl -sS http://127.0.0.1:20200/actuator/health
curl -sS http://127.0.0.1:20200/actuator/health/liveness
curl -sS http://127.0.0.1:20200/actuator/health/readiness
curl -sS http://127.0.0.1:20200/v3/api-docs \
| grep -E '/api/event-history/locations|/api/event-history/raw-frames'
curl -sS http://127.0.0.1:20310/actuator/health
curl -sS http://127.0.0.1:20310/actuator/health/liveness
curl -sS http://127.0.0.1:20310/actuator/health/readiness
```
预期:五个容器均为运行状态;健康检查为 `UP`JT808/GB32960 TCP 端口在 ECS 上监听OpenAPI 中包含通用位置分页查询和 RAW 帧查询接口。history-app 不暴露 `/api/event-history/telemetry/fields`也不持续写入逐字段宽表。JT808 位置帧中有 GPS 总里程时,`vehicle-analytics-app` 按差值法写入 MySQL `vehicle_stat_metric``daily_mileage_km` 指标。
## 运行位置约束
生产身份绑定固定使用 MySQL不再提供 file/memory/sqlite 运行时切换入口。生产服务固定运行在 ECS/Portainer不提供本机守护进程模板。
- `jt808-ingest-app` 监听 TCP `808`HTTP `20400`
- `gb32960-ingest-app` 监听 TCP `32960`HTTP `20100`
- `yutong-mqtt-app` 监听 HTTP `20500`
- `vehicle-history-app` 监听 HTTP `20200`
- `vehicle-analytics-app` 监听 HTTP `20310`
- history 热查询只依赖 Kafka 和 TDengine `raw_frames`history 不需要共享 archive volume。
- GB32960/JT808/Yutong MQTT 接入服务可以继续使用 `SINK_ARCHIVE_PATH` 保存原始冷备,但这不是 history API 的实时查询前置条件。
JT808 注册身份绑定:
- 生产需要把 0x0100 注册信息维护到 MySQL 时,只需要配置 `VEHICLE_IDENTITY_MYSQL_*` 连接参数。
- 需要提供 `VEHICLE_IDENTITY_MYSQL_JDBC_URL``VEHICLE_IDENTITY_MYSQL_USERNAME``VEHICLE_IDENTITY_MYSQL_PASSWORD`
- 服务启动时自动创建 `vehicle_identity_binding``vehicle_identity_binding_registration` 两张表。`vehicle_identity_binding` 只维护 `plate``vin` 两列;`vehicle_identity_binding_registration``protocol + phone` 保存 JT808 0x0100 注册字段。
- 注册帧无真实 VIN 时也会写入 registration 表,`vin` 默认为 `unknown`。外部系统识别车辆后只需要维护车牌和 VIN例如`INSERT INTO vehicle_identity_binding (plate, vin) VALUES ('沪A61559F', 'LNVIN000000000001') ON DUPLICATE KEY UPDATE vin = VALUES(vin);`
- MySQL 绑定在启动时加载到内存索引,`plate/vin` 反写后会按 `VEHICLE_IDENTITY_MYSQL_REFRESH_INTERVAL` 周期刷新,默认 `60s`;帧解析热路径只查内存,不按帧访问 MySQL。
- 后续同一终端上报 raw/event 时,服务会用 registration 表里的 `phone/device_id/plate` 加绑定表里的 `plate/vin` 解析成真实 VIN。
健康检查:
```bash
curl -sS http://127.0.0.1:20400/actuator/health
curl -sS http://127.0.0.1:20400/actuator/health/liveness
curl -sS http://127.0.0.1:20400/actuator/health/readiness
curl -sS http://127.0.0.1:20100/actuator/health
curl -sS http://127.0.0.1:20100/actuator/health/liveness
curl -sS http://127.0.0.1:20100/actuator/health/readiness
curl -sS http://127.0.0.1:20500/actuator/health
curl -sS http://127.0.0.1:20500/actuator/health/liveness
curl -sS http://127.0.0.1:20500/actuator/health/readiness
curl -sS http://127.0.0.1:20200/actuator/health
curl -sS http://127.0.0.1:20200/actuator/health/liveness
curl -sS http://127.0.0.1:20200/actuator/health/readiness
curl -sS http://127.0.0.1:20310/actuator/health
curl -sS http://127.0.0.1:20310/actuator/health/liveness
curl -sS http://127.0.0.1:20310/actuator/health/readiness
```
API / Swagger
- JT808 ingest health: `http://127.0.0.1:20400/actuator/health`
- GB32960 ingest health: `http://127.0.0.1:20100/actuator/health`
- Yutong MQTT ingest health: `http://127.0.0.1:20500/actuator/health`
- History query: `http://127.0.0.1:20200/swagger-ui/index.html`
- Analytics metrics: `http://127.0.0.1:20310/swagger-ui/index.html`
## Kafka Topic 验证
历史服务必须消费事件 topic 和 raw topic
- `vehicle.event.gb32960.v1`
- `vehicle.raw.gb32960.v1`
- `vehicle.event.jt808.v1`
- `vehicle.raw.jt808.v1`
- `vehicle.event.mqtt-yutong.v1`
- `vehicle.raw.mqtt-yutong.v1`
使用 Kafka CLI 或管理工具检查 topic 是否存在,并确认 history consumer group 在测试流量后没有持续堆积。默认 `KAFKA_GROUP_HISTORY=vehicle-history` 时,实际 group 会拆成:
- `vehicle-history-gb32960-event`
- `vehicle-history-gb32960-raw`
- `vehicle-history-jt808-event`
- `vehicle-history-jt808-raw`
- `vehicle-history-yutong-mqtt-event`
- `vehicle-history-yutong-mqtt-raw`
## JT808 实时转发验收
如果外部平台已经把 JT808 报文转发到 ECS TCP `808`,至少观察 60 秒日志和 Kafka 数据。
必须拿到以下证据:
- `jt808-ingest-app` 日志显示接入连接或解析到上游消息 ID。
- Kafka 的 `vehicle.event.jt808.v1``vehicle.raw.jt808.v1` 有新增记录。
- `vehicle-history-app` 日志没有持续出现 consumer 或 TDengine 写入失败。
消费完成后,使用已知终端手机号查询:
```bash
curl -sS 'http://127.0.0.1:20200/api/event-history/jt808/locations?phone=<phone>&dateFrom=2026-06-29T00:00:00%2B08:00&dateTo=2026-06-29T23:59:59%2B08:00&limit=10'
```
预期:响应包含 `items`,并且每条记录包含 `eventTime``phone``longitude``latitude``speedKmh``rawUri``metadataJson`
## GB32960 Snapshot / 字段投影验收
默认高吞吐模式下GB32960 全字段查询不依赖 telemetry_fields 宽表。history 先查 TDengine `raw_frames`,直接使用 RAW 行中的 `parsedJson`/`parsedFields` 生成 snapshot 和字段投影:
```bash
curl -sS 'http://127.0.0.1:20200/api/event-history/gb32960/snapshots?vin=<vin>&dateFrom=2026-06-29T00:00:00%2B08:00&dateTo=2026-06-29T23:59:59%2B08:00&limit=10'
curl -sS 'http://127.0.0.1:20200/api/event-history/gb32960/snapshots/fields?vin=<vin>&fields=VEHICLE.speedKmh,VEHICLE.totalMileageKm,POSITION_V2016.longitude,POSITION_V2016.latitude&dateFrom=2026-06-29T00:00:00%2B08:00&dateTo=2026-06-29T23:59:59%2B08:00&limit=10'
```
预期snapshot 返回 `sourceFrames.rawArchiveUri` 和解析后的 `blocks`;字段投影返回所选字段。新增协议字段后,先通过 replay 或专用解析任务补写 `raw_frames.parsedJson`,再基于历史 RAW 结构化结果查询。
## TDengine 直接检查
使用 TDengine CLI 或 JDBC 客户端检查超级表和子表是否创建:
```sql
USE vehicle_ts;
SHOW STABLES;
SELECT COUNT(*) FROM raw_frames;
SELECT COUNT(*) FROM vehicle_locations;
SELECT COUNT(*) FROM jt808_locations;
```
预期:
- `raw_frames``vehicle_locations``jt808_locations` 存在;有对应协议实时流量后,计数持续增长。
- 字段趋势宽表不属于 history-app 验收范围,由独立字段解析服务负责。
- 有实时流量后,计数持续增加。
## 失败语义
- TDengine 不可达时TDengine 查询接口应该返回 HTTP `503`,消息类似 `tdengine history query failed`
- TDengine 未启用时TDengine 专属接口不应该出现在 OpenAPI 中。
- 路径不存在时API 层应该返回 HTTP `404`,而不是泛化的存储失败。
## 2026-06-29 Superseded Local Live Verification Record
该记录仅保留为解析和存储行为的历史证据;生产运行位置已经收敛到 ECS/Portainer。
当时参与验证的服务:
- `jt808-ingest-app`TCP `808`HTTP `20400`
- `gb32960-ingest-app`TCP `32960`HTTP `20100`
- `vehicle-history-app`HTTP `20200`
当次健康检查均返回 `{"status":"UP"}`
可重复运行 live 验收工具:
```bash
python3 tools/vehicle_ingest_live_verify.py \
--tdengine-rest-url 'http://115.29.185.82:6041/rest/sql/vehicle_ts' \
--tdengine-username root \
--tdengine-password '<tdengine-password>' \
--history-base-url 'http://127.0.0.1:20200' \
--date-from '2026-06-29 00:00:00' \
--date-to '2026-06-29 23:59:59' \
--jt808-peer-like '222.66.200.68:%' \
--gb32960-peer-like '115.29.187.205:%'
```
该工具输出 `pass``warn``fail`:当前正式 GB32960 只有平台登录、没有车辆 `0x02` 时会输出 `warn`,避免把平台在线误判为车辆数据在线。正式车辆数据验收时增加 `--require-gb32960-vehicle-realtime`,缺少非测试 VIN 的 `0x02` 会直接返回非 0 退出码。
JT808 真实转发链路已验证:
- ECS TCP `808` 有外部平台 `222.66.200.68` 持续发送 JT808 报文。
- `vehicle.raw.jt808.v1``vehicle.event.jt808.v1` 被 history 消费。
- `raw_frames``protocol='JT808'` 的真实行数已超过 `36000`
- 外部平台真实 0x0100 注册帧已超过 `347` 条,真实 0x0200 位置帧已超过 `12400` 条。
- `jt808_locations` 中位置行数已超过 `8783`
- 当前高吞吐默认链路不维护 `telemetry_fields`JT808 明细验证以 `raw_frames``vehicle_locations` 和 API 分页结果为准。
- 分页 API 已用终端号 `13079963301` 验证第一页和 nextCursor 第二页,样例位置包含 `longitude=118.913846``latitude=31.927309``statusFlag=3``archive://2026/06/29/JT808/...` 原始帧引用。
- 注册帧样例:终端号 `13079963320`、deviceId `9963320`、deviceType `SEG-9888G`、plate `沪A61559F`、maker `70112`、province `31`、city `113`、plateColor `2`。生产 VIN 反写依赖 MySQL identity store后续 raw/event 的 VIN 解析也从 MySQL 绑定读取。
## JT808 ECS smoke / 轻压测工具
仓库提供 `tools/jt808_e2e_smoke.py`,用于重复验证 TCP `808` 到 TDengine history 的闭环:
```bash
export TDENGINE_REST_URL='http://<tdengine-host>:6041/rest/sql/vehicle_ts'
export TDENGINE_USERNAME='root'
export TDENGINE_PASSWORD='<tdengine-password>'
python3 tools/jt808_e2e_smoke.py \
--connection-mode session \
--start-phone 13079962000 \
--frames 3 \
--expect-history-count 3 \
--verify-pagination \
--archive-root "$PROJECT_ROOT/data/archive-jt808" \
--tdengine-rest-url "$TDENGINE_REST_URL" \
--tdengine-username "$TDENGINE_USERNAME" \
--tdengine-password "$TDENGINE_PASSWORD"
```
默认 `--connection-mode per-frame`,适合单帧连接 smoke。生产验收建议使用 `--connection-mode session` 覆盖同一连接内连续位置帧。脚本会输出发送数量、history 可见数量、分页验证结果、raw archive 检查结果和 `tdengineRawFrames``tdengineRawFrames` 表示本次可见位置记录中的唯一 `rawUri` 有多少个已确认写入 TDengine `raw_frames`
- 合成 0x0100 注册帧终端号 `13079969999` 已验证TCP `808` 返回 `0x8100` 注册 ACK`raw_frames.metadata_json` 包含 `jt808.register.province``jt808.register.city``jt808.register.maker``jt808.register.deviceType``jt808.register.deviceId``jt808.register.plateColor``jt808.register.plate`,对应 raw archive 文件存在。
复查 SQL
```sql
USE vehicle_ts;
SELECT COUNT(*) FROM raw_frames WHERE protocol = 'JT808';
SELECT COUNT(*) FROM jt808_locations WHERE protocol = 'JT808';
SELECT event_time, vehicle_key, phone, message_id, raw_uri
FROM raw_frames
WHERE protocol = 'JT808'
ORDER BY event_time DESC
LIMIT 5;
SELECT ts, frame_id, phone, message_id, metadata_json, raw_uri
FROM raw_frames
WHERE protocol = 'JT808' AND phone = '13079969999' AND message_id = 256
ORDER BY ts DESC
LIMIT 3;
```
复查 API
```bash
curl -sS 'http://127.0.0.1:20200/api/event-history/jt808/locations?phone=13079963301&dateFrom=2026-06-29T00:00:00%2B08:00&dateTo=2026-06-29T23:59:59%2B08:00&order=DESC&limit=3'
curl -sS 'http://127.0.0.1:20200/api/event-history/jt808/raw-frames?phone=13079963320&messageId=256&dateFrom=2026-06-29T00:00:00%2B08:00&dateTo=2026-06-29T23:59:59%2B08:00&order=DESC&limit=3'
```
`/api/event-history/jt808/raw-frames` 用于排查注册、鉴权、心跳、位置等原始帧索引。返回项会保留 `messageId``messageIdHex``rawUri``parseStatus``peer``vin``phone` 和完整 `metadataJson`;注册帧的 `metadataJson` 包含 `jt808.register.province``jt808.register.city``jt808.register.maker``jt808.register.deviceType``jt808.register.deviceId``jt808.register.plateColor``jt808.register.plate``identitySource``identityResolved` 等字段。
GB32960 链路已用正式平台登录和合成实时帧验证:
- 正式平台 `115.29.187.205` 已连接 TCP `32960`,服务收到 `0x05 PLATFORM_LOGIN` 并返回 ACK日志显示账号 `Hyundai`、策略 `ALLOW`
- 2026-06-29 正式平台登录 raw 已有 `3` 条,均落入 TDengine `raw_frames`,可通过 RAW 查询看到 `parsedJson` 中的 `PLATFORM_LOGIN` 解析结果。
- 当前正式平台尚未发送车辆实时上报 `0x02`;非测试 VIN 的 `message_id=2` 计数为 `0`,因此不能把平台登录误判为车辆数据在线。
- TCP `32960` 返回 GB32960 `2323` ACK。
- `raw_frames` 中测试 VIN `LTEST202606290001` 已有 `2` 条 raw 记录。
- 当前默认通过 GB32960 snapshot/fields 即时解析 `raw_frames` 中的 RAW JSON不维护字段宽表增长。
- latest raw 包含 `archive://2026/06/29/GB32960/LTEST202606290001/...`
- snapshots API 可查到 `speedKmh=62.4``totalMileageKm=223456.7``longitude=116.397128``latitude=39.916527`
复查 SQL
```sql
USE vehicle_ts;
SELECT COUNT(*) FROM raw_frames
WHERE protocol = 'GB32960' AND vin = 'LTEST202606290001';
SELECT event_time, vin, raw_uri, frame_id
FROM raw_frames
WHERE protocol = 'GB32960' AND vin = 'LTEST202606290001'
ORDER BY event_time DESC
LIMIT 5;
SELECT ts, vin, message_id, event_time, raw_uri, peer, metadata_json
FROM raw_frames
WHERE protocol = 'GB32960' AND peer LIKE '115.29.187.205:%'
ORDER BY ts DESC
LIMIT 5;
SELECT COUNT(*) FROM raw_frames
WHERE protocol = 'GB32960' AND message_id = 2 AND vin <> '' AND vin NOT LIKE 'LTEST%';
```
也可以直接运行仓库 smoke 工具验证 TCP `32960`、GB32960 snapshot/fields、TDengine `raw_frames`。如果需要复核冷备 `.bin`,再传入接入服务 archive 根目录:
```bash
export TDENGINE_REST_URL='http://<tdengine-host>:6041/rest/sql/vehicle_ts'
export TDENGINE_USERNAME='root'
export TDENGINE_PASSWORD='<tdengine-password>'
python3 tools/gb32960_e2e_smoke.py \
--archive-root "$PROJECT_ROOT/data/archive-jt808" \
--tdengine-rest-url "$TDENGINE_REST_URL" \
--tdengine-username "$TDENGINE_USERNAME" \
--tdengine-password "$TDENGINE_PASSWORD"
```
预期输出包含 `records >= 1``fieldCounts` 中关键字段均大于 0、`tdengineRawFrames` 等于本次可见唯一 `rawUri` 数量。传入 `--archive-root` 时,`archiveChecked >= 1` 说明冷备文件也能复核;这里的 `records` 字段表示 snapshot 数量,用于兼容早期脚本输出名。
复查 API
```bash
curl -sS 'http://127.0.0.1:20200/api/event-history/gb32960/snapshots?vin=LTEST202606290001&dateFrom=2026-06-29T00:00:00%2B08:00&dateTo=2026-06-29T23:59:59%2B08:00&limit=10'
curl -sS 'http://127.0.0.1:20200/api/event-history/gb32960/snapshots/fields?vin=LTEST202606290001&fields=VEHICLE.speedKmh,VEHICLE.totalMileageKm,POSITION_V2016.longitude,POSITION_V2016.latitude&dateFrom=2026-06-29T00:00:00%2B08:00&dateTo=2026-06-29T23:59:59%2B08:00&limit=10'
```
## 当前注意事项
- 当前生产链路使用 TDengine WebSocket JDBC例如 `jdbc:TAOS-WS://<tdengine-host>:6041/vehicle_ts`
- `KAFKA_CONSUMER_AUTO_OFFSET_RESET=latest` 适合生产接入新流量;如果要回放历史 Kafka 数据,需要切换 consumer group 或重置 offset。
- `vehicle-history-app` raw 和 event 使用不同 consumer bindingraw consumer 必须开启,否则 `raw_frames` 不会持续增长。
- JT808 设备没有 VIN 映射时,`vehicle_key` 使用 `jt808:<phone>`;启用 MySQL identity store 后registration 表中已反写 VIN 的 phone/deviceId/plate 会用于后续 raw/event 的 VIN 解析。
- GB32960 正式对端目前只验证到平台登录;等待真实车辆 `0x02` 到达后,再用非测试 VIN 复查 `vehicle_locations`、snapshot/fields API 和导出能力。

View File

@@ -0,0 +1,227 @@
# 100K 车辆接入容量基线
更新时间2026-07-13
## 目标
支撑 100,000 台车辆接入,入口服务在高连接数和高帧率下保持可观测、可背压、可恢复。
## 初始容量假设
| 项 | 目标 |
| --- | --- |
| 连接车辆数 | 100,000 |
| 平均帧间隔 | 10-30 秒 |
| 平均入口 FPS | 3,000-10,000 |
| 短时 burst FPS | 20,000 |
| Gateway p99 parse + enqueue | < 50ms |
| Redis 当前态延迟 | < 3s |
| TDengine 历史写入 lag | 可观测且 burst 后下降 |
| MySQL 当前态 | 异步,不能阻塞 Redis |
## 当前生产观测
当前 ECS `115.29.187.205` 在低负载下健康:
- JT808 连接约 242。
- GB32960 连接约 2。
- Kafka lag 为 0。
- NATS ack pending 为 0。
- Redis 写入约 0ms。
- MySQL 投影约 1-2ms。
- ECS load 约 `0.09 / 0.15 / 0.21`
- 可用内存约 6.6GB。
- 根分区使用约 37%。
## 当前已知缺口
- `net.core.somaxconn=128`,无法作为 100K 连接生产基线。
- Gateway 默认 `TCP_MAX_CONNECTIONS` 已调整为 `120000`,生产仍可通过环境变量覆盖。
- 已新增可重复的 TCP 连接压测工具,后续需要用它跑 10K/50K/100K 阶段测试并记录结果。
- 仍缺读超时按协议维度计数Gateway frame duration 已提供 histogram可用于 parse + enqueue + response 的 p95/p99 估算。
- TDengine history writer 和 NATS fast writer 已使用 batch 写入;后续需要用带帧率压测验证 batch size、flush latency 和存储端承载能力。
- History、realtime、stat、identity writer 均支持单进程多个同 group consumer默认分别为 `HISTORY_WORKERS=3``REALTIME_WORKERS=3``STATS_WORKERS=3``IDENTITY_WRITER_WORKERS=3`。四者按 Kafka 分区并行且保持单车分区顺序;后续压测需联合观察 TDengine/MySQL 写延迟、连接数和各 consumer lag再决定是否增加 worker。
- Kafka topic 当前 12 分区100K 目标下需要结合实际 FPS 再评估分区数。
## ECS OS 参数建议
100K 连接目标需要至少以下系统参数作为起点:
```bash
sysctl -w net.core.somaxconn=65535
sysctl -w net.ipv4.tcp_max_syn_backlog=65535
sysctl -w net.ipv4.ip_local_port_range="10000 65000"
sysctl -w net.ipv4.tcp_tw_reuse=1
```
systemd gateway 已配置 `LimitNOFILE=1048576`,需要持续保持。
## 验收指标
| 层 | 指标 | 目标 |
| --- | --- | --- |
| Gateway | `vehicle_gateway_active_connections` | 能稳定到阶段目标连接数 |
| Gateway | `vehicle_gateway_connection_closes_total{reason}` | `read_error/extract_error/max_connections` 不异常增长 |
| Gateway | `vehicle_gateway_connection_rejections_total` | 容量目标内不增长 |
| Gateway async sink | `vehicle_async_sink_queue_depth{sink}` | 队列深度不持续增长 |
| Gateway async sink | `vehicle_async_sink_enqueue_total{sink,kind,status}` | `timeout/closed` 不增长 |
| Gateway async sink | `vehicle_async_sink_publish_total{sink,kind,status}` | `error` 不增长 |
| Gateway async sink | `vehicle_async_sink_publish_duration_ms_histogram_bucket` | p99 不持续上升 |
| Gateway | `vehicle_gateway_publish_total{status="ok"}` | 持续增长 |
| Gateway | `vehicle_gateway_frame_duration_ms_histogram_bucket` | p99 小于容量目标 |
| History writer | `vehicle_history_batch_flush_total{status}` | `error` 不增长 |
| History writer | `vehicle_history_batch_rows_total{status}` | batch 行数持续增长 |
| History writer | `vehicle_history_batch_pending_messages` | 稳态接近 0burst 后下降 |
| History writer | `vehicle_history_batch_pending_rows` | 稳态接近 0burst 后下降 |
| History writer | `vehicle_history_batch_flush_duration_ms{status}` | flush 延迟不持续上升 |
| History writer | `vehicle_history_config{setting="workers"}` | 不低于 `3` |
| History writer | `vehicle_history_worker_active` | 每个配置 worker 均为 `1` |
| NATS bridge | `vehicle_bridge_nats_consumer_ack_pending` | 稳态为 0 |
| NATS bridge | `vehicle_bridge_nats_consumer_pending` | burst 后下降 |
| NATS bridge | `vehicle_bridge_batch_pending_messages` | 稳态接近 0burst 后下降 |
| NATS bridge | `vehicle_bridge_batch_duration_ms_histogram_bucket` | p99 不持续上升 |
| NATS fast writer | `vehicle_fast_writer_nats_consumer_ack_pending` | 稳态为 0 |
| NATS fast writer | `vehicle_fast_writer_nats_consumer_pending` | burst 后下降 |
| NATS fast writer | `vehicle_fast_writer_batch_pending_messages` | 稳态接近 0burst 后下降 |
| NATS fast writer | `vehicle_fast_writer_batch_pending_envelopes` | 稳态接近 0burst 后下降 |
| NATS fast writer | `vehicle_fast_writer_stage_duration_ms_histogram_bucket` | TDengine/Redis/ack 各阶段 p99 不持续上升 |
| Kafka consumers | `vehicle_*_kafka_lag` | 稳态为 0burst 后下降 |
| Realtime Redis | `vehicle_realtime_store_update_duration_ms_histogram_bucket{store="redis"}` | p99 毫秒级且不持续上升 |
| Realtime MySQL | `vehicle_realtime_store_update_duration_ms_histogram_bucket{store="mysql"}` | p99 不持续上升 |
| Realtime MySQL | async queue depth | 不持续增长 |
| Realtime MySQL | `vehicle_realtime_config{setting="workers"}` | 不低于 `3` |
| Realtime MySQL | `vehicle_realtime_worker_active` | 每个配置 worker 均为 `1` |
| Stat writer | `vehicle_stat_write_duration_ms_histogram_bucket` | p99 不持续上升 |
| Stat writer | `vehicle_stat_config{setting="workers"}` | 不低于 `3` |
| Stat writer | `vehicle_stat_worker_active` | 每个配置 worker 均为 `1` |
| Identity writer | `vehicle_identity_writer_worker_active` | 每个配置 worker 均为 `1` |
## Retention Guardrails
10W 车辆接入时,中间件只承担解耦和短期缓冲职责,不能成为长期历史库。
| Component | Guardrail |
| --- | --- |
| NATS JetStream `VEHICLE_INGEST` | `NATS_STREAM_MAX_BYTES=21474836480``NATS_STREAM_MAX_AGE_HOURS=24``NATS_STREAM_ENSURE_TIMEOUT_SECONDS=60` |
| Kafka `vehicle.raw.go.*` / `vehicle.fields.go.*` | `retention.ms=21600000``segment.ms=600000``segment.bytes=268435456` |
| Gateway NATS publisher | `NATS_ASYNC_RAW_WORKERS=128`,吸收平台按秒集中上报形成的瞬时突发;以 `vehicle_async_sink_queue_wait_recent_p99_ms` 校验,不按稳态 queue depth 猜测 |
| Fast writer | `FAST_WRITER_WORKERS=16``FAST_WRITER_BATCH_SIZE=100``FAST_WRITER_FETCH_WAIT_MS=5``FAST_WRITER_OPERATION_TIMEOUT_MS=1000`;继续使用手工 ACK避免以可靠性换延迟 |
| Root disk | 使用率低于 `80%`,超过 `85%` 进入容量告警 |
如果 NATS `consumer_pending` 下降但仍高,说明系统在追历史积压;如果 `ack_pending=0` 且 Kafka lag 为 `0`,不要重启服务,重点观察 NATS data size 和 pending 下降斜率。
## 分阶段压测
压测入口:
```bash
# 100 连接 smoke test
/opt/lingniu-go-native/current/load-sim \
-protocol jt808 \
-addr 127.0.0.1:808 \
-connections 100 \
-connect-rate 100 \
-send-interval 10s \
-duration 2m \
-template 0200 \
-send=false
```
`load-sim``-send=true` 时会按连接和帧序号生成可被现有解析器解析的唯一协议帧:
- JT808 会变更包头手机号、流水号和 `0x0200` 位置时间,并重新计算转义和校验码。
- GB32960 会变更 VIN 和实时数据时间,并重新计算 BCC。
- JT808 发送模式默认持续读取服务端通用应答;禁止关闭 `-drain-responses` 后用应答写阻塞产生的延迟评价服务端性能。
- JT808 使用 `-cleanup-registration` 时会在结束后只清理指定模拟手机号区间且来源为 loopback 的注册记录RAW 仍保留作为压测证据。每轮发送压测必须使用未在 identity-writer 10 分钟 touch 节流窗口内使用过的新号段,否则注册不会重复写入,`rows_deleted` 可以为 `0`
- 低 FPS 全链路压测必须使用该模式,避免重复静态帧导致 event id 冲突。
- 对生产端口执行 `-send=true` 会写入合成 raw 数据,只能在明确隔离标识和压测窗口后执行。
1. 100 连接 smoke test。
2. 10,000 连接保持测试。
3. 10,000 连接 + 1 FPS/连接短测。
4. 50,000 连接保持测试。
5. 100,000 连接保持测试。
6. 按真实协议分布进行混合帧率测试。
每一阶段必须记录:
- active connections
- gateway frame/publish counters
- NATS pending 和 ack pending
- Kafka lag
- Redis/MySQL/TDengine 写入耗时和 pending depth
- CPU/load/memory/file descriptors
- 错误日志
## 2026-07-03 阶段压测记录
压测方式:在 ECS 本机使用 `/opt/lingniu-go-native/current/load-sim``127.0.0.1:808` 发起 JT808 hold-only 连接,`-send=false`,不发送业务帧,不污染 raw 数据。
压测前基线,时间 `2026-07-03 18:03:10 CST`
| 项 | 值 |
| --- | --- |
| 服务状态 | 6 个 Go systemd 服务均 active |
| JT808 连接 | 约 240 |
| GB32960 连接 | 约 2 |
| load average | `0.43 / 0.22 / 0.23` |
| available memory | 约 `6661 MB` |
| NATS ack pending | `0` |
| NATS pending | 约 `39` |
| history/stat/realtime Kafka lag | `0` |
阶段结果:
| 阶段 | 时间 | 参数 | 峰值连接/FD | 结果 | 资源和队列 |
| --- | --- | --- | --- | --- | --- |
| 100 连接 | `18:03:21-18:04:21 CST` | `connections=100 connect-rate=100 duration=60s send=false` | JT808 active `340``ss``341` | opened `100`failed `0`frames `0`write_errors `0` | load 约 `0.33/0.21/0.22`available memory 约 `6659 MB` |
| 1,000 连接 | `18:04:38-18:05:38 CST` | `connections=1000 connect-rate=500 duration=60s send=false` | JT808 active `1240``ss``1241`gateway FD 约 `1256` | opened `1000`failed `0`frames `0`write_errors `0` | load 约 `0.80/0.33/0.26`available memory 约 `6645 MB` |
| 10,000 连接 | `18:05:59-18:07:59 CST` | `connections=10000 connect-rate=1000 duration=120s send=false` | JT808 active `10240``ss``10241`gateway FD 约 `10256` | opened `10000`failed `0`frames `0`write_errors `0` | 20 秒时 load 约 `0.45/0.33/0.27`available memory 约 `6409 MB`NATS ack pending `0` |
| 50,000 连接 | `18:11:22-18:14:22 CST` | `connections=50000 connect-rate=2000 duration=180s send=false` | JT808 active `50238``ss``50239`gateway FD 约 `50254` | opened `50000`failed `0`frames `0`write_errors `0` | 45 秒时 load 约 `0.07/0.23/0.24`available memory 约 `5590 MB`NATS ack pending 短时 `29`,后续归零 |
| 100,000 总连接 | `18:18:13-18:21:13 CST` | JT808 `50000` + GB32960 `50000`,各 `connect-rate=2000 duration=180s send=false` | JT808 active `50236`GB32960 active `50002`,总 active 约 `100238`gateway FD 约 `100252` | JT808 opened `50000` / failed `0`GB32960 opened `50000` / failed `0`frames `0`write_errors `0` | 75 秒时 load 约 `0.16/0.30/0.29`available memory 约 `4511 MB`NATS ack pending 短时 `10`,后续归零 |
压测后:
- `2026-07-03 18:08:47 CST` JT808 active 回落到 `240``ss``241`
- `2026-07-03 18:15:36 CST` 5W 压测后 JT808 active 回落到约 `238``ss``239`
- `2026-07-03 18:16:49 CST` NATS ack pending 回到 `0`history/stat/realtime Kafka lag 总和均为 `0`
- `2026-07-03 18:23:02 CST` 10W 总连接压测后 JT808 active 回落到 `236`GB32960 active 回落到 `2`NATS ack pending 为 `0`history/stat/realtime Kafka lag 总和均为 `0`
- `20211/20212/20213/20214/20215/20216/20200` readyz 均 OK。
- 压测窗口未出现 gateway `error|failed|panic|fatal|rejected` 日志。
- `vehicle_gateway_connection_rejections_total` 未输出,表示本轮没有连接拒绝计数。
结论:单 ECS 已通过本机 10W 总连接 hold-only 基线测试。该结果证明当前内核参数、gateway 连接上限和 systemd 文件句柄设置可以承载 10W 空闲长连接;下一阶段必须进入带真实帧率的低 FPS 压测验证解析、NATS、Kafka、Redis、TDengine、MySQL 全链路吞吐。
## 2026-07-13 1000 FPS 全链路压测记录
压测参数JT808 `0x0200`1000 个连接,每连接 1 FPS持续 60 秒,手机号区间 `139000000000-139000000999`;开启服务端应答读取和注册自动清理。每轮均打开 1000 个连接、失败 0发送 59,500 帧、写错误 0读取应答约 1.19 MB、读错误 0结束后清理 1000 条模拟注册。
未读取 JT808 应答的早期结果不作为延迟基线:客户端接收缓冲区会填满并反向阻塞 Gateway 写应答,测到的是压测器缺陷而非服务端真实链路。
| 版本 | 观测点 | Gateway 应答 p99 | Redis p99 | Bridge RAW p99 | TDengine history p99 | MySQL stat p99 |
| --- | --- | ---: | ---: | ---: | ---: | ---: |
| topic 串行写 Kafka | 压测中段 | 约 32ms | 约 212ms | 约 304ms | 约 466ms | 约 332ms |
| topic 串行写 Kafka | 压测结束 | 约 10ms | 约 162ms | 约 326ms | 约 427ms | 约 313ms |
| topic 并行写 Kafkaconcurrency=6 | 压测中段 | 约 24ms | 约 207ms | 约 311ms | 约 409ms | 约 247ms |
| topic 并行写 Kafkaconcurrency=6 | 压测结束 | 约 29ms | 约 184ms | 约 288ms | 约 368ms | 约 316ms |
并行版本在同一 NATS 批次内按 Kafka topic 分组并发写RAW 与派生 fields 全部成功后才 ACK 原 NATS 消息;任一 topic 失败仍保留对应源消息待重放。低负载 Bridge p99 从约 110ms 降到约 65ms1000 FPS 下 history p99 下降约 14%Bridge RAW p99 下降约 12%。压测中段 NATS pending 短时峰值 71Gateway async queue 短时 285结束时均归零Kafka lag 和各 writer retry 均为 0。主机为 4 核,瞬时 load average 约 9.68,但采样时 CPU 仍约 52% idle、无 D 状态进程、可用内存约 6.4 GiB。
增加队列等待指标后,使用五个全新 JT808 号段进行了 worker A/B每轮仍是 59,500 帧、发送/读取错误为 0、清理注册 1000 条。以下均为压测中段最近 512 样本窗口,秒级集中发送会比均匀流量更严格:
| Gateway RAW worker | Fast writer worker / fetch | Gateway queue wait p99 | Gateway 应答 p99 | Redis p99 | 结论 |
| ---: | --- | ---: | ---: | ---: | --- |
| 32 | `8 / 20ms` | 约 95ms | 约 92ms | 约 234ms | 入口突发排队明显 |
| 64 | `8 / 20ms` | 约 84ms | 约 79ms | 约 254ms | Gateway 有小幅收益,下游未改善 |
| 128 | `16 / 5ms` | 约 42ms | 约 17ms | 约 151ms | 当前最佳组合 |
| 128 | `32 / 5ms` | 约 50ms | 约 32ms | 约 221ms | pull worker 过多产生调度竞争,回退 |
最终保留 `NATS_ASYNC_RAW_WORKERS=128``FAST_WRITER_WORKERS=16``FAST_WRITER_FETCH_WAIT_MS=5`。Redis 批写自身 p99 低于 `5ms`,约 151ms 的剩余尾延迟主要位于 JetStream 持久化发布和 consumer deliveryNATS/Kafka pending、Kafka lag 与 writer retry 在每轮结束后均归零。
## 2026-07-14 WAL Outbox 验证
Gateway 已切换为分段 WAL + 异步 JetStream PubAck。WAL 使用 16MiB/5s 分段、1ms/256 条 group commit、`fsync=true`、10000 最大 inflight本地 Apple M4 并发基准约 1926 records/s`519144 ns/op`),没有以关闭 `fsync` 换取吞吐。
ECS 生产使用 JT808 `0x0200`、1000 个连接、每连接 1 FPS、持续 60 秒验证:连接成功 1000、失败 0发送 59500 帧、写错误 0、读错误 0WAL `submitted=acked=66776`(含同期真实三协议流量),结束后 WAL backlog/inflight、NATS pending、Kafka lag 全部为 0。压测结束 Gateway JT808 响应 recent p99 约 80ms。
随后在约 1000 FPS 下对 Gateway 执行真实 `SIGKILL`。systemd 约 5 秒后自动拉起;启动后立即从崩溃前 WAL 重放并确认约 563 条记录,最终 WAL backlog/inflight、NATS pending 和 history/stat/realtime/identity Kafka lag 均回到 0日志没有 CRC 损坏、publish error 或 panic。故障窗口内压测客户端的写/读错误是 TCP 连接被强制中断的预期结果,不代表已持久接受的服务端数据丢失。

160
docs/ops/feichi-bridge.md Normal file
View File

@@ -0,0 +1,160 @@
# 飞驰 HTTP → GB/T 32960 桥接服务
## 目标
将飞驰车辆数据平台中的 20 辆 `GB_T32960` 车辆转换为 GB/T 32960.3-2016 标准 TCP 报文,并发送到岭牛内网车辆网关的 32960 端口。
服务必须满足:
- 每 10 秒发现实时增量,不重复转发平台持续返回的最后一帧;
- 使用 `0x05` 完成平台登录,实时数据使用 `0x02`,历史补传使用 `0x03`
- 每帧收到内网网关成功 ACK 后才推进持久化游标;
- 自动完成 CSRF 初始化、验证码识别、RSA 密码登录和 `R_SESS` 会话维护;
- 源会话返回 401/403 时自动重新登录,并重放一次原 API 请求;
- TCP 断线后重新登录,并用相同原始帧重试一次;
- 重启后从本地游标继续,默认补齐最近 1 小时、每次查询 20 分钟;
- 保留源接口调用、目标 ACK、错误和活跃车辆等 Prometheus 指标;
- 不转发离线车辆反复返回的旧快照。
- 离线车辆最后快照按源时间仅补发一次 `0x03`,不伪装成实时 `0x02`
## 数据路径
```text
飞驰 HTTP API
├─ 车辆发现 vehicleRealStatuss
├─ 实时明细 vehicleRealStatussByVId
└─ 历史明细 hisdataQuerys
feichi-bridge
字段映射 → 去重 → 32960 编码 → ACK 后提交游标
内网 vehicle-gateway :32960
```
桥接服务当前编码 V2016 `##` 报文。源平台确认的主要字段如下:
| 32960 数据 | 飞驰字段 |
|---|---|
| 数据时间 | `2000`(兼容 `9999` |
| 车速、里程、挡位 | `2201``2202``2203` |
| 加速/制动踏板 | `2208``2209` |
| 运行模式、DCDC | `2213``2214` |
| 充电状态、车辆状态 | `2301``3201` |
| 总电压、总电流、绝缘电阻 | `2613``2614``2617` |
| SOC | `7615` |
| 原始经纬度 | `2502``2503` |
| 单体电压、温度 | `2003``2103` |
| 极值 | `2601``2612` |
| 燃料电池主要数据 | `2110``2121` |
| 驱动电机复合数据 | `2307``2308` |
定位必须使用 `2502/2503` 原始坐标;`.gd` 字段是页面地图转换坐标,不能写进国标数据。
## 自动登录与认证文件
服务按照网页端协议执行 `first-login → randCode → login`:读取 `x-api-csrf`/`CSRF`,验证码图片交给 ECS 本机、禁网运行的专用 OCR 容器,密码使用网页端同一 RSA 公钥做 PKCS#1 v1.5 加密,登录成功后保存响应中的 `R_SESS`。OCR 结果不会被直接信任登录接口是最终校验错误时会换一张验证码重试。密码、Cookie、验证码图片和响应正文均不写日志。
`/opt/lingniu-go-native/secrets/feichi-auth.json`
```json
{
"username": "replace-me",
"password": "replace-me"
}
```
文件权限必须为 `0600`。飞驰若提供机器账号、长期 Token 或正式接口鉴权,仍应优先切换到供应商承诺的方式。
`/opt/lingniu-go-native/secrets/feichi-target.json`
```json
{
"platformId": "FEICHIBRIDGE00001",
"username": "bridge",
"password": "replace-me"
}
```
`platformId` 必须正好 17 字节。目标网关启用 `GB32960_AUTH_MODE=enforce` 时,需在网关凭据文件中登记同一平台账号和密码。
## 环境配置
`/opt/lingniu-go-native/env/feichi-bridge.env` 最小配置:
```dotenv
FEICHI_BASE_URL=http://mob.fsfeichi.com.cn:8000
FEICHI_AUTH_SECRET_FILE=/opt/lingniu-go-native/secrets/feichi-auth.json
FEICHI_TARGET_SECRET_FILE=/opt/lingniu-go-native/secrets/feichi-target.json
FEICHI_TARGET_ADDR=127.0.0.1:32960
FEICHI_STATE_FILE=/var/lib/lingniu-feichi-bridge/state.json
HEALTH_ADDR=127.0.0.1:20219
```
可调参数及默认值:
| 参数 | 默认值 | 说明 |
|---|---:|---|
| `FEICHI_POLL_INTERVAL_SECONDS` | 10 | 实时轮询间隔 |
| `FEICHI_DISCOVERY_INTERVAL_SECONDS` | 300 | 车辆清单刷新间隔 |
| `FEICHI_FETCH_CONCURRENCY` | 4 | 明细接口并发数 |
| `FEICHI_SOURCE_STALE_SECONDS` | 120 | 超过该时间不作为实时帧转发 |
| `FEICHI_STALE_REISSUE_ENABLED` | true | 将离线旧快照去重后仅补发一次为 `0x03` |
| `FEICHI_HTTP_TIMEOUT_SECONDS` | 15 | HTTP 超时 |
| `FEICHI_OCR_IMAGE` | `lingniu/feichi-captcha-ocr:1.0.0` | 本机验证码 OCR 镜像 |
| `FEICHI_OCR_TIMEOUT_SECONDS` | 20 | 单次验证码识别超时 |
| `FEICHI_LOGIN_MAX_ATTEMPTS` | 20 | 单轮自动登录最多验证码次数 |
| `FEICHI_LOGIN_RETRY_SECONDS` | 1 | 验证码/登录失败重试间隔 |
| `FEICHI_TARGET_TIMEOUT_SECONDS` | 10 | TCP 写入和 ACK 超时 |
| `FEICHI_BACKFILL_ENABLED` | true | 是否启用历史补传 |
| `FEICHI_BACKFILL_LOOKBACK_SECONDS` | 3600 | 无游标时回看范围 |
| `FEICHI_BACKFILL_WINDOW_SECONDS` | 1200 | 单次历史查询窗口 |
| `FEICHI_BACKFILL_SAFETY_SECONDS` | 30 | 历史与实时之间的安全延迟 |
| `FEICHI_BACKFILL_INTERVAL_SECONDS` | 3600 | 补传巡检周期 |
飞驰地址当前是明文 HTTP。部署时应限制服务的出口目的地址并避免认证文件、Cookie、响应原文进入代理访问日志。
## 构建与启动
```bash
cd go/vehicle-gateway
go test ./internal/feichibridge ./cmd/feichi-bridge
CGO_ENABLED=0 go build -trimpath -o feichi-bridge ./cmd/feichi-bridge
docker build -t lingniu/feichi-captcha-ocr:1.0.0 ../../deploy/feichi-ocr
install -d -m 0750 /var/lib/lingniu-feichi-bridge
systemctl daemon-reload
systemctl enable --now lingniu-go-feichi-bridge
```
检查:
```bash
curl -fsS http://127.0.0.1:20219/healthz
curl -fsS http://127.0.0.1:20219/readyz
curl -fsS http://127.0.0.1:20219/metrics | grep vehicle_feichi_bridge
journalctl -u lingniu-go-feichi-bridge -f
jq '.vehicles | to_entries[] | select(.value.last_realtime_ack_at or .value.last_backfill_ack_at) |
{vin: .key, realtime_ack: .value.last_realtime_ack_at, backfill_ack: .value.last_backfill_ack_at}' \
/var/lib/lingniu-feichi-bridge/state.json
```
## 上线验收
先选 1 辆在线车灰度 30 分钟,再开放全部 20 辆。验收需要同时满足:
1. 网关收到一次 `0x05` 登录,后续实时帧为 `0x02`
2. 同一源时间和同一内容只落一条,重启后无持续重复;
3. 断开目标 TCP 后能重连、重新登录并继续收到 ACK
4. 使源会话过期后能自动重登;连续登录失败时 `/readyz` 变为 503游标不前移
5. 两辆离线车不会把历史最后一帧当实时数据周期发送;
6. 原始经纬度、总压、总流、SOC、单体电压和温度与平台页面抽样一致
7. 20 辆车连续运行 24 小时后,源/目标计数和平台记录数差异可解释。
## 已知边界
- 页面接口不是飞驰承诺的稳定开放 API接口结构升级可能要求同步调整客户端
- 验证码识别依赖本机 OCR 镜像和飞驰当前验证码样式;样式变化会触发登录失败告警,不会导致未 ACK 数据推进游标;
- 源接口只提供离散快照,无法恢复平台未保存或查询窗口之外的原始上行帧;
- 映射结果是标准 32960 语义报文,不是车辆原始二进制报文的逐字节复制。

View File

@@ -0,0 +1,179 @@
# Go Service Observability
## ECS
Host: `115.29.187.205`
The Go services expose local-only health and metrics endpoints. They are intended for ECS-local checks, Prometheus scraping through an agent, or SSH troubleshooting. They are not public business APIs.
## Endpoints
| Service | systemd unit | Address | Health | Readiness | Metrics |
| --- | --- | --- | --- | --- | --- |
| Gateway | `lingniu-go-gateway.service` | `127.0.0.1:20211` | `/healthz` | `/readyz` | `/metrics` |
| History writer | `lingniu-go-history-writer.service` | `127.0.0.1:20212` | `/healthz` | `/readyz` | `/metrics` |
| Stat writer | `lingniu-go-stat-writer.service` | `127.0.0.1:20213` | `/healthz` | `/readyz` | `/metrics` |
| NATS Kafka bridge | `lingniu-go-nats-kafka-bridge.service` | `127.0.0.1:20214` | `/healthz` | `/readyz` | `/metrics` |
| Identity writer | `lingniu-go-identity-writer.service` | `127.0.0.1:20217` | `/healthz` | `/readyz` | `/metrics` |
| Realtime API | `lingniu-go-realtime-api.service` | `0.0.0.0:20200` | `/healthz` | `/readyz` | `/metrics` |
## Quick Checks
生产事故处理顺序和阈值解释见 [车辆接入生产运行手册](vehicle-ingest-runbook.md)。
```bash
curl -fsS http://127.0.0.1:20211/readyz
curl -fsS http://127.0.0.1:20211/metrics
curl -fsS http://127.0.0.1:20212/readyz
curl -fsS http://127.0.0.1:20213/readyz
curl -fsS http://127.0.0.1:20214/readyz
curl -fsS http://127.0.0.1:20217/readyz
curl -fsS http://127.0.0.1:20200/readyz
```
本机容量健康摘要:
```bash
/opt/lingniu-go-native/current/capacity-check
```
它会抓取 Gateway、History writer、Stat writer、Identity writer、NATS bridge、Realtime writer/API、NATS fast writer 的本地 `/metrics`,输出 JSON。退出码 `0` 表示当前关键 backlog 和拒绝计数正常,退出码 `2` 表示存在超阈值 pending、Kafka lag、连接拒绝或 metrics 抓取失败,适合接入 cron/告警。
Identity writer 必须消费 `vehicle.raw.go.jt808.v1`,且 Gateway 指标 `vehicle_gateway_jt808_registration_gateway_writes_enabled` 必须为 `0`。两者同时写 `jt808_registration` 会被 `capacity-check` 判为重复 writer。
Gateway 的 NATS 接受边界由 WAL outbox 提供。`vehicle_durable_outbox_backlog_records` 表示已经 `fsync` 但尚未 PubAck 的记录,`vehicle_durable_outbox_inflight` 表示正在等待 PubAck 的记录;健康时两者最终回到 `0``vehicle_durable_outbox_publish_total``submit_error``ack_error``remove_error``close_timeout` 默认按 outbox 错误门禁处理。稳定 event ID/NATS MsgId 负责崩溃恢复期间的 JetStream 去重。
默认还会用 histogram 估算链路 p99Gateway frame `250ms`、Gateway identity `100ms`、async sink `100ms`、bridge batch `5000ms`、fast-writer stage `100ms`、history flush `1000ms`、realtime store `100ms`、stat write `100ms`history/stat/realtime/identity 的 Kafka lag 总和超过 `100` 时降级。Gateway async sink 队列深度默认超过 `10000` 降级,可用 `-async-sink-queue-depth-max` 调整Gateway 身份快照必须 ready 且最近成功刷新时间不能超过 `180s`,分别用 `-gateway-identity-snapshot-required``-gateway-identity-snapshot-max-age-seconds` 调整。Gateway 身份解析缓存默认容量阈值是 `300000`Stat writer 统计缓存默认容量阈值是 `1000000`Realtime API 的 MySQL 写入节流缓存和 VIN 车牌缓存默认容量阈值分别是 `1000000``200000`,可分别用 `-gateway-identity-cache-entries-max``-stat-cache-entries-max``-realtime-throttle-cache-entries-max``-realtime-plate-cache-entries-max` 调整,设为 `0` 表示关闭该项容量检查。history、realtime、stat、identity writer 的 worker 数默认都不得低于 `3`,分别由 `-history-workers-min``-realtime-workers-min``-stat-workers-min``-identity-workers-min` 检查。低样本数会跳过 p99 判断,阈值设为 `0` 可临时关闭对应延迟检查。
批处理内部 in-flight pending 默认允许 `100` 条窗口:`-fast-writer-batch-pending-max=100``-history-batch-pending-max=100``-history-rows-pending-max=100``-stat-batch-pending-max=100`。这类指标用于发现批写卡住,不再因为高频流量下几十条正在 flush/ack 的瞬时批次直接降级NATS consumer pending、ack pending 和 Kafka lag 仍按独立阈值判断真实积压。
topic 流转检查默认只抓完全断层history/realtime 必须消费三类 raw topicstat 必须消费三类 fields topicbridge 未配置 subject 的 `route_error` 默认超过 `0` 条就降级,可用 `-bridge-route-error-max=-1` 临时关闭fast-writer `invalid_json` 默认超过 `0` 条就降级,可用 `-fast-writer-invalid-json-max=-1` 临时关闭history-writer `invalid_json` 默认超过 `0` 条就降级,可用 `-history-invalid-json-max=-1` 临时关闭realtime projector `invalid_json` 默认超过 `0` 条就降级,可用 `-realtime-invalid-json-max=-1` 临时关闭stat-writer `invalid_json` 默认超过 `0` 条就降级,可用 `-stat-invalid-json-max=-1` 临时关闭;某协议 gateway raw publish 达到 `100` 条,扣除 `skipped_non_realtime` 后仍然没有 fields publish 时降级;实时 fields 缺失或 publish error 比例超过 `20%` 时也会降级;某协议 gateway raw/fields publish 达到 `1000` 条但 bridge 没有写对应 Kafka topic 时降级;某协议 gateway raw publish 达到 `1000` 条但 fast-writer 没有消费同 raw subject 并成功写 Redis/ack 时降级;某 raw topic bridge 写 Kafka 达到 `1000` 条但 history 未收到/写入或 realtime 未收到/更新时降级;某 fields topic bridge 写 Kafka 达到 `100` 条但 stat-writer 未配置或未收到该 topic 时降级。解析质量检查默认在单协议 gateway frame 样本达到 `100` 条后启用,`OK` 以外的 `PARTIAL/BAD_FRAME/other` 比例超过 `5%` 时降级,用于发现协议解析器或上游报文质量突然恶化。响应质量检查默认在单协议 gateway response 尝试样本达到 `100` 条后启用,`build_error/write_error` 比例超过 `5%` 时降级;`skipped` 表示该消息类型无需响应,不计入分母。身份解析质量检查默认在单协议 identity 样本达到 `100` 条后启用,`resolved` 以外的 `unresolved/error/timeout` 比例超过 `20%` 时降级,用于发现 VIN/车牌/phone 映射突然失效。history-writer 会对实时数据帧检查 raw envelope 是否携带 `parsed_fields`,默认单 topic/protocol 达到 `100` 帧后,缺失比例超过 `5%` 降级用于发现历史证据层丢失扁平化解析字段。Redis 快路径字段质量检查默认在单 raw subject 写入字段达到 `1000` 个后启用,`skipped_stale/seen` 超过 `20%` 时降级用于发现大面积旧帧补发或设备时间漂移。stat-writer 还会在单个 fields topic 成功 append 达到 `100` 条但未抽取到任何里程样本时降级,或 VIN 缺失、里程缺失、非正里程、source 缺失这类可行动跳过比例超过 `60%` 时降级,用于发现字段映射或身份/source 关联断裂;单 topic 找到 `100` 条里程样本但没有 source tracking或 source endpoint 缺失比例超过 `60%` 时也会降级,用于发现 `vehicle_data_source` 来源管理断层;单 topic 已写入 `100` 条里程样本但最终 `vehicle_daily_mileage` 投影写入仍为 `0` 时也会降级,用于发现 source 候选层到最终查询层的断层;`skipped_same_mileage` 是重复总里程去重,不计入可行动跳过比例。已出现过的成功/received `*_last_*_unix_seconds` 如果超过 `300s` 未刷新,也会降级;刚启动尚未出现过该指标时不会因缺失 last activity 降级。`capacity-check` JSON 会额外派生 `*_last_*_age_seconds`,便于直接读取距今秒数。历史位置派生写入错误使用 `vehicle_history_last_location_write_unix_seconds{status="error"}` 判断是否仍在发生:默认只在最近 `300s` 内出现错误时降级,进程生命周期累计 counter 继续保留用于审计;可用 `-history-location-error-recent-seconds` 调整窗口,设为 `0` 时恢复累计值门禁。特殊环境可用 `-required-consumer-topics=''` 临时关闭消费 topic 合同检查。
`capacity-check` 还会默认请求 Realtime API 的 `/api/stats/daily-metrics/diagnostics/reasons`,把当天每日里程诊断聚合放到 JSON 的 `daily_mileage_diagnostics`。这项默认只展示 `vehicle_total``actionable_issue_total` 和原因分布,不会因为存在业务缺口直接退出 `2`;需要把缺口纳入发布/巡检阻断时,显式传 `-daily-mileage-diagnostics-max-actionable=0` 或其他阈值。诊断 API 不可访问会被视为观测能力故障并降级;特殊环境可用 `-daily-mileage-diagnostics-url=''` 关闭。
ECS 上通过 systemd timer 每分钟执行一次:
```bash
systemctl status lingniu-go-capacity-check.timer
systemctl list-timers lingniu-go-capacity-check.timer
journalctl -u lingniu-go-capacity-check.service --since '10 minutes ago' --no-pager
systemctl start lingniu-go-capacity-check.service
```
`lingniu-go-capacity-check.service` 是 oneshot 服务。容量健康时退出码为 `0`;不健康时退出码为 `2`timer 会保留 failed 结果JSON findings 会写入 journal。
## Core Counters
| Metric | Meaning |
| --- | --- |
| `vehicle_gateway_active_connections` | Current TCP connections by protocol. Labels: `protocol`. |
| `vehicle_gateway_connection_closes_total` | TCP connection closes by protocol and reason. Labels: `protocol`, `reason`. Reasons include `eof`, `read_timeout`, `read_error`, `extract_error`, `context_cancelled`, `max_connections`. |
| `vehicle_gateway_connection_rejections_total` | TCP connection rejections by protocol and reason. Labels: `protocol`, `reason`. |
| `vehicle_gateway_frames_total` | Protocol frames received and parsed by the gateway. Labels: `protocol`, `status`. |
| `vehicle_gateway_last_frame_unix_seconds` | Last observed gateway frame time by protocol and parse status. Labels: `protocol`, `status`. |
| `vehicle_gateway_frame_duration_ms` | Last observed Gateway frame handling duration and histogram buckets for p95/p99. Labels: `protocol`, `status`. |
| `vehicle_gateway_identity_total` | Gateway identity resolve results. Labels: `protocol`, `status`; status includes `resolved`, `unresolved`, `error`, `timeout`. |
| `vehicle_gateway_identity_duration_ms` | Last observed identity resolve duration and histogram buckets for p95/p99. Labels: `protocol`, `status`. |
| `vehicle_gateway_identity_cache_entries` | Gateway identity resolver cache entry count. Labels: `cache`; values include runtime `lookup`/`registration`/`source_code`/`location_touch` and `snapshot_binding`/`snapshot_identifier`/`snapshot_registration`/`snapshot_source`. |
| `vehicle_gateway_identity_cache_max_entries` | Gateway identity resolver cache entry cap per cache. Controlled by `IDENTITY_LOOKUP_CACHE_MAX_ENTRIES`, default `300000`. |
| `vehicle_gateway_identity_snapshot_ready` | `1` after at least one complete atomic identity snapshot refresh; `0` means frames continue ingesting but unknown JT808 identities cannot be resolved. |
| `vehicle_gateway_identity_snapshot_entries` | Current identity snapshot entry counts. Labels: `kind` = `binding`, `identifier`, `registration`, `source`. |
| `vehicle_gateway_identity_snapshot_refresh_total` | Snapshot refresh attempts. Labels: `status` = `ok` or `error`; an error keeps the last known good snapshot. |
| `vehicle_gateway_identity_snapshot_last_success_unix_seconds` | Last successful complete snapshot publication time. |
| `vehicle_gateway_publish_total` | Gateway publish/delegation result. In NATS mode raw is published and fields uses `status="delegated"`; unified appears only when explicitly enabled. Labels: `protocol`, `kind`, `status`. |
| `vehicle_gateway_last_publish_unix_seconds` | Last gateway publish time by protocol/kind/status, useful for protocol freshness checks. Labels: `protocol`, `kind`, `status`. |
| `vehicle_gateway_fields_total` | Gateway fields eligibility result. Labels: `protocol`, `status`; NATS mode uses `delegated_to_bridge`, while compatibility direct mode may use `published`/`publish_error`. |
| `vehicle_gateway_response_total` | Gateway protocol response build/write result. Labels: `protocol`, `message_id`, `status`; status includes `ok`, `skipped`, `build_error`, `write_error`. |
| `vehicle_gateway_last_response_unix_seconds` | Last gateway protocol response build/write result time. Labels: `protocol`, `message_id`, `status`. |
| `vehicle_gateway_response_e2e_recent_p99_ms` | Bounded recent p99 from frame receipt through successful protocol response. Labels: `protocol`; capacity target defaults to `100ms`. |
| `vehicle_gateway_response_e2e_recent_samples` | Samples in the bounded response-latency window. Labels: `protocol`. |
| `vehicle_gateway_authentication_total` | Protocol authentication decisions. Labels: `protocol`, `mode`, `source`, `status`; `source=configured` means a configured credential matched, `source=device` means the JT808 phone's snapshot token matched, and `source=none` means no credential matched. `observe` records mismatches without rejecting, while `enforce` returns a protocol failure and closes the connection. |
| `vehicle_gateway_authentication_mode` | Active authentication mode by protocol. Labels: `protocol`, `mode`. |
| `vehicle_gateway_authentication_credentials` | Configured account count, never credential values. Labels: `protocol`. |
| `vehicle_async_sink_publish_duration_ms_histogram` | Async sink publish duration histogram covering worker publish calls to the delegate sink. Labels: `sink`, `kind`, `status`. |
| `vehicle_async_sink_queue_capacity` | Gateway async sink queue capacity. Labels: `sink`. NATS uses `NATS_ASYNC_QUEUE_SIZE`; Kafka fallback uses `KAFKA_ASYNC_QUEUE_SIZE`. |
| `vehicle_bridge_messages_total` | NATS messages fetched by the bridge. Labels: `subject`, `status`. |
| `vehicle_bridge_fields_projection_total` | Fields projection result from canonical raw. Labels: `protocol`, `status`; `published` is healthy, error/skip labels explain why no fields event was emitted. |
| `vehicle_bridge_fields_projection_count` | Latest flattened field count emitted by bridge per protocol/status. |
| `vehicle_bridge_last_message_unix_seconds` | Last NATS message fetched by subject/status. Labels: `subject`, `status`. |
| `vehicle_bridge_kafka_writes_total` | Bridge writes to Kafka. Labels: `topic`, `status`. |
| `vehicle_bridge_last_kafka_write_unix_seconds` | Last bridge Kafka write by topic/status. Labels: `topic`, `status`. |
| `vehicle_bridge_nats_acks_total` | NATS ack results after Kafka write or unrouted-subject isolation. Labels: `subject`, `status`; status includes `ok`, `error`, and `dropped_route_error`. |
| `vehicle_bridge_last_ack_unix_seconds` | Last NATS ack by subject/status. Labels: `subject`, `status`. |
| `vehicle_bridge_nats_consumer_pending` | JetStream messages pending for the bridge durable consumer. |
| `vehicle_bridge_nats_consumer_ack_pending` | JetStream messages delivered to bridge but not yet acked. |
| `vehicle_bridge_nats_consumer_waiting` | Pull requests waiting on the bridge durable consumer. |
| `vehicle_bridge_batch_pending_messages` | NATS messages fetched by the bridge but not yet written to Kafka and acked. |
| `vehicle_bridge_batch_pending_kafka_messages` | Kafka messages prepared for the current bridge batch. |
| `vehicle_bridge_batch_duration_ms_histogram` | Bridge batch duration histogram covering Kafka write and NATS ack. Labels: `status`. |
| `vehicle_history_batch_flush_duration_ms_histogram` | History writer TDengine batch flush duration histogram. Labels: `status`. |
| `vehicle_realtime_store_update_duration_ms_histogram` | Realtime MySQL/Redis projection duration histogram. Labels: `store`, `protocol`, `status`. |
| `vehicle_stat_write_duration_ms_histogram` | Stat writer MySQL write duration histogram. Labels: `topic`, `status`. |
| `vehicle_stat_samples_total` | Stat writer mileage sample results. Labels: `topic`, `protocol`, `status`; status includes `found`, `written`, `skipped_missing_fields`, `skipped_missing_vin`, `skipped_missing_mileage`, `skipped_non_positive_mileage`, `skipped_same_mileage`, `skipped_missing_source`, `event_time_future_adjusted`. `skipped_missing_fields` means the fields topic carried an envelope without flattened fields and should be treated as a stream-contract issue. The last status means device event time was more than 10 minutes ahead of received time and stats used received time instead. |
| `vehicle_fast_writer_nats_consumer_pending` | JetStream messages pending for the fast writer durable consumer. |
| `vehicle_fast_writer_nats_consumer_ack_pending` | JetStream messages delivered to fast writer but not yet acked. |
| `vehicle_fast_writer_nats_consumer_waiting` | Pull requests waiting on the fast writer durable consumer. |
| `vehicle_fast_writer_batch_pending_messages` | NATS fast writer messages fetched but not yet written to TDengine/Redis and acked. |
| `vehicle_fast_writer_batch_pending_envelopes` | Valid parsed envelopes in the current NATS fast writer batch. |
| `vehicle_fast_writer_messages_total` | Fast writer message results by NATS subject. Labels: `subject`, `status`; status includes `ok`, `error`, `invalid_json`, and `ack_error`. `invalid_json` is acked and isolated so it will not block the raw fast path, but it indicates the stream contract is polluted. |
| `vehicle_fast_writer_stage_duration_ms_histogram` | Fast writer stage duration histogram for TDengine, Redis, and NATS ack. The Redis stage is one batch pipeline when the repository supports batch updates. Labels: `subject`, `stage`, `status`. |
| `vehicle_fast_writer_redis_envelopes_total` | Redis realtime projection envelope result from the NATS fast path. Labels: `subject`, `status`; status includes `seen`, `updated`, `skipped_non_realtime`, `skipped_missing_vin`, `skipped_missing_vehicle_key`, `skipped_missing_fields`. Missing fields is an ingress contract failure: RAW is retained, but Redis online/KV projection is skipped and never re-flattened downstream. |
| `vehicle_fast_writer_redis_fields_total` | Redis realtime KV field write result from the NATS fast path. Labels: `subject`, `status`; status includes `seen`, `written`, `skipped_stale`. `skipped_stale` means an older event-time frame was allowed to add missing fields but was blocked from overwriting newer field values. |
| `vehicle_fast_writer_last_message_unix_seconds` | Last fast-writer message processing result by subject/status. Labels: `subject`, `status`. |
| `vehicle_fast_writer_last_stage_unix_seconds` | Last fast-writer stage completion by subject/stage/status. Labels: `subject`, `stage`, `status`. |
| `vehicle_history_writes_total` | TDengine history writes for valid envelopes only. Labels: `topic`, `status`; invalid JSON messages are committed after `vehicle_history_kafka_messages_total{status="invalid_json"}` and are not counted as writes. |
| `vehicle_history_parsed_fields_total` | Realtime raw envelopes that did or did not carry precomputed `parsed_fields` before TDengine history write. Labels: `topic`, `protocol`, `status`; status includes `present`, `missing`. |
| `vehicle_history_last_message_unix_seconds` | Last history Kafka message by topic/status. Labels: `topic`, `status`. |
| `vehicle_history_last_write_unix_seconds` | Last history TDengine write result by topic/status. Labels: `topic`, `status`. |
| `vehicle_history_last_commit_unix_seconds` | Last history Kafka commit by topic/status. Labels: `topic`, `status`. |
| `vehicle_history_batch_pending_messages` | Messages fetched by history writer but not yet flushed and committed. |
| `vehicle_history_batch_pending_rows` | Parsed raw envelopes waiting in the current TDengine batch. |
| `vehicle_history_retry_pending_messages` | Fetched messages retained in memory behind a failed TDengine write or Kafka commit. Healthy value is `0`; the writer does not fetch a new batch while this is non-zero. |
| `vehicle_history_batch_retries_total` | Failure-closed history batch retries. Labels: `reason`; `write_error` retries only the uncommitted suffix and `commit_error` retries commit without replaying TDengine writes. |
| `vehicle_history_batch_flush_duration_ms` | Last TDengine batch flush duration. Labels: `status`. |
| `vehicle_history_config{setting="workers"}` | Configured in-process history Kafka consumer count; default and capacity minimum are `3`. |
| `vehicle_history_worker_active` | Active history consumer gauge by `worker`; all configured workers should be `1`. |
| `vehicle_stat_writes_total` | MySQL metric writes. Labels: `topic`, `status`. |
| `vehicle_stat_kafka_messages_total` | Stat writer Kafka message results by fields topic. Labels: `topic`, `status`; `invalid_json` messages are committed and isolated before MySQL mileage statistics. |
| `vehicle_stat_last_message_unix_seconds` | Last stat Kafka message by topic/status. Labels: `topic`, `status`. |
| `vehicle_stat_last_write_unix_seconds` | Last stat MySQL append result by topic/status. Labels: `topic`, `status`. |
| `vehicle_stat_last_commit_unix_seconds` | Last stat Kafka commit by topic/status. Labels: `topic`, `status`. |
| `vehicle_stat_write_duration_ms_histogram` | MySQL stat writer append duration histogram. Labels: `topic`, `status`. |
| `vehicle_stat_config{setting="workers"}` | Configured in-process Kafka consumer count. Default and capacity-check minimum are `3`. |
| `vehicle_stat_worker_active` | Active stat consumer gauge by `worker`; all configured workers should remain `1` while the service is running. |
| `vehicle_stat_batch_pending_messages` | Messages fetched by stat writer but not yet appended to MySQL and committed. Controlled by `STATS_BATCH_SIZE` and `STATS_BATCH_WAIT_MS`; capacity check default threshold is `1000`. |
| `vehicle_stat_retry_pending_messages` | Fetched messages retained in memory behind a failed MySQL write or Kafka commit. Healthy value is `0`. |
| `vehicle_stat_batch_retries_total` | Failure-closed stat batch retries. Labels: `reason`; commit-only retries do not append the mileage sample twice in the same process. |
| `vehicle_stat_cache_entries` | Stat writer in-memory cache entries by cache type. Labels: `cache`; cache includes `last_total_mileage`, `source_seen`, `projection`, and `baseline`. Runtime cap is controlled by `STATS_CACHE_MAX_ENTRIES` and defaults to `1000000`. |
| `vehicle_stat_cache_max_entries` | Stat writer configured per-cache entry cap. `0` means unlimited and should not be used in production without an external memory guard. |
| `vehicle_stat_cache_evictions_total` | Cumulative stat writer cache evictions caused by capacity pressure. Labels: `cache`. |
| `vehicle_stat_samples_total` | Mileage samples extracted and written by stat writer. Use this with `vehicle_stat_writes_total`: write `ok` only means the append call succeeded, while `status="written"` proves a sample reached `vehicle_daily_mileage_source`. |
| `vehicle_stat_sources_total` | Stat writer source tracking results for `vehicle_data_source`. Labels: `topic`, `protocol`, `status`; status includes `attempted`, `written`, `skipped_throttled`, `skipped_missing_endpoint`. This intentionally avoids source IP labels to keep metrics low-cardinality. |
| `vehicle_stat_projections_total` | Final daily mileage projection results from `vehicle_daily_mileage_source` to `vehicle_daily_mileage`. Labels: `topic`, `protocol`, `status`; status includes `attempted`, `written`, `skipped_throttled`. |
| `vehicle_realtime_updates_total` | Redis/MySQL realtime projector updates. Labels: `topic`, `status`. |
| `vehicle_realtime_kafka_messages_total` | Realtime projector Kafka message results by raw topic. Labels: `topic`, `status`; `invalid_json` messages are committed and isolated before Redis/MySQL projection. |
| `vehicle_realtime_last_message_unix_seconds` | Last realtime Kafka message by topic/status. Labels: `topic`, `status`. |
| `vehicle_realtime_last_update_unix_seconds` | Last realtime projection result by topic/status. Labels: `topic`, `status`. |
| `vehicle_realtime_last_commit_unix_seconds` | Last realtime Kafka commit by topic/status. Labels: `topic`, `status`. |
| `vehicle_realtime_store_update_duration_ms_histogram` | Redis/MySQL store update duration histogram. Labels: `store`, `protocol`, `status`. |
| `vehicle_realtime_config{setting="workers"}` | Configured in-process realtime Kafka consumer count; default and capacity minimum are `3`. |
| `vehicle_realtime_worker_active` | Active realtime consumer gauge by `worker`; all configured workers should be `1`. |
| `vehicle_realtime_async_queue_total` | Realtime API async secondary projection queue events for MySQL snapshot/location. Labels: `store`, `protocol`, `status`; status includes `queued`, `dropped`, `closed`. |
| `vehicle_realtime_async_queue_depth` | Realtime API async MySQL projection queue depth. Labels: `store`, `protocol`. |
| `vehicle_realtime_async_queue_capacity` | Realtime API async MySQL projection queue capacity. Labels: `store`; controlled by `MYSQL_REALTIME_ASYNC_QUEUE_SIZE`. |
| `vehicle_realtime_retry_pending_messages` | Fetched messages retained in memory behind a failed realtime projection or Kafka commit. Healthy value is `0`. |
| `vehicle_realtime_batch_retries_total` | Failure-closed realtime batch retries. Labels: `reason`; only the uncommitted suffix is projected again after a write failure. |
| `vehicle_identity_writer_retry_pending_messages` | JT808 raw messages retained behind a failed registration MySQL transaction or Kafka commit. Healthy value is `0`; identity writer does not fetch newer offsets while non-zero. |
| `vehicle_identity_writer_config{setting="workers"}` | Configured identity Kafka consumer count; default and capacity minimum are `3`. |
| `vehicle_identity_writer_worker_active` | Active identity consumer gauge by `worker`; every configured worker should remain `1`. |
| `vehicle_identity_writer_batch_retries_total` | Failure-closed identity retries. Labels: `reason`; `write_error` retries the transaction and `commit_error` retries only Kafka commit. |
| `vehicle_realtime_plate_cache_entries` | Realtime API VIN-to-plate binding cache entry count. Default runtime cap is controlled by `PLATE_CACHE_MAX_ENTRIES` and defaults to `200000`. |
| `vehicle_realtime_plate_cache_max_entries` | Realtime API configured VIN-to-plate cache entry cap. `0` means unlimited and should not be used in production without an external memory guard. |
| `vehicle_realtime_plate_cache_evictions_total` | Cumulative VIN-to-plate cache evictions caused by TTL expiry cleanup or capacity pressure. |
| `vehicle_history_kafka_lag` | Estimated Kafka lag for history writer by topic and partition. |
| `vehicle_stat_kafka_lag` | Estimated Kafka lag for stat writer by topic and partition. |
| `vehicle_realtime_kafka_lag` | Estimated Kafka lag for realtime projector by topic and partition. |
## First Principle
These metrics are operational telemetry. They should stay out of MySQL and TDengine business tables unless a product query explicitly requires persisted aggregates. Persisted aggregate metrics should be derived from Kafka replay rather than by widening realtime or raw tables.

View File

@@ -0,0 +1,555 @@
# Go 车辆接入上下文记忆
更新时间2026-07-03
这份文档用于在 Codex 上下文重载后快速恢复项目状态。更完整的运行排查步骤见:
- `docs/ops/vehicle-ingest-runbook.md`
- `docs/architecture/production-data-plane-inventory.md`
## 当前目标
Go 版本车辆数据接入链路已经作为生产主链路运行在 ECS `115.29.187.205`
核心目标:
1. 高性能可靠接收 GB32960、JT808、宇通 MQTT。
2. 协议解析后先写 NATS再桥接 Kafka。
3. Kafka raw 供 TDengine 历史、Redis 实时、MySQL 当前态/统计消费。
4. 尽量避免 Java 老链路、Xinda 相关链路和重复存储。
5. 解析字段只投影一次,避免 TDengine/Redis/MySQL 各自重复 flatten。
6. 新长期目标:朝 10W 车辆生产稳定运行优化,容量基线见 `docs/ops/100k-capacity-baseline.md`
2026-07-03 之后,`nats-fast-writer` 的快速路径是 TDengine 批写 + Redis 批量 pipeline一个 NATS fetch batch 先批量写 TDengine再用 `FastUpdateBatch` 把实时 KV、在线 TTL、last_seen 和协议索引合并到一次 Redis pipeline最后逐条 ack NATS。
TDengine writer 的 raw/location 子表创建有进程内单飞保护:同一子表 key 并发首次写入时,只有一个 goroutine 执行 `CREATE TABLE IF NOT EXISTS` 和 tag 更新,其他 goroutine 等待结果后继续 INSERT避免 10W 车辆启动或回放时对 TDengine 形成重复 DDL 风暴。
容量健康摘要工具:`/opt/lingniu-go-native/current/capacity-check` 会抓本机各 Go 服务 `/metrics` 并输出 JSON。关键 backlog、Kafka lag、连接拒绝或 metrics 抓取失败时退出码为 `2`,可以接 cron/告警。
## 服务和端口
ECS`115.29.187.205`
| 服务 | systemd | 端口 | 作用 |
| --- | --- | --- | --- |
| Gateway | `lingniu-go-gateway.service` | `808``32960``20211` | JT808、GB32960、宇通 MQTT 接入和解析 |
| NATS fast writer | `lingniu-go-nats-fast-writer.service` | 健康端口随 env | NATS 快速写 TDengine/Redis |
| NATS Kafka bridge | `lingniu-go-nats-kafka-bridge.service` | `20214` | NATS 到 Kafka |
| History writer | `lingniu-go-history-writer.service` | `20212` | Kafka raw 写 TDengine |
| Stat writer | `lingniu-go-stat-writer.service` | `20213` | Kafka raw 写每日里程 |
| Realtime API | `lingniu-go-realtime-api.service` | `20200` | Redis/MySQL 投影和 HTTP API |
业务端口:
- JT808`0.0.0.0:808`
- GB32960`0.0.0.0:32960`
- Swagger/API`http://115.29.187.205:20200/swagger-ui/index.html`
## 部署路径
Go 服务使用 systemd 裸机部署,不使用 Docker。
```bash
/opt/lingniu-go-native/current
/opt/lingniu-go-native/releases/*
/opt/lingniu-go-native/env/*.env
```
最近 release
```text
/opt/lingniu-go-native/releases/production-track-telemetry-20260714082451
/opt/lingniu-go-native/releases/100k-hardening-20260703175730
```
服务重启:
```bash
systemctl restart lingniu-go-gateway.service
systemctl restart lingniu-go-nats-fast-writer.service
systemctl restart lingniu-go-nats-kafka-bridge.service
systemctl restart lingniu-go-history-writer.service
systemctl restart lingniu-go-stat-writer.service
systemctl restart lingniu-go-realtime-api.service
```
健康检查:
```bash
for port in 20211 20212 20213 20214 20200; do
curl -fsS "http://127.0.0.1:${port}/readyz"
echo
done
```
## 中间件
Kafka / NATS ECS
- 公网:`114.55.58.251`
- 内网:`172.17.111.56`
- NATS`172.17.111.56:4222`
TDengine
- 内网:`172.17.111.57`
- WS`172.17.111.57:6041`
- 数据库:`lingniu_vehicle_ts`
MySQL RDS
- 内网:`rm-bp179zbv481rnw3e2.mysql.rds.aliyuncs.com:3306`
- 数据库:`lingniu_vehicle_data`
Redis
- 内网:`r-bp1u741kij7e51i481.redis.rds.aliyuncs.com:6379`
- DB`50`
不要在最终回答中明文输出生产密码。
## Topic
当前 Go raw topic
```text
vehicle.raw.go.gb32960.v1
vehicle.raw.go.jt808.v1
vehicle.raw.go.yutong-mqtt.v1
```
fields topic 已存在:
```text
vehicle.fields.go.gb32960.v1
vehicle.fields.go.jt808.v1
vehicle.fields.go.yutong-mqtt.v1
```
当前核心链路仍以 raw topic 作为事实输入raw envelope 里已经包含一次性预计算的 `parsed_fields`
## 代码结构
主要目录:
```text
go/vehicle-gateway/cmd/gateway
go/vehicle-gateway/cmd/history-writer
go/vehicle-gateway/cmd/realtime-api
go/vehicle-gateway/cmd/stat-writer
go/vehicle-gateway/cmd/nats-fast-writer
go/vehicle-gateway/cmd/nats-kafka-bridge
go/vehicle-gateway/internal/protocol/gb32960
go/vehicle-gateway/internal/protocol/jt808
go/vehicle-gateway/internal/protocol/yutongmqtt
go/vehicle-gateway/internal/realtime
go/vehicle-gateway/internal/history
go/vehicle-gateway/internal/identity
```
每次改代码后在本地跑:
```bash
cd go/vehicle-gateway
go test ./...
```
## 解析和存储原则
### 一次解析字段
2026-07-03 已完成优化:
- `FrameEnvelope` 增加:
- `parsed_fields`
- `parsed_field_types`
- Gateway 在发布 raw 前调用 `realtime.EnsureParsedFields(&env)`
- TDengine writer、Redis realtime KV、MySQL snapshot 优先使用 `env.ParsedFields`
- 旧消息没有 `parsed_fields` 时,保留 fallback`Parsed` 展开一次。
相关文件:
```text
go/vehicle-gateway/internal/envelope/envelope.go
go/vehicle-gateway/internal/realtime/kv.go
go/vehicle-gateway/internal/gateway/tcp_server.go
go/vehicle-gateway/internal/gateway/mqtt_client.go
go/vehicle-gateway/internal/history/writer.go
go/vehicle-gateway/internal/realtime/repository.go
go/vehicle-gateway/internal/realtime/snapshot_writer.go
```
### GB32960 多帧
GB32960 有两类实际模式:
1. 单条 `0x02` 同时包含标准 data unit 和 vendor data unit。
2. 标准帧加 vendor 栈分片帧。
例子:
- `LB9A32A24R0LS1426`:单帧完整混合上报,单帧约 `774 bytes``gd_fc_stack.cell_count=108`
- `LNXNEGRR7SR318212`:多帧上报,`gd_fc_stack.cell_count=432`,按 `200 + 200 + 32` 分片。
TDengine raw_frames 必须保持单帧事实,不跨帧合并。
Redis/MySQL 实时视图可以按 `protocol + vin` 增量合并,但不能让后来的 stack-only 帧覆盖标准车辆字段。
### JT808
JT808 主入口端口是 `808`,不再使用 `8089`
解析重点:
- 包头手机号按 BCD 去前导 0。
- `0x0100` 注册帧写 `jt808_registration`
- `0x0102` 鉴权帧尝试按 phone 查 registration/binding 关联 VIN。
- `0x0200` 位置帧即使没有注册/鉴权,也会按 phone 降级查 `vehicle_identity_binding`,建立 session。
- 位置附加信息 `0x01` 是 GPS 总里程,按差值用于每日里程。
## MySQL 身份表
`vehicle_identity_binding` 是外部维护事实表,服务只读,不允许运行时写入。
核心字段:
```text
vin
plate
phone
oem
updated_at
```
用途:
- JT808 用 phone 或 plate 找 VIN。
- Realtime snapshot/location 用 VIN 反查 plate。
- 后续导入 G7s、广安车联等来源时更新 phone/oem。
`jt808_registration` 是 808 注册鉴权运行状态表,以 phone 为主键记录注册、鉴权、source_endpoint、关联 VIN 状态。
## 2026-07-03 数据维护记录
### G7s binding
源文件:
```text
/Users/lingniu/Library/Mobile Documents/com~apple~CloudDocs/rsync/2026_07_02_17_11_37转发配置列表-1(2).xlsx
```
处理规则:
1. 删除原有 G7/G7s 的 `phone/oem`
2. 使用文件中的 `手机号` 列更新。
3. 冲突的不更新,其他更新。
执行结果:
- 清空旧 G7/G7s`666` 行。
- 更新安全 G7s`561` 行。
- 跳过:`72` 行。
- 缺 plate 或 VIN`14`
- 与非 G7/G7s 绑定冲突:`58`
跳过清单:
```text
outputs/g7s_binding_skipped_apply.csv
```
### JT808 unknown VIN 导出
导出条件:
- `jt808_registration.vin` 为空、`unknown``unknow`
-`phone` 不存在于 `vehicle_identity_binding.phone`
结果:
- `30` 个 phone
导出文件:
```text
outputs/jt808_unknown_vin_phone_not_in_binding.csv
```
## 常用查询
历史 raw frame
```bash
curl -s 'http://115.29.187.205:20200/api/history/raw-frames?protocol=GB32960&vin=LB9A32A24R0LS1426&limit=1&includeFields=true'
```
只取指定 parsed fields
```bash
curl -s 'http://115.29.187.205:20200/api/history/raw-frames?protocol=GB32960&vin=LB9A32A24R0LS1426&limit=1&includeFields=true&fields=gb32960.vehicle.speed_kmh&fields=gb32960.gd_fc_stack.stack_water_outlet_temp_c'
```
808 源 IP
```bash
journalctl -u lingniu-go-gateway.service --since '10 minutes ago' --no-pager
ss -tn sport = :808
```
最近已观察到 808 源 IP
```text
115.159.85.149
222.66.200.68
122.152.221.156
115.231.168.135
```
## 部署步骤
本地构建:
```bash
cd go/vehicle-gateway
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build -trimpath -ldflags='-s -w' -o /tmp/lingniu-go-deploy/gateway ./cmd/gateway
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build -trimpath -ldflags='-s -w' -o /tmp/lingniu-go-deploy/history-writer ./cmd/history-writer
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build -trimpath -ldflags='-s -w' -o /tmp/lingniu-go-deploy/realtime-api ./cmd/realtime-api
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build -trimpath -ldflags='-s -w' -o /tmp/lingniu-go-deploy/stat-writer ./cmd/stat-writer
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build -trimpath -ldflags='-s -w' -o /tmp/lingniu-go-deploy/nats-fast-writer ./cmd/nats-fast-writer
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build -trimpath -ldflags='-s -w' -o /tmp/lingniu-go-deploy/nats-kafka-bridge ./cmd/nats-kafka-bridge
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build -trimpath -ldflags='-s -w' -o /tmp/lingniu-go-deploy/capacity-check ./cmd/capacity-check
```
上传部署:
```bash
release="name-$(date +%Y%m%d%H%M%S)"
ssh root@115.29.187.205 "mkdir -p /opt/lingniu-go-native/releases/$release"
scp /tmp/lingniu-go-deploy/* root@115.29.187.205:/opt/lingniu-go-native/releases/$release/
ssh root@115.29.187.205 "chmod 755 /opt/lingniu-go-native/releases/$release/* && ln -sfn /opt/lingniu-go-native/releases/$release /opt/lingniu-go-native/current"
```
部署后验证:
```bash
ssh root@115.29.187.205 '
systemctl is-active lingniu-go-gateway.service lingniu-go-nats-fast-writer.service lingniu-go-nats-kafka-bridge.service lingniu-go-history-writer.service lingniu-go-stat-writer.service lingniu-go-realtime-api.service
for port in 20211 20212 20213 20214 20200; do curl -fsS "http://127.0.0.1:${port}/readyz"; echo; done
journalctl -u lingniu-go-gateway.service --since "2 minutes ago" --no-pager | grep -Ei "error|failed|panic|fatal" || true
'
```
## 2026-07-03 10W 接入优化部署记录
Release
```text
/opt/lingniu-go-native/releases/100k-hardening-20260703175730
```
包含:
- Gateway 默认 `TCP_MAX_CONNECTIONS=120000`
- 新增 `vehicle_gateway_connection_rejections_total`
- 新增 `/opt/lingniu-go-native/current/load-sim` 运维压测工具。
- Redis-first realtime projector 增加 MySQL async queue dropped 回归测试。
ECS 已落地内核参数:
```text
net.core.somaxconn = 65535
net.ipv4.tcp_max_syn_backlog = 65535
net.ipv4.ip_local_port_range = 10000 65000
net.ipv4.tcp_tw_reuse = 1
```
注意:远端 `/etc/sysctl.conf` 原本有 `net.ipv4.tcp_max_syn_backlog = 1024`,已改为 `65535`,同时保留 `/etc/sysctl.d/zz-lingniu-vehicle-ingest.conf`
部署后验证:
- `current` 指向 `100k-hardening-20260703175730`
- `20211/20212/20213/20214/20200` readyz 均为 ok。
- `808``32960``20200``20211-20214` 监听 backlog 均显示 `65535`
- Gateway 重启后持续接收 GB32960、JT808、Yutong MQTT 帧。
- 重启完成后窗口未发现 gateway/realtime-api 持续 error/fatal 日志。
### 10W 阶段压测进展
2026-07-03 已完成 ECS 本机 JT808 hold-only 阶段压测,不发送业务帧:
| 阶段 | 结果 |
| --- | --- |
| 100 连接 / 60s | opened `100`failed `0`frames `0`write_errors `0` |
| 1,000 连接 / 60s | opened `1000`failed `0`frames `0`write_errors `0` |
| 10,000 连接 / 120s | opened `10000`failed `0`frames `0`write_errors `0` |
| 50,000 连接 / 180s | opened `50000`failed `0`frames `0`write_errors `0` |
| 100,000 总连接 / 180s | JT808 opened `50000`、GB32960 opened `50000`failed `0`frames `0`write_errors `0` |
1W 阶段峰值约 `10240` JT808 active connectionsgateway FD 约 `10256`NATS ack pending 为 `0`,压测后连接回落到生产基线约 `240`readyz 全部 OK。
5W 阶段峰值约 `50238` JT808 active connectionsgateway FD 约 `50254`available memory 约 `5590 MB`。NATS ack pending 中间短时到 `29``18:16:49 CST` 回到 `0`history/stat/realtime Kafka lag 总和均为 `0`。压测后 JT808 active 回落到生产基线约 `238-240`readyz 全部 OK。
10W 总连接阶段使用双端口 loopbackJT808 `50000` + GB32960 `50000`,均 `-send=false`。峰值约 `100238` active connectionsgateway FD 约 `100252`available memory 约 `4511 MB`。NATS ack pending 中间短时到 `10``18:23:02 CST` 回到 `0`history/stat/realtime Kafka lag 总和均为 `0`。压测后 JT808 回落到约 `236`GB32960 回落到约 `2`readyz 全部 OK。
### Gateway async publish 指标
2026-07-03 已部署 gateway async sink 指标到 ECS 当前 release
```text
vehicle_async_sink_enqueue_total{sink,kind,status}
vehicle_async_sink_publish_total{sink,kind,status}
vehicle_async_sink_publish_duration_ms{sink,kind,status}
vehicle_async_sink_publish_duration_ms_histogram_bucket{sink,kind,status,le}
vehicle_async_sink_publish_duration_ms_histogram_count{sink,kind,status}
vehicle_async_sink_publish_duration_ms_histogram_sum{sink,kind,status}
vehicle_async_sink_queue_depth{sink}
```
部署后验证:
- `vehicle_async_sink_queue_depth{sink="nats"} 0`
- `vehicle_async_sink_enqueue_total{kind="raw",sink="nats",status="queued"}` 持续增长
- `vehicle_async_sink_publish_total{kind="raw",sink="nats",status="ok"}` 持续增长
- NATS bridge ack pending 为 `0`
- Gateway 启动后窗口无 `error|failed|panic|fatal` 日志
这些指标是进入带帧率压测前的关键保护栏:如果 `queue_depth` 持续增长、`enqueue_total{status="timeout"}` 增长,或 `vehicle_async_sink_publish_duration_ms_histogram_bucket` 的 p99 持续上升,说明入口 publish 队列或下游 NATS/Kafka 写入已成为瓶颈。
### Gateway connection close reason
2026-07-03 已为 TCP gateway 增加连接关闭原因计数:
```text
vehicle_gateway_connection_closes_total{protocol,reason}
```
当前 `reason` 包括 `eof``read_timeout``read_error``extract_error``context_cancelled``max_connections`。10W 长连接和带帧率压测时,应重点关注 `read_error``extract_error``max_connections` 是否异常增长;`read_timeout` 需要结合协议上报周期和在线数判断。
### NATS fast writer stage histogram
2026-07-03 已为 NATS fast writer 增加阶段耗时 histogram
```text
vehicle_fast_writer_stage_duration_ms_histogram_bucket{subject,stage,status,le}
vehicle_fast_writer_stage_duration_ms_histogram_count{subject,stage,status}
vehicle_fast_writer_stage_duration_ms_histogram_sum{subject,stage,status}
```
`stage` 取值包括 `tdengine``redis``ack`,分别对应快速写 TDengine raw/location、Redis realtime 投影、NATS ack。带帧率压测时如果某个 stage 的 p99 持续上升,可以直接定位 fast path 是存储瓶颈还是 ack/网络瓶颈。
2026-07-03 后续优化:`nats-fast-writer` 已从“Fetch batch 后逐条写 TDengine/Redis”改为“Fetch batch 后先调用 `history.Writer.AppendAllBatch` 批量写 TDengine再调用 `realtime.Repository.FastUpdateBatch` 用一次 Redis pipeline 写实时 KV/在线状态/索引,最后逐条 ack NATS”。这样可以同时减少 TDengine 和 Redis round tripack 仍按单条执行,避免存储失败时误 ack。
2026-07-03 后续优化:`nats-fast-writer` 的 TDengine SQL 连接池已从固定 `1` 改为可配置:
```text
FAST_WRITER_TDENGINE_MAX_OPEN_CONNS
FAST_WRITER_TDENGINE_MAX_IDLE_CONNS
```
默认 `max_open=1``max_idle=1`,生产可以通过 env 小步放大。此前实测默认跟随 worker 放大到 8 会导致 `[0x200] db is not specified` / `[0x2616] Database not specified`,根因是 TDengine `USE <database>` 不会自动作用到 `database/sql` 新建连接。2026-07-03 已将 fast-writer 使用的 history writer 改为生成 `database.table` 形式的 database-qualified SQL并在 ECS 上将 pool 调到 `2/2` 验证通过:未再出现选库错误,写入 ok 计数持续增长。
同时新增 batch pending gauge
```text
vehicle_fast_writer_nats_consumer_pending{stream,consumer}
vehicle_fast_writer_nats_consumer_ack_pending{stream,consumer}
vehicle_fast_writer_nats_consumer_waiting{stream,consumer}
vehicle_fast_writer_batch_pending_messages
vehicle_fast_writer_batch_pending_envelopes
```
`vehicle_fast_writer_nats_consumer_*` 表示 JetStream durable consumer 层面的 backlog/ack-pending`vehicle_fast_writer_batch_pending_*` 表示 fast-writer 已拉取但尚未完成 TDengine/Redis/ack 的批内压力。带帧率压测时它们应在稳态接近 0burst 后能够回落;如果持续非 0需要结合 `vehicle_fast_writer_stage_duration_ms_histogram` 判断卡在 NATS 拉取、TDengine、Redis 还是 ack。
### NATS Kafka bridge batch 指标
2026-07-03 已为 NATS -> Kafka bridge 增加 batch pending 和 duration histogram
```text
vehicle_bridge_batch_pending_messages
vehicle_bridge_batch_pending_kafka_messages
vehicle_bridge_batch_duration_ms_histogram_bucket{status,le}
vehicle_bridge_batch_duration_ms_histogram_count{status}
vehicle_bridge_batch_duration_ms_histogram_sum{status}
```
该耗时覆盖一批 NATS 消息路由到 Kafka、Kafka 同步写入、NATS ack 的完整过程。带帧率压测时,如果 pending 持续非 0 或 duration p99 持续上升,应优先排查 Kafka broker、bridge batch size、NATS ack-pending 和网络。
### Realtime store update histogram
2026-07-03 已为 realtime-api Redis/MySQL store update 增加 duration histogram
```text
vehicle_realtime_store_update_duration_ms_histogram_bucket{store,protocol,status,le}
vehicle_realtime_store_update_duration_ms_histogram_count{store,protocol,status}
vehicle_realtime_store_update_duration_ms_histogram_sum{store,protocol,status}
```
该耗时覆盖 Redis 快速实时投影和 MySQL 异步当前态/位置投影的实际写入耗时。带帧率压测时Redis p99 上升会直接影响 realtime consumer 追平MySQL p99 上升需要结合 `vehicle_realtime_async_queue_depth``vehicle_realtime_async_queue_total{status="dropped"}` 判断是否开始丢弃低优先级当前态写入。
### Stat writer write histogram
2026-07-03 已为 stat-writer MySQL append 增加 duration histogram
```text
vehicle_stat_write_duration_ms_histogram_bucket{topic,status,le}
vehicle_stat_write_duration_ms_histogram_count{topic,status}
vehicle_stat_write_duration_ms_histogram_sum{topic,status}
```
该耗时覆盖单条 Kafka 消息解析后写入 `vehicle_daily_mileage` 的 MySQL append 路径。带帧率压测时,如果该 p99 持续上升并伴随 `vehicle_stat_kafka_lag` 增长,优先排查 MySQL 写入、每日里程表索引和 stat-writer 消费能力。
### Gateway frame duration histogram
2026-07-03 已为 Gateway 帧处理耗时增加 histogram
```text
vehicle_gateway_frame_duration_ms_histogram_bucket{protocol,status,le}
vehicle_gateway_frame_duration_ms_histogram_count{protocol,status}
vehicle_gateway_frame_duration_ms_histogram_sum{protocol,status}
```
该耗时覆盖单帧从解析、identity resolve、parsed fields 生成、publish enqueue 到协议响应写入的整体入口处理路径。带帧率压测时用它按协议观察 p95/p99不能只看 `vehicle_gateway_frame_duration_ms` 最后一条 gauge。
### Gateway identity resolve duration
2026-07-11 已为 Gateway 身份解析单独增加耗时 histogram 和超时分类:
```text
vehicle_gateway_identity_total{protocol,status}
vehicle_gateway_identity_duration_ms_histogram_bucket{protocol,status,le}
vehicle_gateway_identity_duration_ms_histogram_count{protocol,status}
vehicle_gateway_identity_duration_ms_histogram_sum{protocol,status}
```
生产默认使用周期身份快照,每帧只查本机内存;`status="timeout"` 主要用于兼容关闭 `IDENTITY_SNAPSHOT_ONLY_ENABLED` 后的回源查询模式。快照模式应重点观察 `vehicle_gateway_identity_snapshot_ready` 和最近成功刷新时间RDS 故障时入口继续接收,上一版快照继续服务,未知 phone 暂时 unresolved。
### History writer TDengine 批写
2026-07-03 已实现 history-writer 第一阶段批写:
- `history.Writer.AppendAllBatch` 按 TDengine child table 聚合 raw/location 多行 `INSERT`
- `cmd/history-writer` 默认 `HISTORY_WORKERS=3``HISTORY_BATCH_SIZE=200``HISTORY_BATCH_WAIT_MS=20`
- TDengine batch 成功后才批量提交 Kafka messages失败不提交 offset。
- 保留单条 `AppendAll` 路径作为回退。
- `production-track-telemetry-20260714082451` 起,`vehicle_locations` stable 增加 `soc_percent`。history-writer 在建表后执行幂等 `ALTER STABLE ... ADD COLUMN soc_percent DOUBLE`,只忽略 TDengine 明确的重复列错误;其他迁移错误会阻止 ready禁止带病继续写入。单条和批量位置 INSERT 都必须携带 SOC。
新增指标:
```text
vehicle_history_batch_flush_total{status}
vehicle_history_batch_rows_total{status}
vehicle_history_batch_pending_messages
vehicle_history_batch_pending_rows
vehicle_history_batch_flush_duration_ms{status}
```
进入带帧率压测时,必须同时观察 `vehicle_history_kafka_lag``vehicle_history_batch_pending_*``vehicle_history_batch_*`。如果 pending 持续非 0 但 Kafka lag 尚未放大,说明 history-writer 已经在本地批写或提交阶段形成早期积压。
## 下次恢复上下文时先做
1. 读本文件。
2.`docs/ops/vehicle-ingest-runbook.md`
3.`docs/architecture/production-data-plane-inventory.md`
4. 执行 `git status --short`
5. 如果涉及生产,先查 ECS 服务健康和最近日志。

View File

@@ -0,0 +1,567 @@
# 车辆接入生产运行手册
## 范围
本文覆盖 ECS `115.29.187.205` 上的 Go 原生车辆接入链路:
- GB32960 TCP 接入
- JT/T 808 TCP 接入
- 宇通 MQTT 接入
- NATS 到 Kafka 桥接
- Kafka 历史、统计、实时消费者
- Redis 实时缓存、MySQL 实时表、TDengine 历史表
运行手册的目标是按自上而下的顺序定位问题:入口、队列、桥接、消费、存储、查询 API。
当前生产服务、topic、表和 Redis key 的清单见 [生产数据面清单](../architecture/production-data-plane-inventory.md)。
10W 车辆容量目标、当前缺口和压测口径见 [100K 车辆接入容量基线](100k-capacity-baseline.md)。
## 服务地图
| 层级 | 服务 | systemd 单元 | 本机端点 |
| --- | --- | --- | --- |
| 接入 | Gateway | `lingniu-go-gateway.service` | `127.0.0.1:20211` |
| 桥接 | NATS Kafka bridge | `lingniu-go-nats-kafka-bridge.service` | `127.0.0.1:20214` |
| 字段投影 | Kafka RAW fields projector | `lingniu-go-fields-projector.service` | `127.0.0.1:20218` |
| 历史 | TDengine writer | `lingniu-go-history-writer.service` | `127.0.0.1:20212` |
| 统计 | MySQL stat writer | `lingniu-go-stat-writer.service` | `127.0.0.1:20213` |
| 实时写入 | Realtime writer | `lingniu-go-realtime-writer.service` | `127.0.0.1:20216` |
| 身份事实 | JT808 identity writer | `lingniu-go-identity-writer.service` | `127.0.0.1:20217` |
| 实时/API | Realtime API | `lingniu-go-realtime-api.service` | `127.0.0.1:20200` |
## Topic 配置基线
生产环境的 env 文件必须和 Go topic 命名空间保持一致:
| 服务 | 必要 topic |
| --- | --- |
| `lingniu-go-history-writer.service` | `vehicle.raw.go.gb32960.v1``vehicle.raw.go.jt808.v1``vehicle.raw.go.yutong-mqtt.v1` |
| `lingniu-go-stat-writer.service` | `vehicle.fields.go.gb32960.v1``vehicle.fields.go.jt808.v1``vehicle.fields.go.yutong-mqtt.v1` |
| `lingniu-go-realtime-writer.service` | `vehicle.raw.go.gb32960.v1``vehicle.raw.go.jt808.v1``vehicle.raw.go.yutong-mqtt.v1` |
| `lingniu-go-identity-writer.service` | `vehicle.raw.go.jt808.v1` |
| `lingniu-go-fields-projector.service` | 消费三类 `vehicle.raw.go.*`,把 RAW 中已有的 `parsed_fields` 投影到对应 `vehicle.fields.go.*` |
| `lingniu-go-nats-kafka-bridge.service` | NATS canonical raw 到 Kafka raw并从 raw 中已有的 `parsed_fields` 投影同协议 Kafka fields`vehicle.event.go.unified.v1` 只在兼容开关打开时使用 |
Gateway 在 NATS 模式下设置 `FIELDS_DERIVE_FROM_RAW_ENABLED=true`,每帧只发布 canonical rawbridge 设置 `BRIDGE_DERIVE_FIELDS_FROM_RAW_ENABLED=true`,只复用 raw envelope 中已计算一次的 `parsed_fields`不会重新解析协议原文。bridge 对同一条 NATS raw 生成 Kafka raw 和可选 Kafka fields两个输出都成功后才 ACK任一输出失败会保留 NATS 消息等待重放。登录、注册、鉴权等非实时帧只写 raw不生成 fields。
如果 stat-writer 少消费某个 fields topic对应协议的每日里程不会进入 `vehicle_daily_mileage`。stat-writer 启动时会拒绝 `vehicle.raw.*` 这类 RAW topic 配置,避免统计链路重新从原始报文解析并破坏 RAW/fields 解耦。修正后可能出现短时间 Kafka lag这是在追补历史 backlog`vehicle_stat_writes_total{status="ok"}` 只表示 append 调用成功,真正证明里程样本入库的是 `vehicle_stat_samples_total{status="written"}` 持续增长且 lag 下降。
stat-writer 默认 `STATS_WORKERS=3`,每个 worker 是同一 consumer group 的独立 Kafka reader。Kafka 按车辆 key 固定分区,因此单车累计里程仍按分区顺序执行,不同分区并行。调整 worker 后同时核对 `vehicle_stat_config{setting="workers"}`、每个 `vehicle_stat_worker_active{worker}`、MySQL 连接数、write p99 和 stat lagworker 不应超过 fields topic 的有效分区并行度。
stat-writer 对死锁、锁等待和网络中断等瞬时 MySQL 故障按 `STATS_RETRY_ATTEMPTS` 有限重试,重试耗尽后保持失败关闭并等待存储恢复。只有 `NULL`、数值越界、类型转换、字段过长和 CHECK 约束这类明确的单消息数据错误,才会在同样的有限重试耗尽后写入 `STATS_QUARANTINE_DIR` 并提交该 Kafka 偏移量,避免一条坏消息阻塞整个分区;表结构、权限和未知程序错误不会被隔离。隔离文件包含原 topic、partition、offset、key、value 和错误,默认权限为 `0600`,处理前不得删除。
history-writer 和 realtime-writer 同样默认分别使用 `HISTORY_WORKERS=3``REALTIME_WORKERS=3`。每个 worker 只顺序处理 Kafka 分配给自己的分区TDengine 表缓存、MySQL 实时节流缓存和车牌缓存均支持并发访问。调整后必须同时核对对应的 `*_config{setting="workers"}``*_worker_active`、pending/retry、后端连接数和 Kafka lag。
history-writer 启动会对既有 `vehicle_locations` stable 幂等补齐 `soc_percent DOUBLE`。发布后除 `/readyz` 外,必须抽查一台支持 SOC 的车辆在 TDengine 历史位置中 `soc_percent IS NOT NULL`;如果迁移报错且不是明确的重复列错误,服务应保持未就绪,先处理 TDengine DDL/权限,不能跳过迁移后强启。
identity-writer 默认使用 `IDENTITY_WRITER_WORKERS=3`。每个 worker 持有独立的位置触达节流器并共享 MySQL 连接池Kafka phone/vehicle key 保证同一终端在单一分区内有序,分区再平衡最多产生一次幂等触达。调整后核对 `vehicle_identity_writer_config{setting="workers"}``vehicle_identity_writer_worker_active`、pending/retry、MySQL 连接数和 identity lag。
Gateway 和 NATS bridge 启动时也会检查 raw/fields 合同Kafka raw topic 必须是 `vehicle.raw.*`fields topic 必须是 `vehicle.fields.*`NATS raw subject 和 fields subject 不允许相同。若服务启动失败并提示 `raw and fields ... must be different``fields kafka topic ... must start with "vehicle.fields."`,应先修正 `/opt/lingniu-go-native/env/*.env` 的 topic/subject 配置。
Realtime API 是当前态投影,默认在没有已提交 offset 时从 latest 开始消费。切换 topic 或新建 consumer group 后,不应让 realtime 追扫历史 raw backlog需要重建当前态时应使用明确的回放任务或手动 reset offset。
history、stat、realtime、identity 四个 Kafka 消费服务都会暴露 `vehicle_kafka_consumer_info{service,group,topic}``capacity-check` 会把缺少该指标的消费服务判为 degraded用来捕获“进程活着但没有挂上 Kafka topic”的配置错误。Identity writer 上线后 Gateway 必须设置 `JT808_REGISTRATION_GATEWAY_WRITES_ENABLED=false`,容量巡检会阻止双写 `jt808_registration`
业务端口:
- GB32960 TCP`0.0.0.0:32960`
- JT/T 808 TCP`0.0.0.0:808`
- 实时 writer`127.0.0.1:20216`
- 身份 writer`127.0.0.1:20217`
- 实时/API`0.0.0.0:20200`
## 协议鉴权配置
Gateway 的鉴权策略分为 `disabled``observe``enforce`。生产新增或变更账号时先使用 `observe`,确认 `vehicle_gateway_authentication_total` 中只有 `accepted` 后再切换 `enforce`
GB32960 多平台账号使用仅 root 可读的 JSON 文件,不把密码提交到 Git 或写入 systemd unit
```json
{
"platform-a": "current-password",
"platform-b": ["migration-old-password", "current-password"]
}
```
`gateway.env` 只配置文件地址和模式:
```bash
GB32960_PLATFORM_CREDENTIALS_FILE=/opt/lingniu-go-native/secrets/gb32960-platform-credentials.json
GB32960_AUTH_MODE=observe
JT808_AUTH_MODE=observe
JT808_REGISTER_AUTH_CODE=issued-auth-code
IDENTITY_RESOLVE_TIMEOUT_MS=50
```
Gateway 生产必须启用独立 WAL outbox不与旧 `NATS_SPOOL_DIR` 混用:
```bash
NATS_DURABLE_OUTBOX_ENABLED=true
NATS_OUTBOX_DIR=/opt/lingniu-go-native/spool/nats-outbox-wal
NATS_OUTBOX_MAX_INFLIGHT=10000
NATS_OUTBOX_ACK_TIMEOUT_MS=3000
NATS_OUTBOX_REPLAY_BATCH_SIZE=1000
NATS_OUTBOX_REPLAY_INTERVAL_MS=1000
NATS_OUTBOX_FSYNC=true
NATS_OUTBOX_CLOSE_TIMEOUT_MS=5000
NATS_OUTBOX_WAL_SEGMENT_BYTES=16777216
NATS_OUTBOX_WAL_SEGMENT_AGE_MS=5000
NATS_OUTBOX_WAL_APPEND_QUEUE_SIZE=100000
NATS_OUTBOX_WAL_COMMIT_BATCH_SIZE=256
NATS_OUTBOX_WAL_COMMIT_INTERVAL_MS=1
```
WAL 在分组 `fsync` 成功后才向协议处理线程返回接受成功网络发布不阻塞设备连接PubAck 前记录始终可恢复。正常值为 `vehicle_durable_outbox_backlog_records=0``vehicle_durable_outbox_inflight=0`,且 `submitted``acked` 最终一致。`submit_error``ack_error``remove_error``close_timeout` 任一增长都必须排查 NATS、磁盘和进程关闭过程。WAL 的 CRC 损坏会拒绝 Gateway 启动,禁止跳过或删除文件后强行启动;应先保留目录副本再恢复。
GB32960 登录密码在鉴权完成后会从 `parsed_fields` 删除,只保留 `password_present`;协议原始帧仍可能包含凭据,因此 RAW 查询接口和 TDengine `raw_hex` 必须按敏感数据控制访问。JT808 的 `0x0102` 会先匹配配置的公共鉴权码,再按规范化手机号匹配 `jt808_registration.auth_token` 的只读内存快照,全程不查询 MySQL指标 `source=device` 表示命中终端历史鉴权码。强制模式拒绝的鉴权帧不会覆盖该可信鉴权码和最近鉴权成功时间。JT808 的 `enforce` 不会拒绝未发注册/认证、直接上报 `0x0200` 的兼容平台;这类来源继续通过身份缺口指标治理。
## 标准发布流程
Go 原生服务发布必须使用 Linux amd64 构建产物,不要直接上传本机默认 `go build` 结果。仓库提供统一脚本:
```bash
cd go/vehicle-gateway
# 首次上线 identity-writer 前准备 systemd、env 和 Gateway 单写开关。
scripts/install-identity-writer-service.sh --host root@115.29.187.205
# 首次上线 fields-projector 前准备 systemd 和 env该步骤不会自动关闭 bridge 的兼容字段投影。
scripts/install-fields-projector-service.sh --host root@115.29.187.205
# 只构建并校验 ELF x86-64 产物。
scripts/deploy-ecs-release.sh --build-only --release verify-$(date +%Y%m%d%H%M%S)
# 构建、上传并校验 release但不切换 current、不重启服务。
scripts/deploy-ecs-release.sh \
--host root@115.29.187.205 \
--release staged-$(date +%Y%m%d%H%M%S) \
--stage-only
# 构建、上传新 release、切换 current、重启服务并执行 readyz/capacity-check。
scripts/deploy-ecs-release.sh \
--host root@115.29.187.205 \
--release production-$(date +%Y%m%d%H%M%S)
```
脚本会构建 `gateway``history-writer``stat-writer``identity-writer``realtime-api``nats-fast-writer``nats-kafka-bridge``fields-projector``capacity-check``load-sim``stats-backfill``identity-import`,并逐个校验 `file` 输出包含 `ELF 64-bit``x86-64`。正式发布按 bridge、fields projector、下游 writer、Gateway 的依赖顺序重启,确保 canonical raw 到 fields 的派生能力先于入口切换生效;发布窗口暂停 `lingniu-go-capacity-check.timer`,发布后恢复 timer、检查 9 个 systemd 服务、9 个 `/readyz``capacity-check``--stage-only` 只落盘 release不改变运行态。生产发布如需把业务统计缺口纳入发布门禁使用 `--capacity-args "-daily-mileage-diagnostics-max-actionable=50"` 这类参数,和 `lingniu-go-capacity-check.service` 的定时巡检口径保持一致。
## 五分钟排查顺序
先回答最核心的问题:数据有没有进来,有没有排队,有没有被消费,最后有没有被查询到。
1. 检查所有服务 `/readyz`
2. 检查 gateway 帧计数和 TCP 活跃连接。
3. 检查 NATS bridge 的 pending 和 ack-pending。
4. 检查 bridge 写 Kafka 与 NATS ack 是否同时增长。
5. 检查 history、stat、realtime、identity 四类 Kafka consumer lag。
6. 检查各 writer 的写入、提交、更新计数。
7. 最后再查存储和业务查询 API。
```bash
for port in 20211 20212 20213 20214 20215 20216 20217 20218 20200; do
curl -fsS "http://127.0.0.1:${port}/readyz"
echo
done
curl -fsS http://127.0.0.1:20211/metrics \
| grep -E 'vehicle_gateway_(active_connections|frames_total|last_frame|publish_total|last_publish)|vehicle_async_sink'
curl -fsS http://127.0.0.1:20214/metrics \
| grep -E 'vehicle_bridge_(nats_consumer|fields_projection|kafka_writes_total|last_kafka_write|nats_acks_total|last_ack|last_message)'
curl -fsS http://127.0.0.1:20212/metrics | grep vehicle_history_kafka_lag
curl -fsS http://127.0.0.1:20212/metrics | grep vehicle_history_batch
curl -fsS http://127.0.0.1:20213/metrics | grep vehicle_stat_kafka_lag
curl -fsS http://127.0.0.1:20213/metrics | grep vehicle_stat_samples_total
curl -fsS http://127.0.0.1:20213/metrics | grep vehicle_stat_sources_total
curl -fsS http://127.0.0.1:20213/metrics | grep vehicle_stat_last_
curl -fsS http://127.0.0.1:20213/metrics | grep vehicle_stat_quarantine_total
find /var/lib/lingniu-go-native/stat-writer-quarantine -maxdepth 1 -type f -printf '%f %s bytes\n'
curl -fsS http://127.0.0.1:20216/metrics | grep vehicle_realtime_kafka_lag
curl -fsS http://127.0.0.1:20217/metrics | grep vehicle_identity_writer_kafka_lag
for port in 20212 20213 20216 20217 20200; do
curl -fsS "http://127.0.0.1:${port}/metrics" | grep vehicle_kafka_consumer_info
done
/opt/lingniu-go-native/current/capacity-check
```
`capacity-check` 会解析 histogram 并做链路耗时早期预警,默认 `-gateway-frame-p99-ms=250``-gateway-identity-p99-ms=100``-async-sink-p99-ms=100``-async-sink-queue-wait-recent-p99-ms=100``-bridge-batch-p99-ms=5000``-fast-writer-stage-p99-ms=100``-history-flush-p99-ms=1000``-realtime-batch-p99-ms=500``-realtime-store-p99-ms=100``-stat-write-p99-ms=100``-kafka-lag-max=1000``-histogram-min-samples=100`。其中 realtime batch 是 Kafka projector 的整批 flush 耗时,主要看吞吐是否吃紧;单车实时投影延迟仍看 fast-writer Redis stage、realtime store update、Kafka lag 和 pending。批处理内部 in-flight pending 默认允许 `-fast-writer-batch-pending-max=1000``-history-batch-pending-max=1000``-history-rows-pending-max=5000``-stat-batch-pending-max=1000``-realtime-batch-pending-max=1000`,用于避免高频流量里百条级正在 flush/ack 的批次误报failure-closed 重试则由 `-consumer-retry-pending-max=0` 单独门禁history/realtime/stat/identity 任一 `*_retry_pending_messages` 非零都会降级设置为负数才关闭。bridge/fast-writer 的 unacked ratio 要达到 `-bridge-message-min-received=5000``-fast-writer-message-min-received=5000` 后才判断。刚重启样本不足时不判断 p99少量 Kafka lag 可能只是高频流量里的 in-flight 消息,只有 history/stat/realtime/identity lag 总和超过阈值才降级;压测阶段可以临时调低阈值。
topic 流转检查默认只抓完全断层:`-required-consumer-topics` 要求 history/realtime 消费三类 raw topic、stat 消费三类 fields topic`-gateway-field-min-raw=100` 检查某协议 raw publish 足够多,扣除登录、注册、鉴权等 `skipped_non_realtime` 后仍然没有 fields 直接发布或委托 bridge`-gateway-field-max-missing-ratio=0.20` 检查实时帧缺少预计算 fields 或发布错误比例是否过高;`-gateway-bridge-min-publishes=1000` 检查某协议 raw/委托 fields 已经由 gateway 接受,但 bridge 没有写对应 Kafka topic`-gateway-fast-writer-min-raw-publishes=1000` 检查某协议 raw 已经由 gateway 发布,但 fast-writer 没有消费同 raw subject 并成功写 Redis/ack`-raw-fanout-min-bridge-writes=1000` 检查某 raw topic 已经由 bridge 写入 Kafka但 history 未收到/写入或 realtime 未收到/更新;`-gateway-parse-min-frames=100``-gateway-parse-max-bad-ratio=0.05` 检查某协议入口帧解析质量是否突然恶化;`-gateway-response-min-samples=100``-gateway-response-max-error-ratio=0.05` 检查 32960 登录 ACK、808 注册/通用 ACK 等协议响应是否大量 build/write 失败,`skipped` 表示该消息类型无需响应,不计入比例;`-gateway-identity-snapshot-required=1``-gateway-identity-snapshot-max-age-seconds=180` 检查内存身份快照是否可用且持续刷新,`-gateway-identity-min-samples=100``-gateway-identity-max-unresolved-ratio=0.20` 检查某协议 VIN/车牌/phone 映射质量是否突然失效;`-history-parsed-field-min-frames=100``-history-parsed-field-max-missing-ratio=0.05` 检查 history-writer 收到的实时 raw envelope 是否缺失预计算 `parsed_fields``-fast-writer-redis-envelope-min-seen=100` 检查 Redis 快路径看到足够 raw envelope 后是否完全没有更新当前态,`-fast-writer-redis-envelope-max-bad-ratio=0.50` 检查缺 VIN/vehicle_key 的实时帧比例是否异常;`-fast-writer-redis-field-min-seen=1000``-fast-writer-redis-field-max-stale-ratio=0.20` 检查 Redis 快路径旧帧字段跳过比例是否异常升高;`-stat-topic-min-bridge-writes=1000` 检查某 fields topic 已经由 bridge 写入 Kafka但 stat-writer 未配置或未收到该 topic。统计链路还会用 `-stat-sample-min-writes=100` 检查“某个 fields topic 已成功 append 但完全抽不到里程样本”的配置/字段映射故障,并用 `-stat-sample-max-actionable-skip-ratio=0.60` 检查 fields 缺失、VIN 缺失、里程缺失、非正里程、source 缺失这类可行动跳过是否占比过高;`-stat-source-min-samples=100``-stat-source-max-missing-ratio=0.60` 检查已找到里程样本但来源端点缺失或 `vehicle_data_source` tracking 没有工作;`-stat-projection-min-sample-writes=100` 检查某 fields topic 已写入 source 候选里程样本但最终 `vehicle_daily_mileage` 投影完全没有写入;`skipped_same_mileage` 只是重复总里程去重,不计入该比例。`-last-activity-stale-seconds=300` 检查已出现过的成功/received `*_last_*_unix_seconds` 是否超过 5 分钟未刷新。上述阈值设为 `0` 可临时关闭;特殊环境可用 `-required-consumer-topics=''` 关闭消费 topic 合同检查。
每日里程业务诊断也会进入 `capacity-check` 的 JSON默认通过 `-daily-mileage-diagnostics-url=http://127.0.0.1:20200/api/stats/daily-metrics/diagnostics/reasons` 查询当天原因汇总,输出 `daily_mileage_diagnostics.vehicle_total``actionable_issue_total``items`。二进制默认 `-daily-mileage-diagnostics-max-actionable=-1` 表示只展示不阻断;生产定时巡检和发布门禁应显式设置一个阈值,例如当前已知 MQTT 稀疏总里程期间可先用 `50`,待问题清理后收紧到 `0`。诊断接口不可访问会降级,因为这表示 MySQL/API 观测链路不可用;临时环境可用 `-daily-mileage-diagnostics-url=''` 关闭。
## 健康基线
当前生产环境的健康特征:
- 所有 `/readyz` 都返回 `status=ok`
- `vehicle_bridge_nats_consumer_ack_pending``0`
- history、stat、realtime 的 Kafka lag 为 `0` 或短时间小幅波动后归零。
- history、stat、realtime 都能看到 `vehicle_kafka_consumer_info`
- `capacity-check` 返回 `status=ok` 且退出码为 `0`
- gateway 的帧计数持续增长。
- bridge 的 Kafka write 和 NATS ack 计数同时增长。
- writer 的成功计数增长,同时 Kafka lag 不持续扩大;统计链路还应看到 `vehicle_stat_samples_total{status="written"}` 随有效里程字段流量增长。
生产默认日志级别保持 `LOG_LEVEL=info`。Gateway 的 TCP 连接打开、关闭、空闲超时属于高频事件,只保留在 debug 级别;排查单车连接问题时可以临时设置 `LOG_LEVEL=debug`,排查结束后应恢复 info常态监控以 `vehicle_gateway_active_connections``vehicle_gateway_connection_closes_total` 为准。
定时容量巡检由 systemd timer 触发:
```bash
systemctl status lingniu-go-capacity-check.timer
journalctl -u lingniu-go-capacity-check.service --since '10 minutes ago' --no-pager
```
如果 `lingniu-go-capacity-check.service` 失败,先看 journal JSON 里的 `findings`,再按对应层级处理。
GB32960 和 JT/T 808 的活跃连接数受上游平台连接方式影响,不能直接等同于车辆数。突然归零或持续异常下降才是信号。
计数器只说明服务启动后累计发生过,不代表当前仍有流量。排查断流时同时看 `*_last_*_unix_seconds`gateway 看 `vehicle_gateway_last_frame_unix_seconds``vehicle_gateway_last_publish_unix_seconds`bridge 看 `vehicle_bridge_last_kafka_write_unix_seconds`fast-writer 看 `vehicle_fast_writer_last_message_unix_seconds``vehicle_fast_writer_last_stage_unix_seconds`history/stat/realtime 分别看对应 `last_message``last_write/update``last_commit`。如果 counter 很高但 last 时间长时间不变,说明对应协议或 topic 已经停止活动。`capacity-check` JSON 会额外派生 `*_last_*_age_seconds`,现场排查优先看 age 是否持续扩大。
## 压测入口
Go 版本提供 `cmd/load-sim` 用于阶段性连接和帧写入压测。压测生产入口前必须先确认上游真实数据窗口,避免和业务流量混淆。
```bash
cd /opt/lingniu-go-native/current/go/vehicle-gateway
go run ./cmd/load-sim \
-protocol jt808 \
-addr 127.0.0.1:808 \
-connections 100 \
-connect-rate 100 \
-send-interval 10s \
-duration 2m \
-template 0200 \
-send=false
```
带帧压测 JT808 时必须读取服务端应答,并使用隔离手机号区间和自动清理:
```bash
export MYSQL_DSN="$(sed -n 's/^MYSQL_DSN=//p' /opt/lingniu-go-native/env/base.env | tail -n 1)"
/opt/lingniu-go-native/current/load-sim \
-protocol jt808 \
-addr 127.0.0.1:808 \
-connections 1000 \
-connect-rate 500 \
-send-interval 1s \
-duration 60s \
-template 0200 \
-jt808-phone-base 139000012000 \
-drain-responses=true \
-cleanup-registration
```
输出必须满足 `connections_failed=0``write_errors=0``response_bytes>0``read_errors=0`。每轮使用未在最近 10 分钟压测过的新手机号区间时,结束后的 `rows_deleted` 应等于模拟连接数;重复号段可能被 identity-writer 的 location touch 节流跳过写入,因此清理数为 `0` 也不表示删除失败。`-cleanup-only` 可用于异常中断后的补清理;删除条件固定受手机号区间和 loopback 来源约束不清理真实注册。Bridge 默认 `BRIDGE_KAFKA_WRITE_CONCURRENCY=6`,同一批次按 topic 并行写 KafkaRAW 和 fields 均成功后才 ACK NATS 源消息。当前生产压测基线使用 `NATS_ASYNC_RAW_WORKERS=128``FAST_WRITER_WORKERS=16``FAST_WRITER_FETCH_WAIT_MS=5`
## 告警阈值建议
| 信号 | 建议阈值 | 含义 |
| --- | --- | --- |
| `/readyz` 非 ok | 立即处理 | 服务或依赖不可用。 |
| Gateway 活跃连接 | 预期有流量时某协议降为 `0` | 上游网络、监听端口或进程可能异常。 |
| `vehicle_gateway_connection_closes_total{reason="read_error"}` | 连续增长 | 入口 TCP 读失败,优先查网络、客户端断连和内核连接状态。 |
| `vehicle_gateway_connection_closes_total{reason="extract_error"}` | 连续增长 | 报文边界或协议提取异常,优先抽查 raw 日志和协议 extractor。 |
| `vehicle_gateway_connection_closes_total{reason="read_timeout"}` | 突然高于历史基线 | 车辆长时间无上报或链路空闲超时,需结合在线数和上游平台状态判断。 |
| `vehicle_gateway_frames_total{status!="OK"}` | 单协议样本达到 `capacity-check -gateway-parse-min-frames` 后,异常比例超过 `capacity-check -gateway-parse-max-bad-ratio` | 解析器或上游报文质量异常。优先抽查 raw 日志、协议 extractor、最近部署的解析规则和上游转发内容。 |
| `vehicle_gateway_frame_duration_ms_histogram_bucket` | p99 超过 `capacity-check -gateway-frame-p99-ms` 或连续 5 分钟超过容量目标 | Gateway parse、identity resolve、publish enqueue 或响应链路变慢。 |
| `vehicle_gateway_response_total{status="build_error"}` / `status="write_error"` | 单协议响应尝试样本达到 `capacity-check -gateway-response-min-samples` 后,错误比例超过 `capacity-check -gateway-response-max-error-ratio` | 协议响应构造或 TCP 写回失败。优先查 32960 登录/实时 ACK、808 注册/通用 ACK、连接断开、响应编码和上游是否提前断链。 |
| `vehicle_gateway_response_e2e_recent_p99_ms` | 单协议样本达到 `capacity-check -histogram-min-samples` 后超过 `capacity-check -gateway-response-recent-p99-ms`,默认 `100ms` | 收帧、解析、内存身份映射、RAW 入队或协议写回变慢。结合 identity、async queue 和 NATS 指标定位具体阶段。 |
| `vehicle_gateway_authentication_total{mode="observe",status!="accepted"}` | 任意增长 | 当前登录/认证与配置不一致但仍放行。先核对账号覆盖、上游迁移密码和 JT808 认证码,确认无误后才能切换 `enforce`。 |
| `vehicle_gateway_authentication_total{mode="enforce",status!="accepted"}` | 任意增长 | 网关已返回协议失败并关闭该连接;检查是否为未授权来源、过期账号或配置错误。 |
| `vehicle_gateway_publish_total{kind="fields"}` | 某协议 `raw` 达到 `capacity-check -gateway-field-min-raw` 但 fields 直接发布或 `delegated` 都为 0 | Gateway 收到实时帧但没有可供 bridge 派生的 `parsed_fields`,优先查协议解析和字段映射。 |
| `vehicle_gateway_fields_total{status="skipped_non_realtime"}` | 持续增长但 fields publish 正常 | Gateway 正确跳过登录、注册、鉴权、心跳等非实时帧,不会进入统计 topic如果实时数据也断流再查上游是否只在发控制帧。 |
| `vehicle_gateway_fields_total{status="skipped_missing_fields"}` / `publish_error` | 单协议达到 `capacity-check -gateway-field-min-raw` 后,缺失比例超过 `capacity-check -gateway-field-max-missing-ratio` | 实时帧没有生成扁平字段或兼容直发模式发布失败。NATS canonical raw 模式优先查协议字段映射和 `parsed_fields`。 |
| `vehicle_gateway_publish_total{kind="raw"}` / `vehicle_gateway_publish_total{kind="fields",status="delegated"}` | 某协议某 kind 达到 `capacity-check -gateway-bridge-min-publishes` 但 bridge 无对应 Kafka topic 写入 | Gateway 已接受 raw/fields 派生任务,但 bridge 没有写 Kafka。优先查 bridge projection、route、NATS subject、Kafka write 和 bridge 日志。 |
| `vehicle_gateway_publish_total{kind="raw"}` | 某协议 raw 达到 `capacity-check -gateway-fast-writer-min-raw-publishes` 但 fast-writer 无同 raw subject `ok` | Gateway 已经发布到 NATS但 Redis 快路径未消费或未成功写入/ack。优先查 `vehicle_fast_writer_messages_total``vehicle_fast_writer_stage_duration_ms_histogram_bucket{stage="redis"}``stage="ack"`、NATS durable pending 和 Redis 健康。 |
| `vehicle_gateway_last_frame_unix_seconds` / `vehicle_gateway_last_publish_unix_seconds` | 已出现过的成功 activity 超过 `capacity-check -last-activity-stale-seconds` 未刷新 | 对应协议入口或 publish 已停止活动,先对比连接数、上游转发状态和 bridge last 指标。 |
| `vehicle_gateway_identity_total{status!="resolved"}` | 单协议样本达到 `capacity-check -gateway-identity-min-samples` 后,异常比例超过 `capacity-check -gateway-identity-max-unresolved-ratio` | VIN/车牌/phone 映射质量下降。优先查 `vehicle_identifier` 导入、`vehicle_identity_binding` 基础车牌/VIN、`jt808_registration` phone 状态和 resolver 缓存 TTL。 |
| `vehicle_gateway_identity_total{status="timeout"}` | 任意持续增长 | Gateway 身份解析超过 `IDENTITY_RESOLVE_TIMEOUT_MS`,帧会继续 partial/unresolved 下发,但 VIN/车牌映射可能滞后。 |
| `vehicle_gateway_identity_duration_ms_histogram_bucket` | p99 超过 `capacity-check -gateway-identity-p99-ms` 或接近 `IDENTITY_RESOLVE_TIMEOUT_MS` | MySQL/缓存身份解析开始拖慢入口链路,优先查 `vehicle_identifier``vehicle_identity_binding``jt808_registration` 索引和 RDS 状态。 |
| `vehicle_gateway_identity_cache_entries` | 超过 `capacity-check -gateway-identity-cache-entries-max`,默认 `300000` | Gateway 身份解析缓存超过容量预期。优先确认 `IDENTITY_LOOKUP_CACHE_MAX_ENTRIES``IDENTITY_LOOKUP_CACHE_TTL_SECONDS``IDENTITY_LOOKUP_CACHE_CLEANUP_INTERVAL_SECONDS`,并检查上游是否出现大量异常 phone、source IP 或历史回放。 |
| `vehicle_gateway_identity_snapshot_ready` / `vehicle_gateway_identity_snapshot_last_success_unix_seconds` | ready 不是 `1`,或最近成功刷新超过 `180s` | Gateway 会继续接入并保留 raw但新 808 phone 可能暂时无法映射 VIN。检查 RDS 连通性和四张身份来源表;恢复后周期刷新会自动发布新快照,不需要重启。 |
| `vehicle_durable_outbox_backlog_records` / `vehicle_durable_outbox_inflight` | 持续非 0 或超过容量门禁 | Gateway 已持久接受但尚未收到 JetStream PubAck。短时突发可自动回落持续增长时检查 NATS 可用性、磁盘延迟、ACK timeout 和 inflight 上限。 |
| `vehicle_durable_outbox_publish_total{status=~"submit_error|ack_error|remove_error|close_timeout"}` | 任意增长 | WAL 后的异步发布、PubAck、分段删除或关闭等待失败。记录不会因提交/ACK 失败丢失;修复依赖后由 replay 自动恢复。 |
| `vehicle_async_sink_queue_depth{sink="nats"}` | 持续增长且不回落,或超过 `capacity-check -async-sink-queue-depth-max`,默认 `10000` | Gateway 到 NATS/Kafka 的异步 publish 队列开始积压。结合 `vehicle_async_sink_queue_capacity` 判断是否接近满队列NATS 入队等待由 `NATS_ASYNC_ENQUEUE_TIMEOUT_MS` 控制Kafka fallback 由 `KAFKA_ASYNC_ENQUEUE_TIMEOUT_MS` 控制。 |
| `vehicle_async_sink_queue_wait_recent_p99_ms` | 样本达到 `capacity-check -histogram-min-samples` 后超过 `-async-sink-queue-wait-recent-p99-ms`,默认 `100ms` | 秒级突发曾造成排队,即使当前 queue depth 已回到 0 也会被最近 512 样本窗口捕获。结合 `vehicle_async_sink_workers`、publish duration 和 NATS pending 判断是 worker 不足还是持久化确认变慢。 |
| `vehicle_async_sink_enqueue_total{status="timeout"}` | 任意增长 | Gateway publish 队列已满或 worker 长时间阻塞,入口会快速失败并记录 publish error避免连接处理 goroutine 长时间堆积。优先查 NATS/Kafka 写入延迟、async worker 数、队列容量和 bridge/fast-writer pending。 |
| `vehicle_async_sink_publish_total{status="error"}` | 连续增长 | NATS/Kafka publish 失败,需要先查中间件连接和日志。 |
| `vehicle_async_sink_publish_duration_ms_histogram_bucket` | p99 超过 `capacity-check -async-sink-p99-ms` 或连续 5 分钟上升 | Gateway async worker 写 NATS/Kafka 变慢,通常会带动 queue depth 增长。 |
| `vehicle_history_batch_flush_total{status="error"}` | 任意增长 | TDengine 批写失败Kafka offset 不会提交,应先查 TDengine 和 SQL 错误。 |
| `vehicle_history_parsed_fields_total{status="missing"}` | 单 topic/protocol 实时帧样本达到 `capacity-check -history-parsed-field-min-frames` 后,缺失比例超过 `capacity-check -history-parsed-field-max-missing-ratio` | history-writer 收到的实时 raw envelope 没有携带预计算 `parsed_fields`TDengine RAW 证据层会缺少扁平化解析字段。优先查 gateway `BuildFieldsEnvelope` / raw envelope 构造、协议字段映射和最近部署。 |
| `vehicle_history_location_writes_total{status="error"}` / `vehicle_history_last_location_write_unix_seconds{status="error"}` | 累计错误超过 `-history-location-error-max`,且最后一次错误距今不超过 `-history-location-error-recent-seconds`(默认 `300s` | TDengine 位置派生表最近仍有写入失败。优先查异常设备时间、坐标投影和 TDengine 时间范围;错误窗口恢复后健康状态自动恢复,累计 counter 仍用于审计。缺少最后错误时间时保持降级,避免观测缺失被误判为恢复。 |
| `vehicle_history_batch_pending_messages` | 超过 `capacity-check -history-batch-pending-max` 或 burst 后不回落 | history-writer 已拉取但未完成写入/提交,可能卡在 TDengine 或 Kafka commit。 |
| `vehicle_history_batch_pending_rows` | 超过 `capacity-check -history-rows-pending-max` 或 burst 后不回落 | TDengine 有批量写入积压,通常早于 Kafka lag 放大。 |
| `vehicle_history_batch_flush_duration_ms_histogram_bucket{status="ok"}` | p99 超过 `capacity-check -history-flush-p99-ms` 或持续上升 | TDengine 写入延迟增加,可能需要降低 batch size 或扩容 TDengine。 |
| `vehicle_history_kafka_messages_total{status="invalid_json"}` | 超过 `capacity-check -history-invalid-json-max`,默认 `0` | Kafka raw topic 中出现无法反序列化的 envelope。history-writer 会 commit 并隔离该消息,避免阻塞历史落库;必须检查 NATS bridge 写入 Kafka 的 payload、raw topic 是否混入非 envelope 数据,以及 gateway 发布合同。 |
| `vehicle_bridge_nats_consumer_ack_pending` | 连续 2 分钟 `> 0` | 消息已投递给 bridge但 Kafka 写入后未完成 ack。 |
| `vehicle_bridge_nats_consumer_pending` | 持续增长且 `> 10000` | bridge 消费 NATS 的速度跟不上生产速度。 |
| `vehicle_bridge_batch_pending_messages` | 持续非 0 或 burst 后不回落 | bridge 已拉取 NATS 消息但尚未完成 Kafka 写入和 NATS ack。 |
| `vehicle_bridge_batch_duration_ms_histogram_bucket` | p99 超过 `capacity-check -bridge-batch-p99-ms` 或连续 5 分钟上升 | Kafka 写入或 NATS ack 开始变慢,通常会先于 ack-pending 扩大。 |
| `vehicle_bridge_fields_projection_total` | `invalid_json``invalid_envelope``protocol_mismatch``skipped_missing_fields``marshal_error` 增长 | canonical raw 无法派生统计 fields。raw 仍进入历史链路,但每日里程可能缺样本;按协议检查 Gateway `parsed_fields` 和字段合同。 |
| `vehicle_bridge_messages_total{status="route_error"}` / `vehicle_bridge_nats_acks_total{status="dropped_route_error"}` | route_error 超过 `capacity-check -bridge-route-error-max`,默认 `0` | NATS stream 中出现 bridge 未配置的 subject。bridge 会 ACK 并丢弃该消息,避免 poison message 阻塞正常 subject必须检查 `NATS_STREAM_SUBJECTS``NATS_FILTER` 和 raw/fields subject 到 Kafka topic 的路由配置。 |
| `vehicle_bridge_kafka_writes_total{topic="vehicle.fields.go.*"}` | 某 fields topic 达到 `capacity-check -stat-topic-min-bridge-writes` 但 stat-writer 无同 topic consumer 或 received | fields 已进 Kafka但统计消费配置或消费循环断层。 |
| `vehicle_bridge_kafka_writes_total{topic="vehicle.raw.go.*"}` | 某 raw topic 达到 `capacity-check -raw-fanout-min-bridge-writes` 但 history/realtime 无同 topic received/write/update | raw 已进 Kafka但历史落库或当前态投影断层。先查对应服务 `vehicle_kafka_consumer_info`,再查 `vehicle_history_kafka_messages_total` / `vehicle_history_writes_total` / `vehicle_realtime_kafka_messages_total` / `vehicle_realtime_updates_total`。 |
| `vehicle_bridge_last_message_unix_seconds` / `vehicle_bridge_last_kafka_write_unix_seconds` / `vehicle_bridge_last_ack_unix_seconds` | 已出现过的成功 activity 超过 `capacity-check -last-activity-stale-seconds` 未刷新 | NATS 到 Kafka 桥接某 subject/topic 停止流动;先查 NATS pending、Kafka 写入和 ack。 |
| `vehicle_fast_writer_nats_consumer_ack_pending` | 超过 `100` 或持续不回落 | 消息已投递给 fast-writer但 Redis 写入后未完成 ack短暂个位数/十几条通常只是高频流量里的 NATS ack 飞行窗口,如果显式启用 TDengine stage也可能卡在 TDengine。 |
| `vehicle_fast_writer_nats_consumer_pending` | 持续增长且 `> 10000` | fast-writer 消费 NATS 的速度跟不上入口写入速度。 |
| `vehicle_fast_writer_batch_pending_messages` | 超过 `capacity-check -fast-writer-batch-pending-max` 或 burst 后不回落 | fast-writer 已拉取 NATS 消息但尚未完成 Redis 写入和 ack如果显式启用 TDengine stage也可能卡在 TDengine。 |
| `vehicle_fast_writer_batch_pending_envelopes` | 持续非 0 或 burst 后不回落 | fast-writer 当前批次已有有效 envelope 在等待落库或 ack。 |
| `vehicle_fast_writer_stage_duration_ms_histogram_bucket` | 某个 stage 的 p99 超过 `capacity-check -fast-writer-stage-p99-ms` 或连续 5 分钟上升 | NATS 快速写链路在 Redis 或 NATS ack 某一阶段变慢;生产默认 `FAST_WRITER_TDENGINE_ENABLED=false`,如果临时启用 TDengine stage再观察 `stage="tdengine"`。 |
| `vehicle_fast_writer_messages_total{status="invalid_json"}` | 超过 `capacity-check -fast-writer-invalid-json-max`,默认 `0` | NATS raw subject 中出现无法反序列化的 envelope。fast-writer 会 ACK 并隔离该消息,避免阻塞 Redis 当前态;必须检查 gateway 发布 payload、NATS stream 是否混入非 envelope 数据,以及 bridge/fast-writer 的 subject 过滤配置。 |
| `vehicle_fast_writer_redis_envelopes_total{status="skipped_non_realtime"}` | 持续增长但 `updated` 正常 | raw topic 中存在登录、注册、鉴权等非实时帧Redis 当前态会跳过它们,不代表污染当前态;如果 `updated=0` 且只有 skip 增长,应检查 gateway 是否停止发送实时帧。 |
| `vehicle_fast_writer_redis_envelopes_total{status="skipped_missing_vin"}` / `skipped_missing_vehicle_key` | 持续增长 | 实时帧进入 Redis 快路径但身份解析不完整,当前态不会写入。优先查 identity resolver、`vehicle_identity_binding``jt808_registration`、MQTT VIN 映射和 GB32960 VIN 字段。 |
| `vehicle_fast_writer_redis_envelopes_total{status="skipped_missing_fields"}` | 任意持续增长 | Gateway 发布了实时 RAW但没有附带一次计算后的 `parsed_fields`。RAW 仍写历史Redis 在线/KV 和 MySQL snapshot/location 均跳过;检查协议 parser、字段映射和 Gateway envelope 构造,禁止在 writer 中增加重新解析兜底。 |
| `vehicle_fast_writer_redis_fields_total{status="skipped_stale"}` | 某协议持续增长且占 `status="seen"` 比例异常升高 | Redis 当前态收到大量旧事件时间字段,旧帧不会覆盖新字段,但会影响上游时序质量判断。优先按 `subject` 查上游平台补发、设备时间漂移、NATS backlog 和 gateway event-time 归一化记录。 |
| `vehicle_fast_writer_last_message_unix_seconds` / `vehicle_fast_writer_last_stage_unix_seconds` | 已出现过的成功 activity 超过 `capacity-check -last-activity-stale-seconds` 未刷新 | Redis 快路径某 raw subject 或具体 stage 停止活动。优先对比 gateway last publish、NATS pending、`vehicle_fast_writer_messages_total`、Redis 健康和 `stage="redis"` / `stage="ack"` 的最近时间。 |
| `vehicle_fast_writer_tdengine_enabled` | 生产非 `0` | fast-writer 正在写 TDengine可能与 history-writer 双写 RAW/位置证据层;`capacity-check` 会降级,除压测或故障切换外应保持 `0`。 |
| `vehicle_kafka_consumer_info` | history/stat/realtime 缺失 | 服务进程可能启动了,但 Kafka broker、group 或 topic 配置没有生效。 |
| `vehicle_history_config{setting="workers"}` / `vehicle_history_worker_active` | 配置低于 `-history-workers-min=3` 或任一 worker 不活跃 | 检查 `HISTORY_WORKERS`、Kafka group 分区分配、进程日志和 TDengine 连接承载。 |
| `capacity-check` finding: `required kafka consumer topic missing` | history/realtime/stat 缺少生产合同 topic | 对照 `/opt/lingniu-go-native/env/*.env``KAFKA_TOPICS``KAFKA_TOPIC` 配置。history/realtime 应有三类 raw topicstat 应有三类 fields topic。 |
| `vehicle_realtime_redis_projector_enabled` | 生产非 `0` | realtime-api 正在从 Kafka 重写 Redis可能与 NATS fast-writer 双写当前态;`capacity-check` 会降级,生产应保持 `0`。 |
| `vehicle_realtime_kafka_messages_total{status="invalid_json"}` | 超过 `capacity-check -realtime-invalid-json-max`,默认 `0` | realtime-api 的 Kafka 当前态投影收到无法反序列化的 raw envelope。服务会 commit 并隔离该消息,避免阻塞 Redis/MySQL 当前态;必须检查 bridge 写入 Kafka 的 payload、raw topic 污染和 gateway 发布合同。 |
| `vehicle_realtime_store_update_duration_ms_histogram_bucket{store="redis"}` | 生产出现或 p99 连续 5 分钟上升 | 仅临时启用 realtime Redis projector 时使用;常规 Redis 写入延迟看 `vehicle_fast_writer_stage_duration_ms_histogram_bucket{stage="redis"}`。 |
| `vehicle_realtime_store_update_duration_ms_histogram_bucket{store="mysql"}` | p99 超过 `capacity-check -realtime-store-p99-ms` 或连续 5 分钟上升 | MySQL 当前态/位置投影变慢,需结合 async queue depth 和 dropped 计数判断。 |
| `vehicle_realtime_config{setting="workers"}` / `vehicle_realtime_worker_active` | 配置低于 `-realtime-workers-min=3` 或任一 worker 不活跃 | 检查 `REALTIME_WORKERS`、Kafka group 分区分配和 MySQL 连接/锁等待。 |
| `vehicle_identity_writer_config{setting="workers"}` / `vehicle_identity_writer_worker_active` | 配置低于 `-identity-workers-min=3` 或任一 worker 不活跃 | 检查 `IDENTITY_WRITER_WORKERS`、JT808 raw 分区分配和 MySQL 连接/锁等待。 |
| `vehicle_realtime_async_queue_total{status="dropped"}` / `status="closed"` | 任意增长 | MySQL 当前态/位置投影的异步二级队列没有接收该帧。`dropped` 表示队列满,优先查 MySQL 写入延迟和 `MYSQL_REALTIME_ASYNC_WORKERS``closed` 只应出现在服务退出窗口,退出时会先 drain 已排队数据再关闭 MySQL 连接。 |
| `vehicle_realtime_plate_cache_entries` | 超过 `capacity-check -realtime-plate-cache-entries-max`,默认 `200000` | Realtime API 的 VIN 到车牌缓存超过容量预期。优先确认 `PLATE_CACHE_MAX_ENTRIES``PLATE_CACHE_TTL_SECONDS``vehicle_identity_binding` 的 VIN 质量;若短时间快速增长,通常说明上游出现大量异常 VIN 或历史回放范围过大。 |
| `vehicle_stat_write_duration_ms_histogram_bucket` | p99 超过 `capacity-check -stat-write-p99-ms` 或连续 5 分钟上升 | MySQL 每日里程统计写入变慢,可能导致 stat Kafka lag 增长。 |
| `vehicle_stat_kafka_messages_total{status="invalid_json"}` | 超过 `capacity-check -stat-invalid-json-max`,默认 `0` | Kafka fields topic 中出现无法反序列化的 envelope。stat-writer 会 commit 并隔离该消息,避免阻塞每日里程统计;必须检查 gateway fields 发布、NATS bridge 路由以及是否误把 RAW/非 envelope 数据写入 fields topic。 |
| `vehicle_stat_batch_pending_messages` | 超过 `capacity-check -stat-batch-pending-max`,默认 `1000` | stat-writer 已从 Kafka 拉取一批 fields 消息,但还没完成 MySQL append 和 Kafka commit。短暂非 0 是批处理窗口;持续升高优先查 MySQL 写入延迟、`vehicle_stat_write_duration_ms_histogram_bucket` 和 stat Kafka lag。 |
| `vehicle_stat_cache_entries` / `vehicle_stat_cache_evictions_total` | entries 超过 `capacity-check -stat-cache-entries-max`,默认 `1000000`evictions 持续增长 | stat-writer 的去重、source touch、投影节流、历史基线缓存达到运行时上限。优先确认 `STATS_CACHE_MAX_ENTRIES``STATS_CACHE_RETENTION_HOURS``STATS_CACHE_CLEANUP_INTERVAL_SECONDS`,并检查是否有大范围历史回放、异常 VIN/source 或 Kafka 积压恢复。 |
| `vehicle_stat_samples_total{status="found"}` | 单 topic `vehicle_stat_writes_total{status="ok"}` 达到 `capacity-check -stat-sample-min-writes``found=0` | fields topic 有消费但完全抽不到里程样本,优先检查 VIN 映射、协议字段映射和总里程字段是否进入 fields该指标带 `protocol` 标签,可直接按 GB32960/JT808/YUTONG_MQTT 聚合。 |
| `vehicle_stat_samples_total{status="skipped_missing_fields"}` | 任意持续增长,或单 topic 可行动跳过数 / `vehicle_stat_writes_total{status="ok"}` 超过 `capacity-check -stat-sample-max-actionable-skip-ratio` | fields topic 收到了 envelope 但没有扁平化 `fields`。优先查 gateway `BuildFieldsEnvelope`、NATS/Kafka bridge topic 映射,以及是否误把 RAW envelope 写进 fields topic。 |
| `vehicle_stat_samples_total{status="skipped_missing_vin"}` / `skipped_missing_mileage` / `skipped_non_positive_mileage` / `skipped_missing_source` | 单 topic 可行动跳过数 / `vehicle_stat_writes_total{status="ok"}` 超过 `capacity-check -stat-sample-max-actionable-skip-ratio` | fields topic 有消费但大量样本无法进入每日里程。按协议和跳过原因分别检查 VIN 映射、总里程字段映射、里程单位/异常值、`vehicle_data_source` 关联。 |
| `vehicle_stat_samples_total{status="event_time_future_adjusted"}` | 连续增长或集中在某协议/source | 设备事件时间比接收时间未来超过 10 分钟统计已回退到接收时间。优先检查上游平台或终端时钟RAW 帧仍保留设备原始时间,业务 location、Redis 当前态和每日里程使用归一后的时间。 |
| `vehicle_stat_samples_total{status="written"}` | stat-writer `ok` 增长但 `written` 不增长 | fields topic 有消费但没有有效里程样本入库,按 `skipped_missing_vin``skipped_missing_mileage``skipped_non_positive_mileage``skipped_missing_source``skipped_same_mileage` 定位原因;如果主要是 `skipped_same_mileage`,通常表示上游总里程未变化,不等同故障。 |
| `vehicle_stat_sources_total{status="written"}` / `skipped_throttled` | `vehicle_stat_samples_total{status="found"}` 增长但 source tracking 无增长 | 来源管理链路没有维护 `vehicle_data_source`。优先查 source endpoint 是否为空、`vehicle_data_source` 表结构、MySQL 写入错误和 `STATS_SOURCE_TOUCH_INTERVAL_SECONDS`。 |
| `vehicle_stat_sources_total{status="skipped_missing_endpoint"}` | 单 topic 缺 endpoint / found 超过 `capacity-check -stat-source-max-missing-ratio` | fields envelope 缺 `source_endpoint`,无法区分平台来源和维护 source。优先查 Gateway 是否填充 TCP remote/MQTT endpoint以及 NATS/Kafka bridge 是否保留 envelope 字段。 |
| `vehicle_stat_projections_total{status="written"}` / `skipped_throttled` | source 候选持续写入但最终日里程更新频率低 | `vehicle_daily_mileage_source` 是每样本候选事实层,`vehicle_daily_mileage` 是最终投影层。`skipped_throttled` 增长通常表示受 `STATS_PROJECT_INTERVAL_SECONDS` 节流,不代表样本丢失;如需强一致核验可临时设投影间隔为 `0`。 |
| `vehicle_history_last_*_unix_seconds` / `vehicle_stat_last_*_unix_seconds` / `vehicle_realtime_last_*_unix_seconds` | 已出现过的成功 activity 超过 `capacity-check -last-activity-stale-seconds` 未刷新 | 下游消费、写入或 commit 某 topic 停止活动;结合 Kafka lag 和对应存储健康判断。 |
| `vehicle_stat_project_interval_seconds` | 生产异常为 `0` 或被调得过小 | `vehicle_daily_mileage_source` 仍每样本更新,但 `vehicle_daily_mileage` 选举投影会被节流;过小会放大 MySQL 压力。 |
| Kafka lag | 连续 5 分钟增长或 `> 10000` | 下游 consumer 或存储存在瓶颈。 |
| Writer 成功计数 | 入口增长但 writer 不增长 | bridge、Kafka、consumer 或存储链路断开。 |
## 事故处理路径
### 没有新数据
1. 先确认监听端口。
```bash
ss -lntp | grep -E ':(808|32960|20200|20211|20212|20213|20214|20215|20216) '
```
2. 检查 gateway readiness 和日志。
```bash
curl -fsS http://127.0.0.1:20211/readyz
journalctl -u lingniu-go-gateway.service --since '10 minutes ago' --no-pager
```
3. 看 gateway 帧计数是否增长。
4. 如果 gateway 有增长但 bridge 没增长,查 NATS publish 和 bridge 日志。
5. 如果 bridge 写入增长但 writer 不增长,查 Kafka lag 和 consumer 日志。
### Kafka lag 持续增长
1. 先从 metrics 定位是哪个服务、哪个 topic、哪个 partition。
2. 检查对应服务 `/readyz`
3. 检查存储依赖history 看 TDenginestat 看 MySQLrealtime 看 Redis/MySQL/TDengine。
4. 先看日志,再决定是否重启。
```bash
journalctl -u lingniu-go-history-writer.service --since '10 minutes ago' --no-pager
journalctl -u lingniu-go-stat-writer.service --since '10 minutes ago' --no-pager
journalctl -u lingniu-go-realtime-api.service --since '10 minutes ago' --no-pager
```
Kafka 是可回放日志。只要 Kafka 还保留消息,恢复 consumer 后 lag 应该能自动追平。不要在 lag 未归零前手工补数。
### NATS pending 持续增长
1. 检查 bridge `/readyz` 和日志。
2. 对比 `vehicle_bridge_kafka_writes_total``vehicle_bridge_nats_acks_total`
3. 如果 Kafka 写入失败,检查 ECS 到 Kafka broker 的网络。
4. 如果 `ack_pending > 0`,优先检查 Kafka 是否监听 `9092`、Kafka 盘是否打满,再看 bridge 日志。
5. 如果 `ack_pending = 0``consumer_pending` 下降,说明 bridge 正在追历史积压;不要反复重启,持续观察下降速度即可。
6. 如果 `consumer_pending` 不下降,才考虑扩容 bridge 或降低 batch/fetch 等配置。
```bash
journalctl -u lingniu-go-nats-kafka-bridge.service --since '10 minutes ago' --no-pager
systemctl restart lingniu-go-nats-kafka-bridge.service
```
生产 stream 必须设置字节上限,避免 NATS JetStream 在 Kafka 或下游故障时吃满根盘:
```bash
grep NATS_STREAM_MAX_BYTES /opt/lingniu-go-native/env/nats-fast-writer.env
grep NATS_STREAM_MAX_BYTES /opt/lingniu-go-native/env/nats-kafka-bridge.env
grep NATS_STREAM_ENSURE_TIMEOUT_SECONDS /opt/lingniu-go-native/env/nats-fast-writer.env
grep NATS_STREAM_ENSURE_TIMEOUT_SECONDS /opt/lingniu-go-native/env/nats-kafka-bridge.env
du -sh /opt/lingniu-nats/data
df -h /
```
当前建议值:`NATS_STREAM_MAX_BYTES=21474836480`,即 `20GiB``NATS_STREAM_ENSURE_TIMEOUT_SECONDS=60`,避免大 stream 元数据更新时被 NATS 客户端默认 5s 超时误杀。Kafka topic 只作为短期缓冲,当前建议保留 `6h`,不要把 Kafka 或 NATS 当长期历史存储;长期历史和 RAW 查询以 TDengine/MySQL 投影为准。
fast-writer 的 `FAST_WRITER_OPERATION_TIMEOUT_MS` 建议为 `1000`。生产默认设置 `FAST_WRITER_TDENGINE_ENABLED=false`,让 fast-writer 只写 Redis 实时投影TDengine RAW/位置由 history-writer 通过 Kafka raw topic 单路写入;如果临时启用 TDengine stage过小的超时会造成 NATS 消息反复重投和重复写压力。
生产 Redis 当前态只由 `nats-fast-writer` 写入;`realtime-writer` 应设置 `REALTIME_ROLE=writer``REALTIME_REDIS_PROJECTOR_ENABLED=false`,继续从 Kafka raw 写 MySQL snapshot/location`realtime-api` 应设置 `REALTIME_ROLE=api`,只提供 Redis/MySQL/TDengine 查询 API。若 `capacity-check` 提示 `realtime redis projector enabled`,应先检查 `/opt/lingniu-go-native/env/realtime-writer.env`
stat-writer 默认 `STATS_WORKERS=3``STATS_PROJECT_INTERVAL_SECONDS=15``vehicle_daily_mileage_source` 每条有效里程样本都会更新,用于保留多源事实和最新总里程;`vehicle_daily_mileage` 是对外查询结果层,会按来源优先级/质量从 source 表投影,投影被节流以降低高频帧下的 MySQL 写放大。需要故障回放或强一致核验时可临时设为 `0` 恢复每样本投影,处理完成后应调回正常值。
`vehicle_data_source.latest_seen_at` 表示平台来源最后一次被本系统接收到的时间stat-writer 优先使用 `received_at_ms` 更新,且 SQL 层禁止该字段被补发、乱序或设备时间异常的帧回退。里程统计日期仍按设备事件时间计算,只有事件时间超出接收时间 `10min` 以上时才回退到接收时间。
`vehicle_realtime_snapshot.access_*` 是车辆/协议级接入证据,由 realtime writer 在快照同一条原子 upsert 中维护。`access_latest_received_at` 只按更大的 `received_at_ms` 且不同 event ID 推进,同时把旧值写入 `access_previous_received_at`,计算 `access_report_interval_ms` 并增加 `access_sample_count`;设备事件时间乱序不影响此路径。迁移 `008` 将存量行标为 `snapshot_backfill`,该时间只是上线基线,不得对外描述为历史首次接入。验证上线时至少观察同一 VIN 的样本数递增、间隔非负、首次时间不变,并确认重复/补发帧没有回退最新接收时间。
### Raw 有数据但实时查不到
1. 查 realtime writer Kafka lag。
2. 查 realtime update 计数。
3. 查 realtime API readiness。
4. 查最新 snapshot/location API。
```bash
curl -fsS http://127.0.0.1:20216/readyz
curl -fsS http://127.0.0.1:20200/readyz
curl -fsS 'http://127.0.0.1:20200/api/realtime/locations?limit=1'
curl -fsS 'http://127.0.0.1:20200/api/realtime/snapshots?limit=1'
```
### 更新 JT808 平台映射
多平台 808 车牌/手机号映射统一进入 `vehicle_identifier`,不要再直接把不同平台的手机号硬塞进 `vehicle_identity_binding` 的唯一列。导入器会先读取映射文件,再用旧 `vehicle_identity_binding` 的车牌/手机号反查 VIN只写入可确定 VIN 的记录;冲突和未匹配记录留在 JSON 报告里人工处理。
```bash
release=$(basename "$(readlink -f /opt/lingniu-go-native/current)")
input=/opt/lingniu-go-native/imports/${release}/jt808_mapping
# 先 dry-run确认 scan.sources、resolved / unresolved / conflicts。
/opt/lingniu-go-native/current/identity-import \
-input "${input}" \
-ensure-schema \
-unresolved-out "/tmp/${release}-identity-unresolved.csv" \
-conflicts-out "/tmp/${release}-identity-conflicts.csv" \
-timeout 5m > /tmp/${release}-identity-dryrun.json
jq '.scan.sources' /tmp/${release}-identity-dryrun.json
jq '.scan.unsupported_items // []' /tmp/${release}-identity-dryrun.json
jq '.source_results' /tmp/${release}-identity-dryrun.json
# dry-run 无冲突后再 apply。
/opt/lingniu-go-native/current/identity-import \
-input "${input}" \
-apply \
-sync-data-sources \
-timeout 5m > /tmp/${release}-identity-apply.json
# 只同步来源表时可以不传 input不带 -apply 只输出候选/跳过数量,不写库。
/opt/lingniu-go-native/current/identity-import \
-sync-data-sources \
-timeout 1m > /tmp/${release}-source-sync-dryrun.json
/opt/lingniu-go-native/current/identity-import \
-sync-data-sources \
-apply \
-timeout 1m > /tmp/${release}-source-sync-apply.json
```
导入后检查:
- `scan.sources` 是否覆盖全部平台目录;重点看每个 `source_code``phone_records``plate_records``skipped`,如果某个平台记录数异常低,先处理源文件表头/列位置,不要直接 apply。
- `scan.unsupported_items` 必须为空;如果出现 `.xls``.xlsb`,先转成 `.xlsx` 再导入,避免某个平台文件被跳过后造成 VIN 映射缺口。`identity-import -apply` 会在连接 MySQL/建表前拒绝带 unsupported 文件的目录。
- `source_results` 是否逐平台呈现合理的 `resolved/unresolved/conflicts`;如果某个平台 `unresolved` 很高,先补 `vehicle_identity_binding` 的 VIN/车牌基础事实;如果 `conflicts` 不为 `0`,先人工确认同一平台下手机号/车牌是否重复指向不同车辆。
- 同一平台同一手机号/车牌重复出现但不冲突时,导入器会合并非空字段,优先保留可用于 VIN 解析的车牌;如果同一标识对应多个不同车牌,会进入 conflict不会静默覆盖。
- `identity-import -apply` 会跳过 unresolved/conflict 项,只写入可确定 VIN 且无冲突的标识;实际写入包在单个 MySQL 事务里,任一写入错误都会回滚本次导入。
- `vehicle_identifier` 总数和各 `source_code` 分布是否符合文件规模。
- `vehicle_data_source.source_code/platform_name/source_kind` 会从 `jt808_registration.phone/source_ip``vehicle_identifier` 推断;同一来源 IP 推断出多个平台时会跳过,且不会覆盖人工维护的平台名。
- 来源维护 API 支持 `sourceCodeMissing=true` 快速列出未绑定平台编码的来源,例如:`/api/stats/data-sources?protocol=JT808&sourceCodeMissing=true&includeTotal=true`
- 来源类型使用 `source_kind` 维护:`PLATFORM` 表示稳定平台源,`DIRECT` 表示车辆直连或动态 IP`UNKNOWN` 表示未分类。每日里程最终投影在同等质量下按 `PLATFORM -> UNKNOWN -> DIRECT` 选择来源,然后再比较 `trust_priority`、样本数和最新时间。
- 每日里程 API 会通过 `source_id` 关联 `vehicle_data_source` 返回选中来源证据,例如:`/api/stats/daily-metrics?protocol=JT808&dateFrom=2026-07-12&dateTo=2026-07-12&limit=1`。排查异常里程时同时核对 `source_id``source_ip``platform_name``source_code``source_kind``latest_total_mileage_km`,不要只看 `daily_mileage_km`
- 每日里程候选来源 API 用于审计多源选举,例如:`/api/stats/daily-metrics/sources?protocol=JT808&dateFrom=2026-07-12&dateTo=2026-07-12&selected=true&limit=1`。它返回 `source_key``phone``sample_count``first_total_mileage_km``latest_total_mileage_km``quality_status``quality_reason``is_selected`,用于确认每个来源独立算出的里程以及最终是否被选中。`first_event_time` 可能是跨日前推得到的历史基线时间,排查时应结合 `quality_reason` 理解。
- 每日里程诊断 API 用于排查“实时在线但统计缺失”,例如:`/api/stats/daily-metrics/diagnostics?protocol=YUTONG_MQTT&date=2026-07-12&diagnosis=NO_SOURCE_SAMPLE&includeTotal=true`。它会从 `vehicle_realtime_snapshot/location` 找当天活跃车辆,再关联最终表和候选来源表输出 `diagnosis``OK` 表示最终日里程已存在;`MISSING_DAILY` 表示候选来源已有样本但最终投影缺失;`NO_SOURCE_SAMPLE` 表示当天确实收到总里程字段但 stat-writer 没抽到候选样本;`NO_TOTAL_MILEAGE` 表示当天实时活跃但没有收到总里程字段。宇通 MQTT 可能稀疏上报,`vehicle_realtime_location.total_mileage_km` 会保留旧值,判断当天是否真实上报要看 `total_mileage_event_time`
- `NO_TOTAL_MILEAGE``reason` 会进一步说明源头问题:`realtime_total_mileage_missing` 表示当前态没有总里程;`realtime_total_mileage_non_positive` 表示源头总里程为 0 或负数;`realtime_total_mileage_time_missing` 表示当前态有总里程旧值但缺少该字段的上报时间证据;`realtime_total_mileage_not_reported_on_stat_date` 表示当前态保留了旧总里程,但指定业务日期没有重新上报总里程字段。
- `vehicle_realtime_snapshot.parsed_json` 是稀疏帧合并后的当前态,字段存在只证明历史上曾上报,不能证明当前帧或统计日上报。字段诊断返回 `mapped_protocol_field_without_fresh_evidence` 时应按 `raw_frame_query_path` 核对原始帧;只有 `vehicle_realtime_location.total_mileage_event_time` 落在统计日内,才把标准总里程视为新鲜证据。
- 每日里程诊断汇总 API 用于大屏和告警入口,例如:`/api/stats/daily-metrics/diagnostics/summary?date=2026-07-12`。它按协议返回 `active_count``ok_count``missing_daily_count``no_source_sample_count``no_total_mileage_count``actionable_issue_count`;健康状态下应优先关注 `actionable_issue_count=0`,异常时再下钻到诊断明细接口。
- 每日里程诊断原因汇总 API 用于告警聚合和运营看板,例如:`/api/stats/daily-metrics/diagnostics/reasons?date=2026-07-12`。它按 `protocol + diagnosis + reason` 返回车辆数;需要只看某类问题时可传 `diagnosis=NO_TOTAL_MILEAGE`,比拉取逐车明细更适合高频轮询。
- `stats-backfill` 用于从 TDengine RAW 证据层补算 MySQL 每日里程。ECS 上直接执行会默认加载 `/opt/lingniu-go-native/env/base.env``/opt/lingniu-go-native/env/stat-writer.env`;默认 `BACKFILL_DRY_RUN=true`,不会写库。补算采用 `BACKFILL_METHOD=last_diff`对每个协议、VIN、来源分别取当日最新累计总里程和同一来源在目标日前最近一次累计总里程`当日最新 - 最近历史总里程` 计算。补算窗口首日会聚合目标日前的历史并取 `LAST(event_time)`因此前一日缺失时会继续向更早日期寻找历史完全为空时才使用当天第一条作基线。JT808 历史补算结束后会按 `source_code` 归并平台来源键,保证历史与实时统计使用同一来源。正式写入前先 dry-run例如
```bash
BACKFILL_PROTOCOLS=YUTONG_MQTT \
BACKFILL_DATE_FROM=2026-07-12 \
BACKFILL_DATE_TO=2026-07-12 \
/opt/lingniu-go-native/current/stats-backfill
BACKFILL_DRY_RUN=false \
BACKFILL_PROTOCOLS=GB32960,JT808,YUTONG_MQTT \
BACKFILL_DATE_FROM=2026-07-12 \
BACKFILL_DATE_TO=2026-07-12 \
/opt/lingniu-go-native/current/stats-backfill
```
生产定时补算使用 systemd timer默认每天 01:30 补算昨天往前 3 天,覆盖上游晚到或断传后恢复的总里程字段:
```bash
cd go/vehicle-gateway
scripts/install-stats-backfill-timer.sh \
--host root@115.29.187.205 \
--days-back 1 \
--window-days 3 \
--on-calendar '*-*-* 01:30:00'
systemctl list-timers lingniu-go-stats-backfill.timer --no-pager
systemctl cat lingniu-go-stats-backfill.service lingniu-go-stats-backfill.timer --no-pager
```
如果只想验证定时任务配置不写库,可以加 `--dry-run` 安装;正式生产不加 `--dry-run`,服务内会设置 `BACKFILL_DRY_RUN=false`。timer 设置 `Persistent=false`,避免首次安装时立刻执行当天已错过的计划;即使 ECS 短暂停机,下一次 3 天窗口也会补到遗漏日期。
- 诊断 API 判断“指定日期活跃”时会同时看 `event_time``received_at``updated_at`,任一时间落在业务日期内即纳入;这样设备时间漂移或未来时间不会遮蔽服务端当天实际收到的数据。
- 每日里程质量规则会把 1km 内的小幅负漂移按 0km 处理,`quality_reason=negative_jitter_clamped`。优先用前一自然日同来源末值前一日缺失时继续向前查找最近历史日末值并按“2500km × 两个累计里程端点相隔自然日数”放大物理合理性上限,但仍把完整差值记在当前统计日,不平均分摊或伪造中间日期。该上限覆盖车辆以约 100km/h 连续运行 24 小时的极端场景,同时继续拒绝约 4000km/日的异常跳变。超过窗口上限的正向突增或超过 1km 的倒退是 `INVALID_DELTA`,不会进入最终 `vehicle_daily_mileage`;历史完全为空时使用当天第一条。如果设备事件时间比接收时间未来超过 10 分钟业务层会统一使用接收时间TDengine RAW 帧仍保留设备原始时间作为证据。
- 来源诊断 API 支持按来源 IP 汇总注册手机号和 `vehicle_identifier` 匹配情况,例如:`/api/stats/data-sources/diagnostics?protocol=JT808&sourceCodeMissing=true&includeTotal=true``reason=no_identifier_match` 表示该来源下手机号还没有进入 `vehicle_identifier``reason=ambiguous_source_code` 表示同一来源 IP 匹配到多个平台编码,需要人工确认;`reason=candidate_available` 表示可以同步或确认候选 `source_code`
- 来源类型建议 API 是只读辅助,例如:`/api/stats/data-sources/kind-suggestions?protocol=JT808&sourceKind=UNKNOWN&includeTotal=true`。它会输出 `suggested_source_kind``suggestion_confidence``suggestion_reason`,用于人工确认后再 PATCH `source_kind`;不要把建议结果无审核地批量写回。
- `source_code` 用于程序稳定识别来源,`platform_name` 用于展示和人工修正;如果两者看起来不一致,优先核对该来源 IP 下的 `jt808_registration.phone` 是否来自同一平台,再决定是否人工修正平台名。
- Gateway 默认启用 `IDENTITY_SOURCE_CODE_LOOKUP_ENABLED=true`。808 VIN 解析会先用 `source_endpoint -> vehicle_data_source.source_code``vehicle_identifier` 精确匹配,精确未命中时再回退到全局 `vehicle_identifier`、旧 `vehicle_identity_binding``jt808_registration`,因此多平台同一手机号/车牌映射不同 VIN 时,应优先维护正确的 `vehicle_data_source.source_code`
- `jt808_registration``vin = 'unknown'` 的手机号是否仍能在 `vehicle_identifier` 中匹配;如果能匹配,等待下一帧 808 位置或注册/鉴权帧触发自动更新。
- Gateway `vehicle_gateway_identity_total{protocol="JT808",status="resolved"}` 应持续增长。
Gateway 默认启用 `IDENTITY_SNAPSHOT_ONLY_ENABLED=true`,启动时并每隔 `IDENTITY_SNAPSHOT_REFRESH_INTERVAL_SECONDS=60` 秒把 `vehicle_data_source``vehicle_identifier`、旧 `vehicle_identity_binding``jt808_registration` 原子载入内存。快照包含 JT808 的 VIN、设备、车牌和最近可信鉴权码每帧身份解析及鉴权均不查询 MySQL。注册/鉴权/位置触达由独立 identity-writer 投影写入 `jt808_registration`,人工改库或新鉴权码最多等待一个刷新周期即可生效。快照刷新失败保留上一版,启动时 RDS 不可用也不阻止接入;未知 phone 会暂时 unresolvedRDS 恢复并完成刷新后自动解析。
## 重启顺序
优先重启最小故障层:
1. 只有一个 consumer lag 异常时,先重启对应 writer。
2. 实时投影或 API 异常时,重启 realtime API。
3. bridge pending 或 ack-pending 卡住时,重启 NATS Kafka bridge。
4. 只有入口监听、解析循环或上游连接处理异常时,才重启 gateway。
```bash
systemctl restart lingniu-go-history-writer.service
systemctl restart lingniu-go-stat-writer.service
systemctl restart lingniu-go-realtime-api.service
systemctl restart lingniu-go-nats-kafka-bridge.service
systemctl restart lingniu-go-gateway.service
```
排查前不要直接全量重启。全量重启会抹掉时间线证据,也会掩盖问题到底发生在入口、队列、存储还是查询层。
## 存储原则
运行指标保留在 `/metrics`,不要为了服务健康计数新增 MySQL 或 TDengine 业务表。
业务存储保持最小化:
- raw frames 保存完整 parsed JSON用于回放和审计。
- realtime snapshot/location 只保存 API 需要的当前业务字段。
- history 表保存可查询的时间序列核心字段。
- 派生统计应能从 Kafka 或 raw history 重新计算。
只有当产品查询明确需要、并且指标定义稳定时,才新增持久化聚合。

View File

@@ -0,0 +1,88 @@
# 车辆定位源仲裁运维说明
## 目标
同一车辆可能同时接入 GB32960、宇通 MQTT、多个 JT808 终端。全局监控不能简单选择最后上报的数据,否则不同设备的定位误差会让车辆在地图上来回跳动。
当前生产规则如下:
- 跨协议默认优先级GB32960、YUTONG_MQTT、JT808候选数据在 10 分钟内有效。
- JT808 按终端保存独立快照,不再由同 VIN 的多个终端互相覆盖。
- JT808 默认优先级:直连终端 20、平台转发 30、未知来源 40数值越小越优先。
- 当前终端有 5 分的黏性权重,新终端连续 3 个有效点后才允许接管。
- 两个新鲜 JT808 来源相差超过 200 米时标记冲突并保持当前来源,避免自动跳点。
- 单终端相邻点超过 `max(500 米, 间隔秒数 × 70 米)` 时拒绝该点。
- 前端仅在定位来源未变化时执行平滑移动;来源切换时直接重置动画,避免绘制虚假轨迹。
## 数据表
- `vehicle_realtime_location_source`JT808 终端级实时位置和质量状态。
- `vehicle_location_source_policy`:车辆、协议、终端级启停和优先级。
- `vehicle_realtime_location`:仲裁后的协议级位置;`source_key` 表示当前终端,`location_conflict``location_conflict_distance_m` 表示冲突状态。
查看某辆车的 JT808 候选来源:
```sql
SELECT source_key, phone, device_id, source_kind, source_endpoint,
quality_status, quality_reason, consecutive_good_samples,
latitude, longitude, received_at
FROM vehicle_realtime_location_source
WHERE vin = ? AND protocol = 'JT808'
ORDER BY received_at DESC;
```
## 指定权威终端
先从候选表复制完整 `source_key`。将权威终端设置为较小优先级;建议保留其他终端作为备用,而不是直接禁用。
```sql
INSERT INTO vehicle_location_source_policy
(vin, protocol, source_key, enabled, priority, remark)
VALUES
(?, 'JT808', ?, 1, 1, '人工确认的权威定位终端')
ON DUPLICATE KEY UPDATE
enabled = VALUES(enabled),
priority = VALUES(priority),
remark = VALUES(remark);
```
禁用确认异常的终端:
```sql
INSERT INTO vehicle_location_source_policy
(vin, protocol, source_key, enabled, priority, remark)
VALUES
(?, 'JT808', ?, 0, 100, '异常终端,停止参与定位仲裁')
ON DUPLICATE KEY UPDATE
enabled = VALUES(enabled),
remark = VALUES(remark);
```
恢复默认策略时删除对应配置:
```sql
DELETE FROM vehicle_location_source_policy
WHERE vin = ? AND protocol = 'JT808' AND source_key = ?;
```
配置在下一条该车辆 JT808 定位消息到达时生效,不需要重启服务。
## 监控与排查
主要观察指标:
- `vehicle_realtime_store_updates_total{protocol="JT808",status="ok",store="mysql"}`
- `vehicle_realtime_store_update_duration_ms_histogram{protocol="JT808",status="ok",store="mysql"}`
- `vehicle_process_heap_alloc_bytes`
冲突车辆查询:
```sql
SELECT vin, source_key, location_conflict, location_conflict_distance_m,
latitude, longitude, received_at
FROM vehicle_realtime_location
WHERE protocol = 'JT808' AND location_conflict = 1
ORDER BY location_conflict_distance_m DESC;
```
冲突期间保留当前来源是预期行为。人工确认哪个终端可信后,再写入策略表;不要通过删除实时快照或修改业务资产数据解决定位冲突。

View File

@@ -1,600 +0,0 @@
# GB32960 Body Parser — Per-Block Exception Isolation Plan
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
**Goal:** `Gb32960BodyParser` 主循环对单信息块的解析异常实现**单块隔离**:一块解析失败不再整帧放弃,改为把失败块兜成 `InfoBlock.Raw` 并继续解析后续信息块。
**Architecture:** 在主循环每次 `parser.parse(body)` 外面套 try/catch仅捕获 `DecodeException` / `BufferUnderflowException` / `IndexOutOfBoundsException`,放行 RuntimeException 以免掩盖 parser bug。失败时
-`fixedLen ≥ 0` 且剩余字节 ≥ fixedLen → 回滚 reader → 按 fixedLen 截取 Raw → position 前进 fixedLen → `continue` 循环
- 否则(变长 parser 或剩余不足)→ 剩余字节全部兜成 Raw → `break` 循环(变长块无法安全找下一块边界)
通过 `Gb32960Properties.Parse.lenientBlockFailure`(默认 true控制开关可以回退到严格模式。**不改 `InfoBlock.Raw` record 结构**(避免 downstream 兼容风险——`Gb32960EventMapper``findBlock(Class)` 类型查找,新增 Raw 不影响其行为)。
**Tech Stack:** Java 25, JUnit 5, AssertJ, Spring Boot ConfigurationProperties, Maven。仅改动 `protocol-gb32960` 模块。
---
## File Structure
- **Modify** `protocol-gb32960/src/main/java/com/lingniu/ingest/protocol/gb32960/config/Gb32960Properties.java` — 新增 `Parse` 内嵌类 + `parse` 字段/getter/setter
- **Modify** `protocol-gb32960/src/main/java/com/lingniu/ingest/protocol/gb32960/codec/Gb32960BodyParser.java` — 主循环 try/catch + recovery 逻辑 + `lenientBlockFailure` 字段
- **Modify** `protocol-gb32960/src/main/java/com/lingniu/ingest/protocol/gb32960/config/Gb32960AutoConfiguration.java` — 把配置穿给 BodyParser
- **Create** `protocol-gb32960/src/test/java/com/lingniu/ingest/protocol/gb32960/codec/Gb32960BodyParserIsolationTest.java` — 3 个隔离场景 + 1 个严格模式回退场景
- **Modify** `bootstrap-all/src/main/resources/application.yml` — 添加 `parse:` 注释段示例
- **Modify** `CHANGELOG.md` — 追加本次变更条目
---
## Task 1: 基线验证 —— 现有测试全绿
**Files:**
- [ ] **Step 1.1: 跑 protocol-gb32960 模块测试**
Run: `mvn -pl protocol-gb32960 test -q`
Expected: BUILD SUCCESS所有现有测试通过`Gb32960DecoderGoldenTest``Gb32960FullBlocksTest``Gb32960DecoderTest``GuangdongFcEndToEndTest`、parser/* 各 Block Parser 单测、profile/* 选择器单测)。
如果基线红了,**先停下来修复基线**再进入 Task 2。
---
## Task 2: 添加 Parse 配置子节点
**Files:**
- Modify: `protocol-gb32960/src/main/java/com/lingniu/ingest/protocol/gb32960/config/Gb32960Properties.java`
- [ ] **Step 2.1: 在 `Gb32960Properties` 类体中(在 `Auth` 字段后、`Tls` 字段前)新增 `parse` 字段**
位置参考:现在 L24 `private Auth auth = new Auth();` 下面一行,在 L27 `private Tls tls = new Tls();` 之前插入:
```java
/**
* 报文解析行为配置。
*
* <p>默认 {@code lenientBlockFailure=true}:单个信息块解析异常时兜成
* {@link com.lingniu.ingest.protocol.gb32960.model.InfoBlock.Raw} 继续解析,
* 不再整帧放弃。关闭后恢复旧行为(任意异常直接抛 {@code DecodeException}
* 放弃整帧),仅在灰度回滚时使用。
*/
private Parse parse = new Parse();
```
- [ ] **Step 2.2: 在类底部(`Tls` 静态类之后、`VendorExtension` 静态类之前)添加 `Parse` 静态内嵌类**
```java
/**
* 报文解析容错策略。
*/
public static class Parse {
/**
* 单块解析异常时是否兜底为 {@link com.lingniu.ingest.protocol.gb32960.model.InfoBlock.Raw}
* 继续解析。默认开启。
*
* <ul>
* <li>{@code true}默认parser 抛 DecodeException / BufferUnderflowException /
* IndexOutOfBoundsException 时,固定长度块按 fixedLen 截取 Raw 后 continue
* 变长块或剩余字节不足时,剩余全部兜成 Raw 后 break 循环。
* <li>{@code false}:保留旧行为,任意异常直接抛出 DecodeException 放弃整帧。
* </ul>
*/
private boolean lenientBlockFailure = true;
public boolean isLenientBlockFailure() { return lenientBlockFailure; }
public void setLenientBlockFailure(boolean lenientBlockFailure) {
this.lenientBlockFailure = lenientBlockFailure;
}
}
```
- [ ] **Step 2.3: 在属性 getter/setter 区域(现有 `getTls` 之后)补充 `getParse` / `setParse`**
```java
public Parse getParse() { return parse; }
public void setParse(Parse parse) { this.parse = parse; }
```
- [ ] **Step 2.4: 编译验证**
Run: `mvn -pl protocol-gb32960 compile -q`
Expected: BUILD SUCCESS。
- [ ] **Step 2.5: 提交**
```bash
git add protocol-gb32960/src/main/java/com/lingniu/ingest/protocol/gb32960/config/Gb32960Properties.java
git commit -m "$(cat <<'EOF'
config(gb32960): add parse.lenientBlockFailure toggle
新增 lingniu.ingest.gb32960.parse.lenientBlockFailure 配置开关(默认 true
为 Gb32960BodyParser 的单块异常隔离行为做回退开关。实现在后续 commit。
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
EOF
)"
```
---
## Task 3: RED —— 写三个失败测试覆盖隔离场景
**Files:**
- Create: `protocol-gb32960/src/test/java/com/lingniu/ingest/protocol/gb32960/codec/Gb32960BodyParserIsolationTest.java`
- [ ] **Step 3.1: 创建测试文件**
完整内容:
```java
package com.lingniu.ingest.protocol.gb32960.codec;
import com.lingniu.ingest.api.spi.DecodeException;
import com.lingniu.ingest.protocol.gb32960.codec.parser.v2016.AlarmV2016BlockParser;
import com.lingniu.ingest.protocol.gb32960.codec.parser.v2016.PositionV2016BlockParser;
import com.lingniu.ingest.protocol.gb32960.codec.parser.v2016.VehicleV2016BlockParser;
import com.lingniu.ingest.protocol.gb32960.model.InfoBlock;
import com.lingniu.ingest.protocol.gb32960.model.InfoBlockType;
import com.lingniu.ingest.protocol.gb32960.model.ProtocolVersion;
import org.junit.jupiter.api.Test;
import java.io.ByteArrayOutputStream;
import java.nio.ByteBuffer;
import java.util.List;
import static org.assertj.core.api.Assertions.assertThat;
import static org.assertj.core.api.Assertions.assertThatThrownBy;
/**
* 验证 {@link Gb32960BodyParser} 单块异常隔离行为。
*
* <p>三个隔离场景:
* <ol>
* <li>固定长度块 parser 抛异常 → 失败块兜 Raw按 fixedLen 截取)+ 后续块继续解析
* <li>固定长度块剩余字节不足(截断帧) → 兜 Raw(剩余) + break 循环
* <li>变长块 parser 抛异常 → 兜 Raw(从失败块起剩余全部) + break 循环
* </ol>
*
* <p>外加一个严格模式回退测试:{@code lenientBlockFailure=false} 时任意异常应抛 DecodeException。
*/
class Gb32960BodyParserIsolationTest {
/** 构造一个"看起来合法但中途 parser 失败"用来注入失败的 Position parser。 */
private static final InfoBlockParser EXPLODING_FIXED_LEN_POSITION = new InfoBlockParser() {
@Override public ProtocolVersion version() { return ProtocolVersion.V2016; }
@Override public int typeCode() { return 0x05; }
@Override public int fixedLength() { return 9; }
@Override public InfoBlock parse(ByteBuffer buffer) {
// 模拟 parser 读了几字节后发现不合法就抛
buffer.get(); buffer.get();
throw new DecodeException("simulated parser failure in Position");
}
};
/** 变长块fixedLength=-1parse 时消费若干字节后抛。模拟 Alarm/Voltage 类列表读越界。 */
private static final InfoBlockParser EXPLODING_VAR_LEN_ALARM = new InfoBlockParser() {
@Override public ProtocolVersion version() { return ProtocolVersion.V2016; }
@Override public int typeCode() { return 0x07; }
@Override public int fixedLength() { return -1; }
@Override public InfoBlock parse(ByteBuffer buffer) {
// 假装读了个长度字段就爆
buffer.get();
throw new DecodeException("simulated parser failure in Alarm list-length read");
}
};
@Test
void fixedLengthBlockFailure_isIsolated_subsequentBlocksStillParsed() {
// 帧布局Vehicle(0x01, 20B) + Position(0x05, 9B 但 parser 爆) + Engine(0x04) 不构造,
// 简化为 Vehicle + FailingPosition + Vehicle 再次,验证"Position 之后还能继续"。
InfoBlockParserRegistry registry = new InfoBlockParserRegistry(List.of(
new VehicleV2016BlockParser(),
EXPLODING_FIXED_LEN_POSITION));
Gb32960BodyParser parser = new Gb32960BodyParser(registry);
ByteArrayOutputStream os = new ByteArrayOutputStream();
writeValidVehicle(os); // 1+20 = 21B
writePositionTypeAnd9ByteBody(os); // 1+9 = 10B (parser 会爆)
writeValidVehicle(os); // 1+20 = 21B
ByteBuffer body = ByteBuffer.wrap(os.toByteArray());
var result = parser.parse(ProtocolVersion.V2016, body);
assertThat(result.blocks()).hasSize(3);
assertThat(result.blocks().get(0)).isInstanceOf(InfoBlock.Gb32960V2016.Vehicle.class);
assertThat(result.blocks().get(1)).isInstanceOfSatisfying(InfoBlock.Raw.class, raw -> {
assertThat(raw.typeCode()).isEqualTo(0x05);
assertThat(raw.type()).isEqualTo(InfoBlockType.RAW);
assertThat(raw.bytes()).hasSize(9); // 按 fixedLen 截取
});
assertThat(result.blocks().get(2)).isInstanceOf(InfoBlock.Gb32960V2016.Vehicle.class);
assertThat(body.hasRemaining()).isFalse();
}
@Test
void truncatedFixedLengthBlock_isWrappedAsRaw_loopTerminates() {
// Vehicle(21B) + Position(type 0x05)但 body 只给 3B —— 不足 9B
InfoBlockParserRegistry registry = new InfoBlockParserRegistry(List.of(
new VehicleV2016BlockParser(),
new PositionV2016BlockParser()));
Gb32960BodyParser parser = new Gb32960BodyParser(registry);
ByteArrayOutputStream os = new ByteArrayOutputStream();
writeValidVehicle(os);
os.write(0x05); // Position typeCode
os.write(0); os.write(0); os.write(0); // 只 3B远远不够 9B
ByteBuffer body = ByteBuffer.wrap(os.toByteArray());
var result = parser.parse(ProtocolVersion.V2016, body);
assertThat(result.blocks()).hasSize(2);
assertThat(result.blocks().get(0)).isInstanceOf(InfoBlock.Gb32960V2016.Vehicle.class);
assertThat(result.blocks().get(1)).isInstanceOfSatisfying(InfoBlock.Raw.class, raw -> {
assertThat(raw.typeCode()).isEqualTo(0x05);
assertThat(raw.bytes()).hasSize(3); // 剩余字节全兜成 Raw
});
}
@Test
void variableLengthBlockFailure_swallowsRemainderAsRaw_loopBreaks() {
// Vehicle(21B) + FailingAlarm(0x07)后面还有一个 Vehicle但因 Alarm 变长无法安全跳过,
// 整段 Alarm 起的剩余字节都被兜成 Raw 后 break。
InfoBlockParserRegistry registry = new InfoBlockParserRegistry(List.of(
new VehicleV2016BlockParser(),
EXPLODING_VAR_LEN_ALARM));
Gb32960BodyParser parser = new Gb32960BodyParser(registry);
ByteArrayOutputStream os = new ByteArrayOutputStream();
writeValidVehicle(os); // 21B
os.write(0x07); // Alarm typeCode
for (int i = 0; i < 10; i++) os.write(0xAA); // 10B 的 alarm bodyparser 爆)
writeValidVehicle(os); // 21B应当不再被解析
ByteBuffer body = ByteBuffer.wrap(os.toByteArray());
var result = parser.parse(ProtocolVersion.V2016, body);
assertThat(result.blocks()).hasSize(2);
assertThat(result.blocks().get(0)).isInstanceOf(InfoBlock.Gb32960V2016.Vehicle.class);
assertThat(result.blocks().get(1)).isInstanceOfSatisfying(InfoBlock.Raw.class, raw -> {
assertThat(raw.typeCode()).isEqualTo(0x07);
// 剩余 10B alarm body + 1+20=21B 后续 Vehicle = 31B
assertThat(raw.bytes()).hasSize(31);
});
}
@Test
void strictMode_throwsOnAnyBlockFailure() {
InfoBlockParserRegistry registry = new InfoBlockParserRegistry(List.of(
new VehicleV2016BlockParser(),
EXPLODING_FIXED_LEN_POSITION));
Gb32960BodyParser parser = new Gb32960BodyParser(registry);
parser.setLenientBlockFailure(false);
ByteArrayOutputStream os = new ByteArrayOutputStream();
writeValidVehicle(os);
writePositionTypeAnd9ByteBody(os);
ByteBuffer body = ByteBuffer.wrap(os.toByteArray());
assertThatThrownBy(() -> parser.parse(ProtocolVersion.V2016, body))
.isInstanceOf(DecodeException.class);
}
// ------------------------------------------------------------------------
// helpers
// ------------------------------------------------------------------------
/** 写一个完整的合法 V2016 Vehicle 块typeCode + 20B body。 */
private static void writeValidVehicle(ByteArrayOutputStream os) {
os.write(0x01); // typeCode Vehicle
os.write(0x01); // vehicleState=1
os.write(0x01); // chargingState=1
os.write(0x01); // runningMode=1
os.write(0); os.write(0); // speed=0
os.write(0); os.write(0); os.write(0); os.write(0); // mileage=0
os.write(0); os.write(0); // totalVoltage=0
os.write(0); os.write(0); // totalCurrent=0offset 1000原值0
os.write(50); // soc=50%
os.write(0x01); // dcdc
os.write(0); // gear
os.write(0); os.write(0); // insulation
os.write(0); // accelerator
os.write(0); // brake
}
/** 写 Position typeCode(0x05) + 9 字节 body内容不重要parser 替换成 EXPLODING 的)。 */
private static void writePositionTypeAnd9ByteBody(ByteArrayOutputStream os) {
os.write(0x05);
for (int i = 0; i < 9; i++) os.write(0);
}
}
```
- [ ] **Step 3.2: 跑测试验证 RED**
Run: `mvn -pl protocol-gb32960 test -Dtest=Gb32960BodyParserIsolationTest -q`
Expected:
- `fixedLengthBlockFailure_isIsolated_subsequentBlocksStillParsed` FAIL当前会抛 DecodeException
- `truncatedFixedLengthBlock_isWrappedAsRaw_loopTerminates` FAIL当前 L132 会抛 DecodeException
- `variableLengthBlockFailure_swallowsRemainderAsRaw_loopBreaks` FAIL异常冒出整帧失败
- `strictMode_throwsOnAnyBlockFailure` FAIL没有 `setLenientBlockFailure` 方法,编译就红)
**验证点**编译失败strictMode test 里 `setLenientBlockFailure` 尚未存在)是**预期** RED 信号——下一步实现。
---
## Task 4: GREEN —— 实现单块异常隔离
**Files:**
- Modify: `protocol-gb32960/src/main/java/com/lingniu/ingest/protocol/gb32960/codec/Gb32960BodyParser.java`
- [ ] **Step 4.1: 在类字段区新增 `lenientBlockFailure` 字段 + setter**
`private final VendorExtensionSelector selector;`(现 L54下面插入
```java
/**
* 单块解析异常时是否兜底为 Raw 后继续。由
* {@link com.lingniu.ingest.protocol.gb32960.config.Gb32960Properties.Parse#isLenientBlockFailure()}
* 注入;默认 true。
*/
private boolean lenientBlockFailure = true;
public void setLenientBlockFailure(boolean lenientBlockFailure) {
this.lenientBlockFailure = lenientBlockFailure;
}
```
- [ ] **Step 4.2: 改造主循环:替换 L130~L150 的整个 "parse + 长度校验" 段**
**完整替换块**:以下代码替换原来从 `int fixedLen = parser.fixedLength();`L130开始到 `blocks.add(block);`L150结束的整块。
```java
int fixedLen = parser.fixedLength();
int posBefore = body.position();
// 旧行为fixedLen 预检不足 → 直接抛。新行为lenient 模式):走统一恢复路径。
if (fixedLen >= 0 && body.remaining() < fixedLen) {
if (!lenientBlockFailure) {
throw new DecodeException(
"info block 0x" + Integer.toHexString(typeCode)
+ " needs " + fixedLen + " bytes but got " + body.remaining());
}
// 剩余字节数不足 fixedLen —— 无法按块截取,整尾兜 Raw + break
int remaining = body.remaining();
byte[] tail = new byte[remaining];
body.get(tail);
log.warn("[gb32960] truncated block typeCode=0x{} declaredFixedLen={} remaining={} — wrapping tail as Raw",
Integer.toHexString(typeCode), fixedLen, remaining);
blocks.add(new InfoBlock.Raw(typeCode, InfoBlockType.RAW, tail));
break;
}
InfoBlock block;
try {
block = parser.parse(body);
} catch (DecodeException | java.nio.BufferUnderflowException | IndexOutOfBoundsException e) {
if (!lenientBlockFailure) {
if (e instanceof DecodeException de) throw de;
throw new DecodeException(
"parser " + parser.getClass().getSimpleName() + " failed: " + e.getMessage(), e);
}
// 恢复路径:先回滚 position 到 parse 入口
body.position(posBefore);
if (fixedLen >= 0) {
// 固定长度块:按 fixedLen 截取 → 兜 Raw → continue 循环
byte[] corrupt = new byte[fixedLen];
body.get(corrupt);
log.warn("[gb32960] block parse failed typeCode=0x{} parser={} fixedLen={} — isolated as Raw, continuing",
Integer.toHexString(typeCode), parser.getClass().getSimpleName(), fixedLen, e);
blocks.add(new InfoBlock.Raw(typeCode, InfoBlockType.RAW, corrupt));
continue;
} else {
// 变长块:无法安全找下一块边界 → 剩余整段兜 Raw + break
int remaining = body.remaining();
byte[] tail = new byte[remaining];
body.get(tail);
log.warn("[gb32960] variable-length block parse failed typeCode=0x{} parser={} remaining={} — wrapping remainder as Raw, stopping loop",
Integer.toHexString(typeCode), parser.getClass().getSimpleName(), remaining, e);
blocks.add(new InfoBlock.Raw(typeCode, InfoBlockType.RAW, tail));
break;
}
}
int consumed = body.position() - posBefore;
if (fixedLen >= 0 && consumed != fixedLen) {
// 这是 parser 契约违反(声明 fixedLen=X 但实际读了 Y保持抛异常以暴露 parser bug。
throw new DecodeException(
"parser " + parser.getClass().getSimpleName()
+ " consumed " + consumed
+ " bytes but declared " + fixedLen);
}
if (log.isDebugEnabled()) {
log.debug("[gb32960] block typeCode=0x{} parser={} pos {}→{} consumed={} declaredFixed={}",
Integer.toHexString(typeCode), parser.getClass().getSimpleName(),
typeStartPos, body.position(), consumed, fixedLen);
}
blocks.add(block);
```
- [ ] **Step 4.3: 运行隔离测试验证 GREEN**
Run: `mvn -pl protocol-gb32960 test -Dtest=Gb32960BodyParserIsolationTest -q`
Expected: 4 个测试全部 PASS。
- [ ] **Step 4.4: 运行全模块测试检查回归**
Run: `mvn -pl protocol-gb32960 test -q`
Expected: BUILD SUCCESS所有原有测试含 Golden、FullBlocks、GuangdongFcEndToEnd依旧通过。
如果有原有测试红了:**停下来诊断**。最可能的原因是某个现有测试之前靠"抛异常"行为验证错误帧的,需要手动判断这是测试问题还是实现问题。
- [ ] **Step 4.5: 提交**
```bash
git add protocol-gb32960/src/main/java/com/lingniu/ingest/protocol/gb32960/codec/Gb32960BodyParser.java \
protocol-gb32960/src/test/java/com/lingniu/ingest/protocol/gb32960/codec/Gb32960BodyParserIsolationTest.java
git commit -m "$(cat <<'EOF'
feat(gb32960): isolate per-block parse failures in body parser
主循环改造:单信息块 parser 抛 DecodeException / BufferUnderflowException /
IndexOutOfBoundsException 时不再放弃整帧。
- 固定长度块fixedLen≥0回滚 reader按 fixedLen 截取字节兜成 InfoBlock.Raw
position 前进 fixedLencontinue 循环继续解析后续块
- 变长块fixedLen=-1或剩余不足 fixedLen剩余字节全部兜成 Raw 后 break
- parser 契约违反consumed != declared fixedLen仍抛异常以暴露 parser bug
- 通过 lingniu.ingest.gb32960.parse.lenientBlockFailure=false 可回退严格模式
- 新增 Gb32960BodyParserIsolationTest 覆盖 3 种失败场景 + 1 严格模式
不改 InfoBlock.Raw record 结构;下游 Gb32960EventMapper 按 findBlock(Class) 类型
查找,新增 Raw 不影响其行为。
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
EOF
)"
```
---
## Task 5: 把配置注入 BodyParserAutoConfiguration 连线)
**Files:**
- Modify: `protocol-gb32960/src/main/java/com/lingniu/ingest/protocol/gb32960/config/Gb32960AutoConfiguration.java`
- [ ] **Step 5.1: 定位 `Gb32960BodyParser` bean 定义**
先查看:`grep -n "Gb32960BodyParser" protocol-gb32960/src/main/java/com/lingniu/ingest/protocol/gb32960/config/Gb32960AutoConfiguration.java`
根据返回的行号,在构造 `Gb32960BodyParser`@Bean 方法里,构造完成后立即调用:
```java
Gb32960BodyParser parser = new Gb32960BodyParser(profileRegistry, selector);
parser.setLenientBlockFailure(properties.getParse().isLenientBlockFailure());
return parser;
```
(如果目前的构造返回一行表达式,改成先赋给 local 变量再设 flag 再 return。
- [ ] **Step 5.2: 编译并运行全模块测试**
Run: `mvn -pl protocol-gb32960 test -q`
Expected: BUILD SUCCESS。
- [ ] **Step 5.3: 提交**
```bash
git add protocol-gb32960/src/main/java/com/lingniu/ingest/protocol/gb32960/config/Gb32960AutoConfiguration.java
git commit -m "$(cat <<'EOF'
config(gb32960): wire parse.lenientBlockFailure into BodyParser bean
Gb32960AutoConfiguration 在创建 Gb32960BodyParser bean 后,将
Gb32960Properties.Parse.lenientBlockFailure 通过 setter 注入,让运行期配置生效。
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
EOF
)"
```
---
## Task 6: bootstrap-all 配置文档化
**Files:**
- Modify: `bootstrap-all/src/main/resources/application.yml`
- [ ] **Step 6.1: 在 `lingniu.ingest.gb32960` 节点的 `vendor-extensions` 段之后、`jt808` 之前,插入 parse 配置示例**
参考位置:现有 application.yml L57 末尾(`vendor-extensions` list 结束)。在 L58 `jt808:` 之前加入:
```yaml
# 报文解析容错。默认启用单块异常隔离:某个信息块解析失败时兜成 Raw 后继续,
# 不再整帧丢弃。仅在需要严格失败语义(灰度回滚或抓虫)时设为 false。
parse:
lenient-block-failure: true
```
- [ ] **Step 6.2: 启动一次 bootstrap-all 的 dry-run 编译**
Run: `mvn -pl bootstrap-all compile -q`
Expected: BUILD SUCCESS只是配置段注释变化不会影响编译
- [ ] **Step 6.3: 提交**
```bash
git add bootstrap-all/src/main/resources/application.yml
git commit -m "$(cat <<'EOF'
config(gb32960): document parse.lenient-block-failure in application.yml
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
EOF
)"
```
---
## Task 7: 全仓库回归测试
**Files:** 无(只是验证)
- [ ] **Step 7.1: 跑完整项目测试**
Run: `mvn test -q`
Expected: BUILD SUCCESS。重点关注
- `protocol-gb32960` 模块全绿Golden、FullBlocks、Isolation、GuangdongFcEndToEnd、Decoder、Mapper、所有 parser 单测、profile
- 其它模块(`ingest-core``sink-kafka` 等)不受影响(无改动应自动通过)
若有红,停下来诊断。
---
## Task 8: 更新 CHANGELOG
**Files:**
- Modify: `CHANGELOG.md`
- [ ] **Step 8.1: 在 CHANGELOG.md 文件顶部(在 `## [0.1.0] — 2026-04-15` 标题之前)新增一个 `## [Unreleased]` 段落;如已存在则追加**
新增段落内容(如已存在 Unreleased 段则在其 `### Added` / `### Changed` 内部追加):
```markdown
## [Unreleased]
### Changed —— GB/T 32960 Body Parser 单块异常隔离
- `Gb32960BodyParser` 主循环对单信息块的 parser 异常不再放弃整帧:
- 固定长度块(`fixedLength ≥ 0`):回滚 reader → 按 fixedLen 截取字节兜成
`InfoBlock.Raw` → 继续解析后续块;
- 变长块或剩余字节不足:剩余字节全部兜成 `Raw` 后终止循环;
- `parser` 契约违反(声明 fixedLen=X 但实际读了 Y仍抛 `DecodeException`
以暴露 parser bug不被静默吞掉
- 只捕获 `DecodeException` / `BufferUnderflowException` / `IndexOutOfBoundsException`
`RuntimeException` 继续向上抛,保留 bug 可见性。
- 新增配置 `lingniu.ingest.gb32960.parse.lenient-block-failure`(默认 `true`
回退严格模式用 `false`
- `InfoBlock.Raw` record 结构**未变**,下游 `Gb32960EventMapper``findBlock(Class)`
类型匹配,新增 Raw 不影响业务事件映射。
- 覆盖测试:`Gb32960BodyParserIsolationTest` —— 3 种隔离场景 + 严格模式回退。
```
- [ ] **Step 8.2: 提交**
```bash
git add CHANGELOG.md
git commit -m "$(cat <<'EOF'
docs(changelog): record gb32960 body parser block isolation
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
EOF
)"
```
---
## Self-Review Checklist
- [x] **Spec coverage**: 原讨论的 A1~A7 全部覆盖——A1 状态机Task 4 Step 4.2、A2 三类错误策略、A3 收窄异常种类(同,只 catch 三种、A4 日志分类warn 带 parser 名/typeCode/原因、A5 配置开关Task 2、A6 合成帧测试Task 3 三场景 + 严格回退、A7 下游兼容(不改 Raw record`findBlock` 类型匹配不受影响plan 中已说明)。
- [x] **Placeholder scan**: 无 TBD / TODO / "实现上面的" 等占位;每个 Step 含实际代码 or 命令。
- [x] **Type consistency**: `lenientBlockFailure` 在 Properties / BodyParser 字段 / setter / 测试中拼写一致。`InfoBlockType.RAW` 枚举值依赖当前存在(已核对 InfoBlock.Raw record
- [x] **测试命名一致**`Gb32960BodyParserIsolationTest` 在 Task 3 创建、Task 4/7 引用一致。
- [x] **未引入新枚举/proto 变更**InfoBlock.Raw 沿用现有 `(int typeCode, InfoBlockType type, byte[] bytes)` 构造。

View File

@@ -1,817 +0,0 @@
> **Superseded:** This 2026-06-23 DuckDB hot-store plan is historical context.
> Use `docs/target-architecture.md` and
> `docs/superpowers/specs/2026-06-29-vehicle-ingest-redesign.md` for the
> current production architecture: TDengine stores hot history, `sink-archive`
> owns raw bytes, and `event-file-store` / DuckDB are no longer part of the
> current build surface.
# GB32960 History DuckDB Hot Store Implementation Plan
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
**Goal:** Replace the current high-write-risk Parquet rewrite history path with a DuckDB hot history store that supports fast `vin + time` queries and full RAW frame replay.
**Architecture:** Add a new `DuckDbHotEventFileStore` implementation behind the existing `EventFileStore` interface, backed by append/upsert DuckDB tables instead of rewriting per-vehicle Parquet files. Keep the current HTTP history API and RAW archive reader intact, then switch `vehicle-history-app` to the hot DuckDB store by configuration.
**Tech Stack:** Java 26, Spring Boot, DuckDB JDBC, JUnit 5, AssertJ, existing Kafka `VehicleEnvelope`, existing `ArchiveStore`.
---
## Scope
This plan implements Phase 1 from `docs/superpowers/specs/2026-06-23-gb32960-production-readiness-design.md`.
Included:
- DuckDB hot event table for `EventFileRecord`.
- Idempotent writes by `event_id`.
- Query by protocol/date/VIN/eventType/eventTime/order/limit.
- Lookup by `rawArchiveUri`.
- Spring configuration to select the hot store.
- Tests that prove history endpoints still replay RAW frames.
- Local verification using the existing `vehicle-history-app`.
Not included in this plan:
- MySQL daily statistics.
- VIN-to-platform local mapping.
- Full telemetry point columnar table.
- Alarm timeline MySQL tables.
- Cold Parquet export.
Those will be separate plans after this foundation lands.
## Files
- Create: `modules/sinks/event-file-store/src/main/java/com/lingniu/ingest/eventfilestore/DuckDbHotEventFileStore.java`
- Create: `modules/sinks/event-file-store/src/test/java/com/lingniu/ingest/eventfilestore/DuckDbHotEventFileStoreTest.java`
- Modify: `modules/sinks/event-file-store/src/main/java/com/lingniu/ingest/eventfilestore/config/EventFileStoreProperties.java`
- Modify: `modules/sinks/event-file-store/src/main/java/com/lingniu/ingest/eventfilestore/config/EventFileStoreAutoConfiguration.java`
- Modify: `modules/sinks/event-file-store/src/test/java/com/lingniu/ingest/eventfilestore/config/EventFileStoreAutoConfigurationTest.java`
- Modify: `modules/apps/vehicle-history-app/src/main/resources/application.yml`
- Modify: `modules/apps/vehicle-history-app/src/test/java/com/lingniu/ingest/historyapp/VehicleHistoryAppDefaultsTest.java`
- Modify: `modules/services/event-history-service/src/test/java/com/lingniu/ingest/eventhistory/Gb32960DecodedFrameServiceTest.java`
## Task 1: Add Hot Store Configuration
**Files:**
- Modify: `modules/sinks/event-file-store/src/main/java/com/lingniu/ingest/eventfilestore/config/EventFileStoreProperties.java`
- Modify: `modules/sinks/event-file-store/src/main/java/com/lingniu/ingest/eventfilestore/config/EventFileStoreAutoConfiguration.java`
- Test: `modules/sinks/event-file-store/src/test/java/com/lingniu/ingest/eventfilestore/config/EventFileStoreAutoConfigurationTest.java`
- [ ] **Step 1: Write the failing auto-configuration test**
Add a test that sets `lingniu.ingest.event-file-store.storage=duckdb-hot` and expects the bean class to be `DuckDbHotEventFileStore`.
```java
@Test
void createsDuckDbHotStoreWhenStorageIsDuckDbHot() {
contextRunner
.withPropertyValues(
"lingniu.ingest.event-file-store.enabled=true",
"lingniu.ingest.event-file-store.storage=duckdb-hot",
"lingniu.ingest.event-file-store.path=" + tempDir.resolve("history"))
.run(context -> assertThat(context)
.hasSingleBean(EventFileStore.class)
.getBean(EventFileStore.class)
.isInstanceOf(DuckDbHotEventFileStore.class));
}
```
- [ ] **Step 2: Run the new test and verify it fails**
Run:
```bash
mvn -pl :event-file-store -Dtest=EventFileStoreAutoConfigurationTest#createsDuckDbHotStoreWhenStorageIsDuckDbHot test
```
Expected: compilation failure because `DuckDbHotEventFileStore` and `storage` property do not exist.
- [ ] **Step 3: Add the `storage` property**
Add to `EventFileStoreProperties`:
```java
/**
* Storage backend. `duckdb-hot` is the production backend; `parquet-sidecar`
* keeps the previous implementation available for compatibility tests.
*/
private String storage = "duckdb-hot";
public String getStorage() {
return storage;
}
public void setStorage(String storage) {
this.storage = storage;
}
```
- [ ] **Step 4: Switch auto-configuration by storage mode**
Change `eventFileStore(...)` to:
```java
@Bean
@ConditionalOnMissingBean
public EventFileStore eventFileStore(EventFileStoreProperties properties,
ObjectProvider<ObjectMapper> objectMapper) {
ObjectMapper mapper = mapper(objectMapper);
Path root = Path.of(properties.getPath());
ZoneId zoneId = ZoneId.of(properties.getZoneId());
String storage = properties.getStorage() == null ? "" : properties.getStorage().trim();
return switch (storage) {
case "", "duckdb-hot" -> new DuckDbHotEventFileStore(root, zoneId, mapper);
case "parquet-sidecar" -> new DuckDbParquetEventFileStore(root, zoneId, mapper);
default -> throw new IllegalStateException(
"unsupported event-file-store storage: " + properties.getStorage());
};
}
```
Import `DuckDbHotEventFileStore`.
- [ ] **Step 5: Run configuration tests**
Run:
```bash
mvn -pl :event-file-store -Dtest=EventFileStoreAutoConfigurationTest test
```
Expected: tests pass after the store class exists in Task 2.
## Task 2: Implement DuckDB Hot Store Schema and Writes
**Files:**
- Create: `modules/sinks/event-file-store/src/main/java/com/lingniu/ingest/eventfilestore/DuckDbHotEventFileStore.java`
- Create: `modules/sinks/event-file-store/src/test/java/com/lingniu/ingest/eventfilestore/DuckDbHotEventFileStoreTest.java`
- [ ] **Step 1: Write failing append/query/idempotency tests**
Create `DuckDbHotEventFileStoreTest`:
```java
class DuckDbHotEventFileStoreTest {
@TempDir
Path tempDir;
@Test
void appendsRecordsAndQueriesByVinTypeAndTime() throws Exception {
EventFileStore store = store();
store.append(rawRecord("raw-1", "VIN001", "2026-06-23T01:00:00Z"));
store.append(rawRecord("raw-2", "VIN002", "2026-06-23T01:00:01Z"));
store.append(rawRecord("raw-3", "VIN001", "2026-06-23T01:00:02Z"));
EventFileQuery query = new EventFileQuery(
ProtocolId.GB32960,
LocalDate.parse("2026-06-23"),
LocalDate.parse("2026-06-23"),
Instant.parse("2026-06-23T01:00:00Z"),
Instant.parse("2026-06-23T01:00:03Z"),
EventFileQuery.Order.DESC,
2,
"VIN001",
"RAW_ARCHIVE");
assertThat(store.query(query))
.extracting(EventFileRecord::eventId)
.containsExactly("raw-3", "raw-1");
}
@Test
void appendAllIsIdempotentByEventId() throws Exception {
EventFileStore store = store();
EventFileRecord original = rawRecord("same-id", "VIN001", "2026-06-23T01:00:00Z");
EventFileRecord replacement = new EventFileRecord(
"same-id",
ProtocolId.GB32960,
"RAW_ARCHIVE",
"VIN001",
Instant.parse("2026-06-23T01:00:05Z"),
Instant.parse("2026-06-23T01:00:06Z"),
"archive://replacement.bin",
Map.of("source", "replacement"),
"{\"replacement\":true}");
store.appendAll(List.of(original, replacement));
assertThat(store.query(new EventFileQuery(
ProtocolId.GB32960,
LocalDate.parse("2026-06-23"),
LocalDate.parse("2026-06-23"),
EventFileQuery.Order.ASC,
10,
"VIN001",
"RAW_ARCHIVE")))
.singleElement()
.satisfies(record -> {
assertThat(record.eventId()).isEqualTo("same-id");
assertThat(record.rawArchiveUri()).isEqualTo("archive://replacement.bin");
});
}
@Test
void findsRecordByRawArchiveUri() throws Exception {
EventFileStore store = store();
EventFileRecord record = rawRecord("raw-uri", "VIN001", "2026-06-23T01:00:00Z");
store.append(record);
EventFileRecord found = store.findByRawArchiveUri(record.rawArchiveUri());
assertThat(found).isNotNull();
assertThat(found.eventId()).isEqualTo("raw-uri");
}
private EventFileStore store() {
return new DuckDbHotEventFileStore(tempDir, ZoneId.of("Asia/Shanghai"), new ObjectMapper());
}
private static EventFileRecord rawRecord(String id, String vin, String eventTime) {
return new EventFileRecord(
id,
ProtocolId.GB32960,
"RAW_ARCHIVE",
vin,
Instant.parse(eventTime),
Instant.parse(eventTime).plusMillis(100),
"archive://" + id + ".bin",
Map.of("platformAccount", "Hyundai", "command", "REALTIME_REPORT"),
"{\"eventId\":\"" + id + "\"}");
}
}
```
- [ ] **Step 2: Run the test and verify it fails**
Run:
```bash
mvn -pl :event-file-store -Dtest=DuckDbHotEventFileStoreTest test
```
Expected: compilation failure because `DuckDbHotEventFileStore` does not exist.
- [ ] **Step 3: Create the hot store class**
Create `DuckDbHotEventFileStore` with this structure:
```java
public final class DuckDbHotEventFileStore implements EventFileStore {
private static final TypeReference<Map<String, String>> STRING_MAP = new TypeReference<>() {};
private final Path root;
private final Path dbPath;
private final ZoneId partitionZone;
private final ObjectMapper objectMapper;
private volatile boolean initialized;
public DuckDbHotEventFileStore(Path root, ZoneId partitionZone) {
this(root, partitionZone, new ObjectMapper());
}
public DuckDbHotEventFileStore(Path root, ZoneId partitionZone, ObjectMapper objectMapper) {
if (root == null) {
throw new IllegalArgumentException("root must not be null");
}
this.root = root.toAbsolutePath();
this.dbPath = this.root.resolve("events.duckdb");
this.partitionZone = partitionZone == null ? ZoneId.of("Asia/Shanghai") : partitionZone;
this.objectMapper = objectMapper == null ? new ObjectMapper() : objectMapper;
}
@Override
public synchronized void appendAll(List<EventFileRecord> records) throws IOException {
if (records == null || records.isEmpty()) {
return;
}
ensureInitialized();
try (Connection connection = DriverManager.getConnection(jdbcUrl())) {
connection.setAutoCommit(false);
try {
upsertRecords(connection, records);
connection.commit();
} catch (SQLException | IOException e) {
connection.rollback();
throw e;
} finally {
connection.setAutoCommit(true);
}
} catch (SQLException e) {
throw new IOException("write duckdb hot event store failed", e);
}
}
@Override
public List<EventFileRecord> query(EventFileQuery query) throws IOException {
ensureInitialized();
// Implement in Task 3.
return List.of();
}
@Override
public EventFileRecord findByRawArchiveUri(String rawArchiveUri) throws IOException {
ensureInitialized();
// Implement in Task 3.
return null;
}
}
```
- [ ] **Step 4: Add schema initialization**
Add:
```java
private void ensureInitialized() throws IOException {
if (initialized) {
return;
}
synchronized (this) {
if (initialized) {
return;
}
Files.createDirectories(root);
try (Connection connection = DriverManager.getConnection(jdbcUrl());
Statement statement = connection.createStatement()) {
statement.execute("""
CREATE TABLE IF NOT EXISTS event_records (
event_id VARCHAR PRIMARY KEY,
protocol VARCHAR NOT NULL,
event_type VARCHAR NOT NULL,
vin VARCHAR NOT NULL,
event_time_ms BIGINT NOT NULL,
ingest_time_ms BIGINT NOT NULL,
partition_date DATE NOT NULL,
raw_archive_uri VARCHAR NOT NULL,
metadata_json VARCHAR NOT NULL,
payload_json VARCHAR NOT NULL
)
""");
statement.execute("""
CREATE INDEX IF NOT EXISTS event_records_protocol_date_time_idx
ON event_records(protocol, partition_date, event_time_ms)
""");
statement.execute("""
CREATE INDEX IF NOT EXISTS event_records_vin_date_time_idx
ON event_records(protocol, vin, partition_date, event_time_ms)
""");
statement.execute("""
CREATE INDEX IF NOT EXISTS event_records_vin_type_date_time_idx
ON event_records(protocol, vin, event_type, partition_date, event_time_ms)
""");
statement.execute("""
CREATE INDEX IF NOT EXISTS event_records_raw_archive_uri_idx
ON event_records(raw_archive_uri)
""");
initialized = true;
} catch (SQLException e) {
throw new IOException("initialize duckdb hot event store failed", e);
}
}
}
```
- [ ] **Step 5: Add idempotent batch upsert**
Use DuckDB `INSERT OR REPLACE` inside one transaction:
```java
private void upsertRecords(Connection connection, List<EventFileRecord> records)
throws SQLException, IOException {
try (PreparedStatement ps = connection.prepareStatement("""
INSERT OR REPLACE INTO event_records VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
""")) {
for (EventFileRecord record : records) {
ps.setString(1, record.eventId());
ps.setString(2, record.protocol().name());
ps.setString(3, record.eventType());
ps.setString(4, record.vin());
ps.setLong(5, record.eventTime().toEpochMilli());
ps.setLong(6, record.ingestTime().toEpochMilli());
ps.setString(7, LocalDate.ofInstant(record.eventTime(), partitionZone).toString());
ps.setString(8, record.rawArchiveUri());
ps.setString(9, objectMapper.writeValueAsString(record.metadata()));
ps.setString(10, record.payloadJson());
ps.addBatch();
}
ps.executeBatch();
}
}
```
- [ ] **Step 6: Run the hot store tests**
Run:
```bash
mvn -pl :event-file-store -Dtest=DuckDbHotEventFileStoreTest test
```
Expected: query tests still fail until Task 3 implements reads; append initialization should compile.
## Task 3: Implement Hot Store Queries and Raw URI Lookup
**Files:**
- Modify: `modules/sinks/event-file-store/src/main/java/com/lingniu/ingest/eventfilestore/DuckDbHotEventFileStore.java`
- Test: `modules/sinks/event-file-store/src/test/java/com/lingniu/ingest/eventfilestore/DuckDbHotEventFileStoreTest.java`
- [ ] **Step 1: Implement `query(EventFileQuery)`**
Use prepared statements for every external value:
```java
@Override
public List<EventFileRecord> query(EventFileQuery query) throws IOException {
ensureInitialized();
String order = query.order() == EventFileQuery.Order.DESC ? "DESC" : "ASC";
StringBuilder where = new StringBuilder("""
WHERE protocol = ?
AND partition_date BETWEEN CAST(? AS DATE) AND CAST(? AS DATE)
""");
if (query.vin() != null) {
where.append(" AND vin = ?\n");
}
if (query.eventType() != null) {
where.append(" AND event_type = ?\n");
}
if (query.eventTimeFrom() != null) {
where.append(" AND event_time_ms >= ?\n");
}
if (query.eventTimeTo() != null) {
where.append(" AND event_time_ms <= ?\n");
}
String sql = """
SELECT event_id, protocol, event_type, vin, event_time_ms, ingest_time_ms,
raw_archive_uri, metadata_json, payload_json
FROM event_records
%s
ORDER BY event_time_ms %s, ingest_time_ms %s, event_id %s
LIMIT ?
""".formatted(where, order, order, order);
try (Connection connection = DriverManager.getConnection(jdbcUrl());
PreparedStatement ps = connection.prepareStatement(sql)) {
bindQuery(ps, query);
try (ResultSet rs = ps.executeQuery()) {
List<EventFileRecord> out = new ArrayList<>();
while (rs.next()) {
out.add(record(rs));
}
return out;
}
} catch (SQLException e) {
throw new IOException("query duckdb hot event store failed", e);
}
}
```
- [ ] **Step 2: Add query binding helper**
```java
private static void bindQuery(PreparedStatement ps, EventFileQuery query) throws SQLException {
int index = 1;
ps.setString(index++, query.protocol().name());
ps.setString(index++, query.dateFrom().toString());
ps.setString(index++, query.dateTo().toString());
if (query.vin() != null) {
ps.setString(index++, query.vin());
}
if (query.eventType() != null) {
ps.setString(index++, query.eventType());
}
if (query.eventTimeFrom() != null) {
ps.setLong(index++, query.eventTimeFrom().toEpochMilli());
}
if (query.eventTimeTo() != null) {
ps.setLong(index++, query.eventTimeTo().toEpochMilli());
}
ps.setInt(index, query.limit());
}
```
- [ ] **Step 3: Implement `findByRawArchiveUri`**
```java
@Override
public EventFileRecord findByRawArchiveUri(String rawArchiveUri) throws IOException {
if (rawArchiveUri == null || rawArchiveUri.isBlank()) {
return null;
}
ensureInitialized();
try (Connection connection = DriverManager.getConnection(jdbcUrl());
PreparedStatement ps = connection.prepareStatement("""
SELECT event_id, protocol, event_type, vin, event_time_ms, ingest_time_ms,
raw_archive_uri, metadata_json, payload_json
FROM event_records
WHERE raw_archive_uri = ?
ORDER BY event_type = 'RAW_ARCHIVE' DESC, ingest_time_ms DESC, event_id DESC
LIMIT 1
""")) {
ps.setString(1, rawArchiveUri);
try (ResultSet rs = ps.executeQuery()) {
return rs.next() ? record(rs) : null;
}
} catch (SQLException e) {
throw new IOException("query duckdb hot store by raw archive uri failed", e);
}
}
```
- [ ] **Step 4: Add record mapper and helpers**
```java
private EventFileRecord record(ResultSet rs) throws SQLException, IOException {
return new EventFileRecord(
rs.getString("event_id"),
protocol(rs.getString("protocol")),
rs.getString("event_type"),
rs.getString("vin"),
Instant.ofEpochMilli(rs.getLong("event_time_ms")),
Instant.ofEpochMilli(rs.getLong("ingest_time_ms")),
rs.getString("raw_archive_uri"),
readMetadata(rs.getString("metadata_json")),
rs.getString("payload_json"));
}
private Map<String, String> readMetadata(String json) throws IOException {
if (json == null || json.isBlank()) {
return Map.of();
}
return objectMapper.readValue(json, STRING_MAP);
}
private static ProtocolId protocol(String value) {
if (value == null || value.isBlank()) {
return ProtocolId.UNKNOWN;
}
try {
return ProtocolId.valueOf(value);
} catch (IllegalArgumentException ex) {
return ProtocolId.UNKNOWN;
}
}
private String jdbcUrl() {
return "jdbc:duckdb:" + dbPath;
}
```
- [ ] **Step 5: Run hot store tests**
Run:
```bash
mvn -pl :event-file-store -Dtest=DuckDbHotEventFileStoreTest test
```
Expected: all `DuckDbHotEventFileStoreTest` tests pass.
## Task 4: Preserve Legacy Store Tests and Update Defaults
**Files:**
- Modify: `modules/sinks/event-file-store/src/test/java/com/lingniu/ingest/eventfilestore/DuckDbParquetEventFileStoreTest.java`
- Modify: `modules/apps/vehicle-history-app/src/main/resources/application.yml`
- Modify: `modules/apps/vehicle-history-app/src/test/java/com/lingniu/ingest/historyapp/VehicleHistoryAppDefaultsTest.java`
- [ ] **Step 1: Keep Parquet tests explicitly legacy**
No behavior change is needed in `DuckDbParquetEventFileStoreTest`; leave it instantiating `DuckDbParquetEventFileStore` directly. Add a class comment:
```java
/**
* Compatibility coverage for the legacy Parquet sidecar backend.
* Production history uses DuckDbHotEventFileStore through auto-configuration.
*/
class DuckDbParquetEventFileStoreTest {
```
- [ ] **Step 2: Set vehicle-history-app storage default**
Add to `modules/apps/vehicle-history-app/src/main/resources/application.yml`:
```yaml
event-file-store:
enabled: ${EVENT_FILE_STORE_ENABLED:true}
storage: ${EVENT_FILE_STORE_STORAGE:duckdb-hot}
path: ${EVENT_FILE_STORE_PATH:./target/event-store/}
zone-id: ${EVENT_FILE_STORE_ZONE_ID:Asia/Shanghai}
batch-size: ${EVENT_FILE_STORE_BATCH_SIZE:1000}
flush-interval-millis: ${EVENT_FILE_STORE_FLUSH_INTERVAL_MILLIS:1000}
```
Keep existing indentation and only add `storage`; if `batch-size` already exists, update it to `1000`.
- [ ] **Step 3: Update app default test**
In `VehicleHistoryAppDefaultsTest`, assert:
```java
assertThat(context.getEnvironment()
.getProperty("lingniu.ingest.event-file-store.storage"))
.isEqualTo("duckdb-hot");
```
- [ ] **Step 4: Run app default tests**
Run:
```bash
mvn -pl :vehicle-history-app -Dtest=VehicleHistoryAppDefaultsTest test
```
Expected: default configuration test passes.
## Task 5: Verify History Ingest Still Archives RAW Bytes
**Files:**
- Modify: `modules/services/event-history-service/src/test/java/com/lingniu/ingest/eventhistory/EventHistoryEnvelopeIngestorTest.java`
- Test: `modules/services/event-history-service/src/test/java/com/lingniu/ingest/eventhistory/Gb32960DecodedFrameServiceTest.java`
- [ ] **Step 1: Add an integration-style test using the hot store**
Add a test that writes a RAW envelope through `EventHistoryEnvelopeIngestor` into `DuckDbHotEventFileStore`, then finds it by URI.
```java
@Test
void rawArchiveEnvelopeCanBeFoundFromDuckDbHotStoreByUri(@TempDir Path tempDir) throws Exception {
EventFileStore store = new DuckDbHotEventFileStore(tempDir, ZoneId.of("Asia/Shanghai"), OBJECT_MAPPER);
CapturingArchiveStore archive = new CapturingArchiveStore();
EventHistoryEnvelopeIngestor ingestor =
new EventHistoryEnvelopeIngestor(store, new TelemetryEnvelopeRecordMapper(), archive);
String rawArchiveKey = "2026/06/23/GB32960/VINRAW001/raw-event-hot.bin";
String rawArchiveUri = "archive://" + rawArchiveKey;
byte[] rawBytes = new byte[]{0x23, 0x23, 0x02, 0x01};
VehicleEnvelope envelope = VehicleEnvelope.newBuilder()
.setSchemaVersion("1.0")
.setEventId("raw-event-hot")
.setVin("VINRAW001")
.setSource("GB32960")
.setProtocolVersion("V2016")
.setEventTimeMs(1_782_112_400_000L)
.setIngestTimeMs(1_782_112_401_000L)
.putMetadata(RawArchiveKeys.META_KEY, rawArchiveKey)
.putMetadata(RawArchiveKeys.META_URI, rawArchiveUri)
.setRawArchive(RawArchiveRef.newBuilder()
.setUri(rawArchiveUri)
.setSizeBytes(rawBytes.length)
.setData(ByteString.copyFrom(rawBytes))
.build())
.build();
EnvelopeIngestResult result = ingestor.tryIngest(envelope.toByteArray());
assertThat(result.status()).isEqualTo(EnvelopeIngestResult.Status.STORED);
assertThat(store.findByRawArchiveUri(rawArchiveUri))
.isNotNull()
.extracting(EventFileRecord::eventId)
.isEqualTo("raw-event-hot");
assertThat(archive.bytesByKey).containsEntry(rawArchiveKey, rawBytes);
}
```
- [ ] **Step 2: Run the event history ingestor test**
Run:
```bash
mvn -pl :event-history-service -Dtest=EventHistoryEnvelopeIngestorTest test
```
Expected: test passes.
- [ ] **Step 3: Run decoded frame service tests**
Run:
```bash
mvn -pl :event-history-service -Dtest=Gb32960DecodedFrameServiceTest test
```
Expected: existing replay/snapshot tests pass. If they use a fake store, no change is needed.
## Task 6: Full Module Verification
**Files:**
- No source changes unless failures expose missing imports or config assertions.
- [ ] **Step 1: Run sink module tests**
Run:
```bash
mvn -pl :event-file-store test
```
Expected: all event-file-store tests pass, including both hot and legacy stores.
- [ ] **Step 2: Run event-history-service tests**
Run:
```bash
mvn -pl :event-history-service test
```
Expected: all event-history-service tests pass.
- [ ] **Step 3: Package vehicle-history-app**
Run:
```bash
mvn -pl :vehicle-history-app -am package -DskipTests
```
Expected: package succeeds.
## Task 7: Local Runtime Verification
**Files:**
- No committed source changes.
- [ ] **Step 1: Stop any old history service on port 20200**
Run:
```bash
lsof -tiTCP:20200 -sTCP:LISTEN | xargs -r kill
```
Expected: no command output, or the old process exits.
- [ ] **Step 2: Start vehicle-history-app with hot DuckDB store**
Run:
```bash
EVENT_FILE_STORE_STORAGE=duckdb-hot \
EVENT_FILE_STORE_PATH=./target/live-history-event-store \
SINK_ARCHIVE_PATH=./target/live-history-archive \
KAFKA_BROKERS=114.55.58.251:9092 \
KAFKA_CONSUMER_ENABLED=true \
KAFKA_GROUP_HISTORY=vehicle-history-hot-$(date +%Y%m%d%H%M%S) \
java --sun-misc-unsafe-memory-access=allow \
-jar modules/apps/vehicle-history-app/target/vehicle-history-app.jar
```
Expected:
- app starts on `http://127.0.0.1:20200`;
- logs show Kafka consumer subscribed;
- `target/live-history-event-store/events.duckdb` is created.
- [ ] **Step 3: Verify health**
Run:
```bash
curl -sS http://127.0.0.1:20200/actuator/health
```
Expected:
```json
{"status":"UP"}
```
- [ ] **Step 4: Verify a recent VIN query**
Run with a VIN observed in live archive:
```bash
curl -sS 'http://127.0.0.1:20200/api/event-history/gb32960/telemetry-snapshots?vin=LNXNEGRR9SR318194&platformAccount=Hyundai&dateFrom=2026-06-23&dateTo=2026-06-24&order=DESC&limit=3'
```
Expected:
- response is a JSON array;
- if live traffic for that VIN exists after service start, at least one snapshot appears;
- `rawArchiveUris` point to existing files under `target/live-history-archive`.
- [ ] **Step 5: Verify raw frame replay**
Use one `rawArchiveUri` from Step 4:
```bash
curl -sS 'http://127.0.0.1:20200/api/event-history/gb32960/frame?rawArchiveUri=archive://REPLACE_ME&platformAccount=Hyundai'
```
Expected:
- response contains `vin`, `command`, `eventTime`, and parsed `blocks`;
- no `raw archive is missing` warning for the selected URI.
## Self-Review Checklist
- [ ] The new hot store does not rewrite Parquet files on every append.
- [ ] `appendAll` writes one batch in one transaction.
- [ ] Duplicate `event_id` replay is idempotent.
- [ ] Existing history HTTP APIs still use `EventFileStore`, so controller contracts are unchanged.
- [ ] RAW bytes still land in `ArchiveStore`.
- [ ] `findByRawArchiveUri` is backed by DuckDB index.
- [ ] No MySQL credential, RDS host, username, or password appears in the plan or source files.
- [ ] This plan does not claim the full production goal is complete; it only lands the history foundation.

File diff suppressed because it is too large Load Diff

View File

@@ -1,46 +0,0 @@
# JT808 Kafka Streaming Mileage Plan
## Current Goal
Consume JT808 Kafka location events and write the derived daily mileage into the common vehicle-stat metric repository.
## Current Design
- Only supported message backbone: Kafka.
- Source topic: `vehicle.event.jt808.v1`
- Runtime app: `vehicle-analytics-app`
- Runtime state: none outside `vehicle_stat_metric`
- Metric output: `VehicleStatRepository.recordDailyMileageSample(...)`
- Production metric storage: JDBC/MySQL `vehicle_stat_metric`
- Date boundary: `Asia/Shanghai`
- Calculation method: `JT808_TOTAL_MILEAGE_DIFF`
JT808 daily mileage is calculated only from the GPS total mileage reported by JT808 location additional information:
```text
daily_mileage_km = max_total_mileage_km - min_total_mileage_km
```
The first valid local-day sample writes `daily_mileage_km=0.0`. Later ordered or replayed samples update the same metric row with:
```text
metric_value = max_total_mileage_km - min_total_mileage_km
metric_key = daily_mileage_km
```
The local-day minimum and maximum GPS total mileage values stay on that same metric row as calculation source columns so restarts recover from MySQL without Redis or another state table.
## Runtime Properties
```text
KAFKA_TOPIC_JT808_EVENT=vehicle.event.jt808.v1
VEHICLE_STAT_ENABLED=true
VEHICLE_STAT_JT808_MILEAGE_ENABLED=true
MYSQL_JDBC_URL=<jdbc-url>
MYSQL_USERNAME=<user>
MYSQL_PASSWORD=<password>
```
## Notes
Do not create or write a protocol-specific JT808 daily-mileage table. Do not add another message backbone, distance accumulation, integral calculation, Redis state, or memory state back into this path unless the mileage definition changes again.

View File

@@ -1,720 +0,0 @@
# Go Vehicle Ingest Redesign Phase 1 Implementation Plan
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
**Goal:** Build the first production-capable Go runtime for GB32960, JT808, and Yutong MQTT ingestion with unified Kafka, TDengine, MySQL statistics, and Redis realtime state boundaries.
**Architecture:** Create a new single Go module at `go/vehicle-gateway` and migrate only the useful pieces from the existing `go/ingest-edge` and `go/vehicle-state` prototypes. The gateway produces unified envelopes to Kafka; independent consumers write TDengine history, MySQL daily metrics, and Redis realtime state.
**Tech Stack:** Go 1.26+, Kafka (`segmentio/kafka-go`), Redis (`redis/go-redis`), MySQL (`go-sql-driver/mysql`), TDengine official Go connector, MQTT (`eclipse/paho.mqtt.golang`), standard library TCP.
---
## File Structure
- Create: `go/vehicle-gateway/go.mod`
- Create: `go/vehicle-gateway/cmd/gateway/main.go`
- Create: `go/vehicle-gateway/cmd/history-writer/main.go`
- Create: `go/vehicle-gateway/cmd/stat-writer/main.go`
- Create: `go/vehicle-gateway/cmd/realtime-api/main.go`
- Create: `go/vehicle-gateway/internal/envelope/envelope.go`
- Create: `go/vehicle-gateway/internal/envelope/envelope_test.go`
- Create: `go/vehicle-gateway/internal/protocol/jt808/*`
- Create: `go/vehicle-gateway/internal/protocol/gb32960/*`
- Create: `go/vehicle-gateway/internal/protocol/yutongmqtt/*`
- Create: `go/vehicle-gateway/internal/gateway/*`
- Create: `go/vehicle-gateway/internal/identity/*`
- Create: `go/vehicle-gateway/internal/eventbus/*`
- Create: `go/vehicle-gateway/internal/history/*`
- Create: `go/vehicle-gateway/internal/stats/*`
- Create: `go/vehicle-gateway/internal/realtime/*`
- Create: `go/vehicle-gateway/internal/observability/*`
- Modify: `README.md`
- Modify: `docs/target-architecture.md`
- Create: `deploy/portainer/docker-compose-go.yml`
The existing `go/ingest-edge` and `go/vehicle-state` directories are migration sources only. After phase 1 is verified, remove or mark them superseded.
---
### Task 1: Create Single Go Module
**Files:**
- Create: `go/vehicle-gateway/go.mod`
- Create: `go/vehicle-gateway/internal/observability/logger.go`
- Create: `go/vehicle-gateway/cmd/gateway/main.go`
- [ ] **Step 1: Create module manifest**
Add `go/vehicle-gateway/go.mod`:
```go
module lingniu-vehicle-ingest/go/vehicle-gateway
go 1.26
require (
github.com/eclipse/paho.mqtt.golang v1.5.1
github.com/go-sql-driver/mysql v1.9.3
github.com/redis/go-redis/v9 v9.17.2
github.com/segmentio/kafka-go v0.4.49
github.com/taosdata/driver-go/v3 v3.8.1
)
```
- [ ] **Step 2: Add logger helper**
Add `go/vehicle-gateway/internal/observability/logger.go`:
```go
package observability
import (
"log/slog"
"os"
)
func NewLogger(service string) *slog.Logger {
handler := slog.NewJSONHandler(os.Stdout, &slog.HandlerOptions{AddSource: true})
return slog.New(handler).With("service", service)
}
```
- [ ] **Step 3: Add temporary gateway entrypoint**
Add `go/vehicle-gateway/cmd/gateway/main.go`:
```go
package main
import "lingniu-vehicle-ingest/go/vehicle-gateway/internal/observability"
func main() {
logger := observability.NewLogger("vehicle-gateway")
logger.Info("vehicle gateway scaffold started")
}
```
- [ ] **Step 4: Verify scaffold builds**
Run:
```bash
cd go/vehicle-gateway
go mod tidy
go test ./...
go build ./cmd/gateway
```
Expected: all commands exit `0`.
---
### Task 2: Define Unified Envelope
**Files:**
- Create: `go/vehicle-gateway/internal/envelope/envelope.go`
- Create: `go/vehicle-gateway/internal/envelope/envelope_test.go`
- [ ] **Step 1: Write envelope tests**
Add `go/vehicle-gateway/internal/envelope/envelope_test.go`:
```go
package envelope
import "testing"
func TestFrameEnvelopeVehicleKeyPrefersVIN(t *testing.T) {
e := FrameEnvelope{Protocol: ProtocolJT808, VIN: "LNBVIN00000000001", Phone: "013307795425"}
if got := e.VehicleKey(); got != "LNBVIN00000000001" {
t.Fatalf("VehicleKey() = %q", got)
}
}
func TestFrameEnvelopeVehicleKeyFallsBackToPhone(t *testing.T) {
e := FrameEnvelope{Protocol: ProtocolJT808, Phone: "013307795425"}
if got := e.VehicleKey(); got != "JT808:013307795425" {
t.Fatalf("VehicleKey() = %q", got)
}
}
func TestFrameEnvelopeEventIDStable(t *testing.T) {
e := FrameEnvelope{
Protocol: ProtocolJT808,
MessageID: "0x0200",
Phone: "013307795425",
Sequence: 1,
EventTimeMS: 1782745114000,
ReceivedAtMS: 1782745114999,
RawHex: "7e02000000ff7e",
}
a := e.StableEventID()
b := e.StableEventID()
if a == "" || a != b {
t.Fatalf("event id must be non-empty and stable: %q %q", a, b)
}
}
```
- [ ] **Step 2: Run tests and confirm failure**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/envelope
```
Expected: fail because `FrameEnvelope` is not defined.
- [ ] **Step 3: Implement envelope**
Add `go/vehicle-gateway/internal/envelope/envelope.go`:
```go
package envelope
import (
"crypto/sha256"
"encoding/hex"
"encoding/json"
"fmt"
"strings"
)
type Protocol string
const (
ProtocolGB32960 Protocol = "GB32960"
ProtocolJT808 Protocol = "JT808"
ProtocolYutongMQTT Protocol = "YUTONG_MQTT"
)
type ParseStatus string
const (
ParseOK ParseStatus = "OK"
ParsePartial ParseStatus = "PARTIAL"
ParseBadFrame ParseStatus = "BAD_FRAME"
)
type FrameEnvelope struct {
EventID string `json:"event_id"`
TraceID string `json:"trace_id"`
Protocol Protocol `json:"protocol"`
MessageID string `json:"message_id"`
Sequence uint16 `json:"sequence"`
VIN string `json:"vin,omitempty"`
VehicleKeyHint string `json:"vehicle_key,omitempty"`
Phone string `json:"phone,omitempty"`
DeviceID string `json:"device_id,omitempty"`
Plate string `json:"plate,omitempty"`
SourceEndpoint string `json:"source_endpoint,omitempty"`
EventTimeMS int64 `json:"event_time_ms"`
ReceivedAtMS int64 `json:"received_at_ms"`
RawHex string `json:"raw_hex,omitempty"`
RawText string `json:"raw_text,omitempty"`
Parsed map[string]any `json:"parsed,omitempty"`
Fields map[string]any `json:"fields,omitempty"`
ParseStatus ParseStatus `json:"parse_status"`
ParseError string `json:"parse_error,omitempty"`
}
func (e FrameEnvelope) VehicleKey() string {
if key := strings.TrimSpace(e.VIN); key != "" {
return key
}
if key := strings.TrimSpace(e.VehicleKeyHint); key != "" {
return key
}
if key := strings.TrimSpace(e.Phone); key != "" {
return string(e.Protocol) + ":" + key
}
if key := strings.TrimSpace(e.DeviceID); key != "" {
return string(e.Protocol) + ":" + key
}
return string(e.Protocol) + ":unknown"
}
func (e FrameEnvelope) StableEventID() string {
if strings.TrimSpace(e.EventID) != "" {
return e.EventID
}
input := fmt.Sprintf("%s|%s|%s|%d|%d|%s",
e.Protocol, e.MessageID, e.VehicleKey(), e.Sequence, e.EventTimeMS, e.RawHex)
sum := sha256.Sum256([]byte(input))
return hex.EncodeToString(sum[:16])
}
func (e FrameEnvelope) MarshalJSONBytes() ([]byte, error) {
if e.EventID == "" {
e.EventID = e.StableEventID()
}
if e.ParseStatus == "" {
e.ParseStatus = ParseOK
}
return json.Marshal(e)
}
```
- [ ] **Step 4: Verify tests pass**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/envelope
```
Expected: pass.
---
### Task 3: Implement JT808 Frame and 0200 Parser
**Files:**
- Create: `go/vehicle-gateway/internal/protocol/jt808/parser.go`
- Create: `go/vehicle-gateway/internal/protocol/jt808/parser_test.go`
- [ ] **Step 1: Add sample frame tests**
Use the production sample provided in the thread:
```text
7E020000320133077954250001000000000048000301D2C4C707376139000A00E6004F26063016235701040001900C2504000000000202000030011F31010F867E
```
Expected assertions:
- message id is `0x0200`
- phone keeps normalized BCD string `013307795425`
- sequence is `1`
- `fields.total_mileage_km` is parsed from additional item `0x01`
- latitude, longitude, speed, direction, alarm, status, and device time are present
- [ ] **Step 2: Implement parser**
Implementation rules:
- strip `0x7e` start/end delimiters
- unescape `0x7d 0x02 -> 0x7e`
- unescape `0x7d 0x01 -> 0x7d`
- verify XOR checksum
- parse 2011/2013 common header
- parse BCD phone from 6 bytes and keep both raw and normalized values in `parsed.header`
- parse 0200 fixed body
- parse additional item list as `id,length,value_hex`
- parse additional `0x01` as `total_mileage_km = uint32 / 10`
- [ ] **Step 3: Verify**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/protocol/jt808
```
Expected: pass.
---
### Task 4: Implement GB32960 Frame and Data Unit Parser
**Files:**
- Create: `go/vehicle-gateway/internal/protocol/gb32960/parser.go`
- Create: `go/vehicle-gateway/internal/protocol/gb32960/parser_test.go`
- [ ] **Step 1: Add parser tests**
Tests must cover:
- `##` frame boundary
- command id
- response flag
- 17-byte VIN
- encryption flag
- payload length
- BCC verification
- realtime data command `0x02`
- reissue data command `0x03`
- data unit `0x01` vehicle status fields
- data unit `0x05` position fields
- [ ] **Step 2: Implement parser**
Implementation rules:
- parse header without allocating large temporary buffers
- keep RAW as hex in envelope
- write all recognized data units to `Parsed`
- write core fields to `Fields`
- unsupported data unit stays in `Parsed["unknown_units"]`
- [ ] **Step 3: Verify**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/protocol/gb32960
```
Expected: pass.
---
### Task 5: Implement Kafka Sink
**Files:**
- Create: `go/vehicle-gateway/internal/eventbus/kafka_sink.go`
- Create: `go/vehicle-gateway/internal/eventbus/kafka_sink_test.go`
- [ ] **Step 1: Add sink interface**
Create a sink interface:
```go
package eventbus
import (
"context"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
)
type Sink interface {
PublishRaw(context.Context, envelope.FrameEnvelope) error
PublishUnified(context.Context, envelope.FrameEnvelope) error
Close() error
}
```
- [ ] **Step 2: Implement Kafka topic routing**
Rules:
- `GB32960 -> vehicle.raw.gb32960.v1`
- `JT808 -> vehicle.raw.jt808.v1`
- `YUTONG_MQTT -> vehicle.raw.yutong-mqtt.v1`
- unified topic is `vehicle.event.unified.v1`
- message key is `env.VehicleKey()`
- [ ] **Step 3: Verify routing with unit tests**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/eventbus
```
Expected: pass.
---
### Task 6: Implement TDengine History Writer
**Files:**
- Create: `go/vehicle-gateway/internal/history/schema.go`
- Create: `go/vehicle-gateway/internal/history/writer.go`
- Create: `go/vehicle-gateway/internal/history/writer_test.go`
- Create: `go/vehicle-gateway/cmd/history-writer/main.go`
- [ ] **Step 1: Add schema bootstrap SQL**
Implement schema strings for:
- database `lingniu_vehicle_ts`
- stable `raw_frames`
- stable `vehicle_locations`
- stable `vehicle_mileage_points`
- [ ] **Step 2: Implement writer**
Rules:
- `AppendRawFrame` always writes one raw row.
- `AppendLocation` writes only when longitude and latitude exist.
- `AppendMileagePoint` writes only when `total_mileage_km` exists.
- child table name is deterministic hash of `protocol + vehicle_key`.
- escape tag values.
- [ ] **Step 3: Use TDengine official driver**
Import WebSocket driver in command:
```go
import _ "github.com/taosdata/driver-go/v3/taosWS"
```
Default driver name:
```text
TDENGINE_DRIVER=taosWS
```
Short-term compatibility:
```text
TDENGINE_DRIVER=taosSql
```
Only use `taosSql` when ECS TDengine WebSocket is not available.
- [ ] **Step 4: Verify**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/history
go build ./cmd/history-writer
```
Expected: pass.
---
### Task 7: Implement MySQL Daily Metric Writer
**Files:**
- Create: `go/vehicle-gateway/internal/stats/schema.go`
- Create: `go/vehicle-gateway/internal/stats/daily_metric.go`
- Create: `go/vehicle-gateway/internal/stats/daily_metric_test.go`
- Create: `go/vehicle-gateway/cmd/stat-writer/main.go`
- [ ] **Step 1: Add schema bootstrap**
Implement `vehicle_daily_metric` schema from the design spec.
- [ ] **Step 2: Add metric derivation tests**
Test cases:
- no `total_mileage_km` produces no metric
- one sample produces:
- `daily_mileage_km = 0`
- `daily_total_mileage_km = sample`
- later larger sample updates:
- `latest_total_mileage_km`
- `daily_mileage_km`
- `daily_total_mileage_km`
- out-of-order smaller sample updates:
- `first_total_mileage_km`
- `daily_mileage_km`
- [ ] **Step 3: Implement MySQL upsert**
Use one table and one idempotent upsert:
```sql
INSERT INTO vehicle_daily_metric
(vin, stat_date, protocol, metric_key, metric_value, metric_unit,
first_total_mileage_km, latest_total_mileage_km, sample_count, calculation_method)
VALUES (?, ?, ?, ?, ?, 'km', ?, ?, 1, 'TOTAL_MILEAGE_DIFF')
ON DUPLICATE KEY UPDATE
first_total_mileage_km = LEAST(first_total_mileage_km, VALUES(first_total_mileage_km)),
latest_total_mileage_km = GREATEST(latest_total_mileage_km, VALUES(latest_total_mileage_km)),
metric_value = CASE
WHEN metric_key = 'daily_mileage_km'
THEN GREATEST(latest_total_mileage_km, VALUES(latest_total_mileage_km))
- LEAST(first_total_mileage_km, VALUES(first_total_mileage_km))
WHEN metric_key = 'daily_total_mileage_km'
THEN GREATEST(latest_total_mileage_km, VALUES(latest_total_mileage_km))
ELSE VALUES(metric_value)
END,
sample_count = sample_count + 1,
updated_at = CURRENT_TIMESTAMP
```
- [ ] **Step 4: Verify**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/stats
go build ./cmd/stat-writer
```
Expected: pass.
---
### Task 8: Implement Redis Realtime API
**Files:**
- Create: `go/vehicle-gateway/internal/realtime/repository.go`
- Create: `go/vehicle-gateway/internal/realtime/repository_test.go`
- Create: `go/vehicle-gateway/internal/realtime/http.go`
- Create: `go/vehicle-gateway/cmd/realtime-api/main.go`
- [ ] **Step 1: Add repository contract**
Methods:
- `Update(ctx, envelope.FrameEnvelope) error`
- `GetMerged(ctx, vin string) (Snapshot, error)`
- `GetProtocol(ctx, vin string, protocol envelope.Protocol) (Snapshot, error)`
- `IsOnline(ctx, vin string) (OnlineStatus, error)`
- [ ] **Step 2: Implement merge logic**
Rules:
- update `vehicle:latest:{vin}:{protocol}`
- update `vehicle:latest:{vin}` with newest fields
- update `vehicle:online:{vin}`
- update sorted set `vehicle:last_seen`
- [ ] **Step 3: Implement HTTP API**
Routes:
- `GET /api/realtime/vehicles/{vin}`
- `GET /api/realtime/vehicles/{vin}/online`
- `GET /api/realtime/vehicles/{vin}/protocols/{protocol}`
- [ ] **Step 4: Verify**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/realtime
go build ./cmd/realtime-api
```
Expected: pass.
---
### Task 9: Wire Gateway Runtime
**Files:**
- Create: `go/vehicle-gateway/internal/gateway/tcp_server.go`
- Create: `go/vehicle-gateway/internal/gateway/mqtt_client.go`
- Modify: `go/vehicle-gateway/cmd/gateway/main.go`
- [ ] **Step 1: Implement TCP server**
Rules:
- one goroutine per accepted connection
- bounded max connections
- read timeout and idle timeout
- protocol-specific frame extractor
- structured peer endpoint
- graceful shutdown on SIGTERM
- [ ] **Step 2: Implement MQTT client**
Rules:
- connect with official production config from environment
- subscribe configured topic list
- convert each message into envelope
- publish raw and unified events
- reconnect with backoff
- [ ] **Step 3: Verify local JSON mode**
Run without Kafka:
```bash
cd go/vehicle-gateway
GB32960_TCP_ADDR=:132960 JT808_TCP_ADDR=:18080 go run ./cmd/gateway
```
Expected: service starts and logs configured listeners.
---
### Task 10: Docker and ECS Deployment
**Files:**
- Create: `go/vehicle-gateway/Dockerfile`
- Create: `deploy/portainer/docker-compose-go.yml`
- Modify: `docs/operations/current-ecs-deployment.md`
- [ ] **Step 1: Add multi-stage Dockerfile**
Build all commands:
- `gateway`
- `history-writer`
- `stat-writer`
- `realtime-api`
- [ ] **Step 2: Add Portainer compose**
Services:
- `go-vehicle-gateway`
- `go-history-writer`
- `go-stat-writer`
- `go-realtime-api`
Each service must include:
- restart policy
- memory limit
- Kafka env
- MySQL/TDengine/Redis env as needed
- logging options
- [ ] **Step 3: Verify Linux build**
Run:
```bash
cd go/vehicle-gateway
GOOS=linux GOARCH=amd64 go build ./cmd/gateway
GOOS=linux GOARCH=amd64 go build ./cmd/history-writer
GOOS=linux GOARCH=amd64 go build ./cmd/stat-writer
GOOS=linux GOARCH=amd64 go build ./cmd/realtime-api
```
Expected: all commands exit `0`.
---
### Task 11: Production Verification
**Files:**
- Create: `docs/operations/go-vehicle-gateway-verification.md`
- [ ] **Step 1: Record test commands**
Document commands to verify:
- gateway process health
- Kafka topic consumption
- TDengine row counts
- MySQL daily metric rows
- Redis realtime lookup
- [ ] **Step 2: Validate real traffic**
Evidence required:
- one real 32960 VIN with RAW, location, mileage point, daily metric, Redis snapshot
- one real JT808 phone/VIN with RAW, location, mileage point, daily metric, Redis snapshot
- one real Yutong MQTT VIN with RAW and Redis snapshot
- [ ] **Step 3: Keep old Java services until evidence is captured**
Only disable Java equivalents after the evidence file contains successful command outputs and timestamps.
---
## Self-Review Checklist
- The plan creates one new Go module instead of extending scattered prototypes.
- The plan covers GB32960, JT808, and Yutong MQTT ingress.
- The plan covers Kafka, TDengine, MySQL, and Redis.
- The plan includes 32960 and 808 daily mileage and daily total mileage.
- The plan includes local and ECS verification.
- The plan does not restore Xinda Push.
- The plan keeps Java services untouched until Go evidence exists.

View File

@@ -0,0 +1,489 @@
# NATS Kafka Ingest Refactor Implementation Plan
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
**Goal:** Move vehicle ingress publishing from direct Kafka writes to NATS JetStream, then bridge NATS to existing Kafka topics so downstream history, realtime, and stats services keep working.
**Architecture:** `gateway` publishes parsed RAW and UNIFIED envelopes to NATS JetStream subjects. A new `nats-kafka-bridge` durable pull consumer reads from JetStream, writes batches to Kafka, and only ACKs NATS after Kafka succeeds. Existing Kafka consumers remain unchanged in this phase.
**Tech Stack:** Go 1.26, `github.com/nats-io/nats.go`, NATS JetStream, Kafka via `segmentio/kafka-go`, existing `eventbus.Sink`, systemd single-binary deployment on ECS.
---
## File Structure
- Create `go/vehicle-gateway/internal/eventbus/nats_sink.go`: NATS JetStream implementation of `eventbus.Sink`.
- Create `go/vehicle-gateway/internal/eventbus/nats_sink_test.go`: topic/subject routing and envelope serialization tests.
- Modify `go/vehicle-gateway/cmd/gateway/main.go`: choose NATS sink when `NATS_URL` is configured; keep Kafka path as fallback.
- Create `go/vehicle-gateway/cmd/nats-kafka-bridge/main.go`: bridge process that creates JetStream stream/consumer, pulls NATS messages, writes Kafka batches, then ACKs.
- Create `go/vehicle-gateway/cmd/nats-kafka-bridge/main_test.go`: bridge batch ACK behavior and topic routing tests.
- Modify `go/vehicle-gateway/go.mod`: add `github.com/nats-io/nats.go`.
- Create `deploy/nats/nats-server.conf`: single-node JetStream config for Kafka ECS.
- Modify ECS deploy process: include `nats-kafka-bridge` binary and systemd service.
## Task 1: Add NATS JetStream Sink
**Files:**
- Create: `go/vehicle-gateway/internal/eventbus/nats_sink.go`
- Create: `go/vehicle-gateway/internal/eventbus/nats_sink_test.go`
- Modify: `go/vehicle-gateway/go.mod`
- [x] **Step 1: Write failing subject routing test**
Create `go/vehicle-gateway/internal/eventbus/nats_sink_test.go` with a fake JetStream publisher:
```go
package eventbus
import (
"context"
"encoding/json"
"testing"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
)
func TestNATSSinkRoutesRawAndUnifiedSubjects(t *testing.T) {
publisher := &recordingNATSPublisher{}
sink := newNATSSinkWithPublisher(publisher, NATSConfig{
RawSubjects: map[envelope.Protocol]string{
envelope.ProtocolJT808: "vehicle.raw.jt808.v1",
},
UnifiedSubject: "vehicle.event.unified.v1",
})
env := envelope.FrameEnvelope{Protocol: envelope.ProtocolJT808, Phone: "13307795425", MessageID: "0x0200"}
if err := sink.PublishRaw(context.Background(), env); err != nil {
t.Fatalf("PublishRaw() error = %v", err)
}
if err := sink.PublishUnified(context.Background(), env); err != nil {
t.Fatalf("PublishUnified() error = %v", err)
}
if got, want := publisher.messages[0].subject, "vehicle.raw.jt808.v1"; got != want {
t.Fatalf("raw subject = %q, want %q", got, want)
}
if got, want := publisher.messages[1].subject, "vehicle.event.unified.v1"; got != want {
t.Fatalf("unified subject = %q, want %q", got, want)
}
var decoded envelope.FrameEnvelope
if err := json.Unmarshal(publisher.messages[0].data, &decoded); err != nil {
t.Fatalf("raw payload is not envelope json: %v", err)
}
if decoded.Phone != "13307795425" {
t.Fatalf("decoded phone = %q", decoded.Phone)
}
}
type recordingNATSPublisher struct {
messages []recordedNATSMessage
}
type recordedNATSMessage struct {
subject string
data []byte
}
func (p *recordingNATSPublisher) Publish(_ context.Context, subject string, data []byte, _ ...NATSPublishOption) error {
p.messages = append(p.messages, recordedNATSMessage{subject: subject, data: append([]byte(nil), data...)})
return nil
}
```
Run: `go test ./internal/eventbus -run TestNATSSinkRoutesRawAndUnifiedSubjects -count=1`
Expected: FAIL because `NATSConfig`, `newNATSSinkWithPublisher`, and `NATSPublishOption` do not exist.
- [x] **Step 2: Implement NATS sink**
Create `go/vehicle-gateway/internal/eventbus/nats_sink.go`:
```go
package eventbus
import (
"context"
"errors"
"fmt"
"github.com/nats-io/nats.go"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
)
type NATSConfig struct {
URL string
Stream string
RawSubjects map[envelope.Protocol]string
UnifiedSubject string
}
type NATSSink struct {
conn *nats.Conn
publisher natsPublisher
rawSubjects map[envelope.Protocol]string
unifiedSubject string
}
type NATSPublishOption = nats.PubOpt
type natsPublisher interface {
Publish(context.Context, string, []byte, ...NATSPublishOption) error
}
func NewNATSSink(cfg NATSConfig) (*NATSSink, error) {
if cfg.URL == "" {
return nil, errors.New("nats url is required")
}
conn, err := nats.Connect(cfg.URL, nats.Name("lingniu-vehicle-gateway"))
if err != nil {
return nil, err
}
js, err := conn.JetStream()
if err != nil {
conn.Close()
return nil, err
}
sink := newNATSSinkWithPublisher(natsJetStreamPublisher{js: js}, cfg)
sink.conn = conn
return sink, nil
}
func newNATSSinkWithPublisher(publisher natsPublisher, cfg NATSConfig) *NATSSink {
rawSubjects := map[envelope.Protocol]string{
envelope.ProtocolGB32960: "vehicle.raw.gb32960.v1",
envelope.ProtocolJT808: "vehicle.raw.jt808.v1",
envelope.ProtocolYutongMQTT: "vehicle.raw.yutong-mqtt.v1",
}
for protocol, subject := range cfg.RawSubjects {
if subject != "" {
rawSubjects[protocol] = subject
}
}
unifiedSubject := cfg.UnifiedSubject
if unifiedSubject == "" {
unifiedSubject = "vehicle.event.unified.v1"
}
return &NATSSink{publisher: publisher, rawSubjects: rawSubjects, unifiedSubject: unifiedSubject}
}
func (s *NATSSink) PublishRaw(ctx context.Context, env envelope.FrameEnvelope) error {
subject, ok := s.rawSubjects[env.Protocol]
if !ok || subject == "" {
return fmt.Errorf("raw subject not configured for protocol %s", env.Protocol)
}
return s.publish(ctx, subject, env)
}
func (s *NATSSink) PublishUnified(ctx context.Context, env envelope.FrameEnvelope) error {
return s.publish(ctx, s.unifiedSubject, env)
}
func (s *NATSSink) Close() error {
if s != nil && s.conn != nil {
s.conn.Drain()
s.conn.Close()
}
return nil
}
func (s *NATSSink) publish(ctx context.Context, subject string, env envelope.FrameEnvelope) error {
payload, err := env.MarshalJSONBytes()
if err != nil {
return err
}
return s.publisher.Publish(ctx, subject, payload, nats.MsgId(env.StableEventID()))
}
type natsJetStreamPublisher struct {
js nats.JetStreamContext
}
func (p natsJetStreamPublisher) Publish(ctx context.Context, subject string, data []byte, opts ...NATSPublishOption) error {
_, err := p.js.Publish(subject, data, append(opts, nats.Context(ctx))...)
return err
}
```
- [x] **Step 3: Run eventbus tests**
Run: `go get github.com/nats-io/nats.go@latest && go test ./internal/eventbus -count=1`
Expected: PASS.
## Task 2: Gateway Sink Selection
**Files:**
- Modify: `go/vehicle-gateway/cmd/gateway/main.go`
- [x] **Step 1: Write failing buildSink test**
Create `go/vehicle-gateway/cmd/gateway/main_test.go` if it does not exist, or add:
```go
func TestBuildSinkPrefersNATSWhenConfigured(t *testing.T) {
t.Setenv("NATS_URL", "nats://127.0.0.1:4222")
t.Setenv("KAFKA_BROKERS", "127.0.0.1:9092")
if got := chooseSinkMode(); got != "nats" {
t.Fatalf("sink mode = %q, want nats", got)
}
}
```
Run: `go test ./cmd/gateway -run TestBuildSinkPrefersNATSWhenConfigured -count=1`
Expected: FAIL because `chooseSinkMode` does not exist.
- [x] **Step 2: Implement mode choice and NATS sink branch**
In `cmd/gateway/main.go`, add:
```go
func chooseSinkMode() string {
if strings.TrimSpace(os.Getenv("NATS_URL")) != "" {
return "nats"
}
return "kafka"
}
```
At the top of `buildSink`, add:
```go
if chooseSinkMode() == "nats" {
sink, err := eventbus.NewNATSSink(eventbus.NATSConfig{
URL: env("NATS_URL", ""),
Stream: env("NATS_STREAM", "vehicle-ingest"),
RawSubjects: map[envelope.Protocol]string{
envelope.ProtocolGB32960: env("NATS_SUBJECT_GB32960_RAW", "vehicle.raw.gb32960.v1"),
envelope.ProtocolJT808: env("NATS_SUBJECT_JT808_RAW", "vehicle.raw.jt808.v1"),
envelope.ProtocolYutongMQTT: env("NATS_SUBJECT_YUTONG_MQTT_RAW", "vehicle.raw.yutong-mqtt.v1"),
},
UnifiedSubject: env("NATS_SUBJECT_UNIFIED", "vehicle.event.unified.v1"),
})
if err != nil {
return nil, err
}
logger.Info("nats jetstream sink enabled", "url", env("NATS_URL", ""), "stream", env("NATS_STREAM", "vehicle-ingest"))
return eventbus.NewAsyncSink(sink, eventbus.AsyncConfig{
QueueSize: envInt("NATS_ASYNC_QUEUE_SIZE", 100000),
Workers: envInt("NATS_ASYNC_WORKERS", 8),
OperationTimeout: time.Duration(envInt("NATS_PUBLISH_TIMEOUT_MS", 30000)) * time.Millisecond,
OnError: func(err error) {
logger.Warn("nats async publish failed", "error", err)
},
}), nil
}
```
- [x] **Step 3: Run gateway tests**
Run: `go test ./cmd/gateway -count=1`
Expected: PASS.
## Task 3: NATS to Kafka Bridge
**Files:**
- Create: `go/vehicle-gateway/cmd/nats-kafka-bridge/main.go`
- Create: `go/vehicle-gateway/cmd/nats-kafka-bridge/main_test.go`
- [x] **Step 1: Write failing bridge test**
Create a test with fake NATS messages and fake Kafka writer. It must prove that the bridge ACKs only after Kafka write succeeds:
```go
func TestBridgeAcksAfterKafkaWrite(t *testing.T) {
nats := &fakeNATSBatchReader{messages: []bridgeMessage{
{subject: "vehicle.raw.jt808.v1", data: []byte(`{"protocol":"JT808","phone":"13307795425"}`)},
}}
kafka := &fakeKafkaBatchWriter{}
bridge := bridgeProcessor{nats: nats, kafka: kafka, topics: defaultBridgeTopics()}
if err := bridge.processBatch(context.Background(), 10); err != nil {
t.Fatalf("processBatch() error = %v", err)
}
if kafka.writeCalls != 1 {
t.Fatalf("kafka write calls = %d, want 1", kafka.writeCalls)
}
if !nats.messages[0].acked {
t.Fatal("nats message was not acked")
}
}
```
Run: `go test ./cmd/nats-kafka-bridge -run TestBridgeAcksAfterKafkaWrite -count=1`
Expected: FAIL because bridge types do not exist.
- [x] **Step 2: Implement minimal bridge processor**
Implement small interfaces in `main.go`:
```go
type bridgeMessage struct {
subject string
data []byte
ack func() error
}
type bridgeNATSReader interface {
Fetch(context.Context, int) ([]bridgeMessage, error)
}
type bridgeKafkaWriter interface {
Write(context.Context, []kafka.Message) error
}
type bridgeProcessor struct {
nats bridgeNATSReader
kafka bridgeKafkaWriter
topics map[string]string
}
func (p bridgeProcessor) processBatch(ctx context.Context, batchSize int) error {
messages, err := p.nats.Fetch(ctx, batchSize)
if err != nil {
return err
}
kafkaMessages := make([]kafka.Message, 0, len(messages))
for _, msg := range messages {
topic, ok := p.topics[msg.subject]
if !ok {
return fmt.Errorf("no kafka topic for nats subject %s", msg.subject)
}
key := kafkaKeyFromEnvelope(msg.data)
kafkaMessages = append(kafkaMessages, kafka.Message{Topic: topic, Key: key, Value: msg.data})
}
if len(kafkaMessages) == 0 {
return nil
}
if err := p.kafka.Write(ctx, kafkaMessages); err != nil {
return err
}
for _, msg := range messages {
if msg.ack != nil {
if err := msg.ack(); err != nil {
return err
}
}
}
return nil
}
```
- [x] **Step 3: Implement real NATS and Kafka adapters**
Use `nats.PullSubscribe`, `Fetch`, `Ack`, and `kafka.Writer.WriteMessages`. The bridge must create or reuse a durable consumer named by `NATS_CONSUMER`, default `nats-kafka-bridge`.
- [x] **Step 4: Run bridge tests**
Run: `go test ./cmd/nats-kafka-bridge -count=1`
Expected: PASS.
## Task 4: NATS Server Deployment
**Files:**
- Create: `deploy/nats/nats-server.conf`
- [x] **Step 1: Create NATS config**
Create:
```conf
server_name: lingniu-nats-01
port: 4222
http_port: 8222
jetstream {
store_dir: "/data/nats/jetstream"
max_file_store: 20GB
max_mem_store: 512MB
}
```
- [x] **Step 2: Deploy on Kafka ECS**
On Kafka ECS, install or run `nats:2` with:
```bash
mkdir -p /opt/lingniu-nats /data/nats/jetstream
docker run -d --name lingniu-nats --restart unless-stopped \
-p 4222:4222 -p 8222:8222 \
-v /opt/lingniu-nats/nats-server.conf:/etc/nats/nats-server.conf:ro \
-v /data/nats:/data/nats \
nats:2 -c /etc/nats/nats-server.conf
```
- [x] **Step 3: Verify NATS**
Run from gateway ECS:
```bash
timeout 3 bash -c '</dev/tcp/172.17.111.56/4222'
```
Expected: exit 0.
## Task 5: ECS Cutover and Verification
**Files:**
- Modify ECS env files only.
- [x] **Step 1: Deploy binaries**
Build and upload: `gateway`, `nats-kafka-bridge`, `history-writer`, `stat-writer`, `realtime-api`.
- [x] **Step 2: Configure gateway**
Add to `/opt/lingniu-go-native/env/gateway.env`:
```bash
NATS_URL=nats://172.17.111.56:4222
NATS_STREAM=vehicle-ingest
NATS_ASYNC_QUEUE_SIZE=100000
NATS_ASYNC_WORKERS=8
NATS_PUBLISH_TIMEOUT_MS=30000
```
- [x] **Step 3: Configure bridge service**
Create `/opt/lingniu-go-native/env/nats-kafka-bridge.env` with:
```bash
NATS_URL=nats://172.17.111.56:4222
NATS_STREAM=vehicle-ingest
NATS_CONSUMER=nats-kafka-bridge
NATS_BATCH_SIZE=500
KAFKA_BROKERS=172.17.111.56:9092
KAFKA_TOPIC_GB32960_RAW=vehicle.raw.go.gb32960.v1
KAFKA_TOPIC_JT808_RAW=vehicle.raw.go.jt808.v1
KAFKA_TOPIC_YUTONG_MQTT_RAW=vehicle.raw.go.yutong-mqtt.v1
KAFKA_TOPIC_UNIFIED=vehicle.event.go.unified.v1
```
- [x] **Step 4: Verify production**
Run:
```bash
systemctl is-active lingniu-go-gateway.service lingniu-go-nats-kafka-bridge.service lingniu-go-history-writer.service lingniu-go-realtime-api.service
ss -Htan state established "( sport = :32960 )"
ss -Htan state established "( sport = :808 )" | wc -l
journalctl -u lingniu-go-gateway.service --since '2 minutes ago' --no-pager | grep -E 'ERROR|deadline|nats async publish failed'
journalctl -u lingniu-go-nats-kafka-bridge.service --since '2 minutes ago' --no-pager | grep -E 'ERROR|failed'
```
Expected:
- All services active.
- 32960 Recv-Q remains 0 or near 0.
- 808 connections remain stable.
- No gateway publish deadline errors.
- Kafka downstream consumers continue updating TDengine and Redis.
## Self-Review
- Spec coverage: gateway NATS publish, bridge to Kafka, ECS NATS deployment, and production verification are covered.
- Scope intentionally excludes rewriting history/stat/realtime to consume NATS directly; they remain Kafka consumers in this phase.
- No placeholders remain; later downlink `notice/reply` subjects are intentionally deferred because they are not needed for first reliable ingest cutover.

View File

@@ -0,0 +1,336 @@
# Storage Simplification Implementation Plan
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
**Goal:** 精简车辆接入落库模型,去掉重复持久化,同时保持 raw 可回放、实时可查、历史位置和日指标可用。
**Architecture:** TDengine 只保留 raw 证据和位置历史两类高写入时间序列MySQL 只保留身份映射、实时当前态和日指标。先兼容 API再停止重复写入最后在生产验证后删除旧表/旧列。
**Tech Stack:** Go, TDengine, MySQL, Redis, Kafka, NATS JetStream, `go test`.
---
## File Structure
- Modify: `go/vehicle-gateway/internal/history/schema.go`
- Stop creating `vehicle_mileage_points` after API compatibility is in place.
- Modify: `go/vehicle-gateway/internal/history/writer.go`
- Stop calling `AppendMileagePoint`; keep total mileage in `vehicle_locations`.
- Modify: `go/vehicle-gateway/internal/history/query.go`
- Make `MileagePointRepository` read from `vehicle_locations` with `total_mileage_km IS NOT NULL`.
- Modify: `go/vehicle-gateway/internal/history/query_test.go`
- Update mileage point query expectations to `vehicle_locations`.
- Modify: `go/vehicle-gateway/internal/history/writer_test.go`
- Prove no `vehicle_mileage_points` stable/table is created or written.
- Modify: `go/vehicle-gateway/internal/history/schema.go`
- In a later task, remove `fields_json` from new `raw_frames` schema.
- Modify: `go/vehicle-gateway/internal/history/writer.go`
- In a later task, stop writing `fields_json`.
- Modify: `go/vehicle-gateway/internal/history/query.go`
- In a later task, make raw query tolerate schemas without `fields_json`.
- Modify: `docs/architecture/storage-minimal-contract.md`
- Keep the target storage contract updated as implementation lands.
### Task 1: Move mileage point queries onto `vehicle_locations`
**Files:**
- Modify: `go/vehicle-gateway/internal/history/query.go`
- Modify: `go/vehicle-gateway/internal/history/query_test.go`
- [x] **Step 1: Write the failing test**
Update `TestMileagePointHandlerReturnsMileageByVehicleKey` in `go/vehicle-gateway/internal/history/query_test.go` so it expects `vehicle_locations`, not `vehicle_mileage_points`:
```go
mock.ExpectQuery("SELECT COUNT\\(\\*\\) FROM lingniu_vehicle_ts.vehicle_locations").
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(17))
mock.ExpectQuery("SELECT ts, event_id, frame_id, received_at, total_mileage_km, speed_kmh, longitude, latitude, protocol, vehicle_key, vin, phone, device_id FROM lingniu_vehicle_ts.vehicle_locations").
WillReturnRows(sqlmock.NewRows([]string{
"ts", "event_id", "frame_id", "received_at", "total_mileage_km", "speed_kmh", "longitude", "latitude",
"protocol", "vehicle_key", "vin", "phone", "device_id",
}).AddRow("2026-07-02 10:00:00.000", "event-1", "frame-1", "2026-07-02 10:00:01.000", 8792.8, 8.0, 121.07764, 31.22436, "JT808", "JT808:13307811350", "VIN001", "13307811350", "dev001"))
```
- [x] **Step 2: Run test to verify it fails**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/history -run 'TestMileagePointHandlerReturnsMileageByVehicleKey|TestBuildMileagePointSQLUsesLiteralsForTDengine' -count=1
```
Expected: FAIL because `MileagePointRepository.tableName()` still returns `vehicle_mileage_points`.
- [x] **Step 3: Point mileage repository at location table**
Change `MileagePointRepository.tableName()` in `go/vehicle-gateway/internal/history/query.go`:
```go
func (r *MileagePointRepository) tableName() string {
if r.database == "" {
return "vehicle_locations"
}
return r.database + ".vehicle_locations"
}
```
Change `mileagePointWhere` so it filters only rows with total mileage:
```go
func mileagePointWhere(query MileagePointQuery) []string {
where := []string{"total_mileage_km IS NOT NULL"}
add := func(condition string) {
where = append(where, condition)
}
if query.Protocol != "" {
add("protocol = '" + quote(query.Protocol) + "'")
}
if query.VehicleKey != "" {
add("vehicle_key = '" + quote(query.VehicleKey) + "'")
}
if query.VIN != "" {
add("vin = '" + quote(query.VIN) + "'")
}
if query.Phone != "" {
add("phone = '" + quote(query.Phone) + "'")
}
if query.DeviceID != "" {
add("device_id = '" + quote(query.DeviceID) + "'")
}
if query.DateFrom != "" {
add("ts >= '" + quote(normalizeDateTimeLiteral(query.DateFrom)) + "'")
}
if query.DateTo != "" {
add("ts <= '" + quote(normalizeDateTimeLiteral(query.DateTo)) + "'")
}
return where
}
```
- [x] **Step 4: Run tests**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/history -count=1
```
Expected: PASS.
- [x] **Step 5: Commit**
```bash
git add go/vehicle-gateway/internal/history/query.go go/vehicle-gateway/internal/history/query_test.go
git commit -m "refactor(go): read mileage points from locations"
```
### Task 2: Stop writing duplicate mileage point table
**Files:**
- Modify: `go/vehicle-gateway/internal/history/schema.go`
- Modify: `go/vehicle-gateway/internal/history/writer.go`
- Modify: `go/vehicle-gateway/internal/history/writer_test.go`
- [x] **Step 1: Write the failing tests**
Update `go/vehicle-gateway/internal/history/writer_test.go`:
```go
for _, want := range []string{"CREATE DATABASE IF NOT EXISTS test_ts", "raw_frames", "vehicle_locations"} {
if got := countSQL(exec.calls, want); got == 0 {
t.Fatalf("expected schema statement containing %q", want)
}
}
if got := countSQL(exec.calls, "vehicle_mileage_points"); got != 0 {
t.Fatalf("schema should not create vehicle_mileage_points, got %d", got)
}
if got := countSQL(exec.calls, "INSERT INTO mil_"); got != 0 {
t.Fatalf("history writer should not insert duplicate mileage point rows, got %d", got)
}
```
- [x] **Step 2: Run test to verify it fails**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/history -run 'TestWriter' -count=1
```
Expected: FAIL because schema and writer still create/write `vehicle_mileage_points`.
- [x] **Step 3: Remove duplicate write path**
Change `AppendAll` in `go/vehicle-gateway/internal/history/writer.go`:
```go
func (w *Writer) AppendAll(ctx context.Context, env envelope.FrameEnvelope) error {
if err := w.AppendRawFrame(ctx, env); err != nil {
return err
}
return w.AppendLocation(ctx, env)
}
```
Remove `vehicle_mileage_points` from `SchemaStatements` in `go/vehicle-gateway/internal/history/schema.go`. Keep `AppendMileagePoint` only if tests still use it directly; otherwise remove the function and its helper after compile errors guide cleanup.
- [x] **Step 4: Run tests**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/history ./cmd/history-writer -count=1
```
Expected: PASS.
- [x] **Step 5: Commit**
```bash
git add go/vehicle-gateway/internal/history/schema.go go/vehicle-gateway/internal/history/writer.go go/vehicle-gateway/internal/history/writer_test.go
git commit -m "refactor(go): stop writing duplicate mileage history"
```
### Task 3: Remove `fields_json` from new raw writes
**Files:**
- Modify: `go/vehicle-gateway/internal/history/schema.go`
- Modify: `go/vehicle-gateway/internal/history/writer.go`
- Modify: `go/vehicle-gateway/internal/history/writer_test.go`
- Modify: `go/vehicle-gateway/internal/history/query.go`
- Modify: `go/vehicle-gateway/internal/history/query_test.go`
- [x] **Step 1: Write failing writer test**
Add this assertion in the raw insert test:
```go
rawInsert := findSQL(exec.calls, "INSERT INTO raw_")
if strings.Contains(rawInsert, "fields_json") {
t.Fatalf("raw insert should not write duplicate fields_json: %s", rawInsert)
}
```
- [x] **Step 2: Run test to verify it fails**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/history -run 'TestWriter' -count=1
```
Expected: FAIL because raw insert still includes `fields_json`.
- [x] **Step 3: Remove `fields_json` from schema and writer**
In `go/vehicle-gateway/internal/history/schema.go`, remove:
```sql
fields_json BINARY(4096),
```
In `go/vehicle-gateway/internal/history/writer.go`, remove the `fieldsJSON` variable and `fields_json` insert column. Keep `parsed_json` complete.
- [x] **Step 4: Make raw query backward compatible**
Query only columns that exist in both old and new schemas. Old production tables may still contain `fields_json`, but the API no longer selects or returns it:
```go
func buildRawFrameSQL(table string, query RawFrameQuery) (string, []any) {
sqlText := `SELECT ts, frame_id, event_id, message_id, event_time, received_at, raw_size_bytes, raw_hex, raw_text, parsed_json, parse_status, parse_error, source_endpoint, protocol, vehicle_key, vin, phone, device_id FROM ` + table
where := rawFrameWhere(query)
if len(where) > 0 {
sqlText += " WHERE " + strings.Join(where, " AND ")
}
sqlText += " ORDER BY " + rawFrameOrderColumn(query.OrderBy) + " DESC LIMIT " + strconv.Itoa(query.Limit) + " OFFSET " + strconv.Itoa(query.Offset)
return sqlText, nil
}
```
The new-schema test scans a row without `fields_json` and asserts the JSON response omits `fields_json`. This also works on old schemas because selecting a subset of columns is valid when the table has extra columns.
- [x] **Step 5: Run tests**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/history ./cmd/history-writer ./cmd/realtime-api -count=1
```
Expected: PASS.
- [x] **Step 6: Commit**
```bash
git add go/vehicle-gateway/internal/history/schema.go go/vehicle-gateway/internal/history/writer.go go/vehicle-gateway/internal/history/writer_test.go go/vehicle-gateway/internal/history/query.go go/vehicle-gateway/internal/history/query_test.go
git commit -m "refactor(go): remove duplicate raw fields json"
```
### Task 4: Production migration and cleanup gate
**Files:**
- Modify: `docs/architecture/storage-minimal-contract.md`
- Modify: `docs/ops/vehicle-ingest-runbook.md`
- [x] **Step 1: Deploy the code changes**
Built Linux amd64 binaries for `history-writer` and `realtime-api`, uploaded them to the current ECS release, and restarted:
```bash
systemctl restart lingniu-go-history-writer.service
systemctl restart lingniu-go-realtime-api.service
```
- [x] **Step 2: Verify service health**
Run on ECS:
```bash
for port in 20212 20200; do
curl -fsS "http://127.0.0.1:${port}/readyz"
echo
done
curl -fsS http://127.0.0.1:20212/metrics | grep vehicle_history_kafka_lag
curl -fsS http://127.0.0.1:20212/metrics | grep vehicle_history_writes_total
```
Expected:
- both readiness endpoints return `status=ok`
- history lag drains to `0` or remains bounded
- history writes continue increasing
- [x] **Step 3: Verify no new duplicate mileage table writes**
Because the system is not live yet and historical data can be discarded, the TDengine history database was dropped and recreated from the new schema:
```sql
DROP DATABASE IF EXISTS lingniu_vehicle_ts;
DESCRIBE lingniu_vehicle_ts.raw_frames;
SHOW lingniu_vehicle_ts.STABLES;
```
Observed stables after recreation: `raw_frames`, `raw_frame_payload_chunks`, `vehicle_locations`. `vehicle_mileage_points` is gone.
- [x] **Step 4: Drop old table only after observation**
Old history data was intentionally discarded before launch. No separate observation window is required for `vehicle_mileage_points`.
- [x] **Step 5: Commit docs update**
```bash
git add docs/architecture/storage-minimal-contract.md docs/ops/vehicle-ingest-runbook.md
git commit -m "docs: record storage simplification migration"
```
## Self-Review
- Spec coverage: The plan addresses minimal storage, duplicate mileage storage, raw field duplication, production verification, and delayed destructive cleanup.
- Red-flag scan: The plan has no open-ended task. Task 3 specifies the compatibility option and the new/old schema test expectations.
- Type consistency: The plan keeps existing public API types and changes only their backing tables.

View File

@@ -0,0 +1,550 @@
# 100K Vehicle Ingest Hardening Implementation Plan
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
**Goal:** Move the Go vehicle ingest platform toward stable production operation for 100,000 vehicles.
**Architecture:** Keep the current Gateway -> NATS -> Kafka -> TDengine/Redis/MySQL architecture, but make capacity explicit, add reproducible load testing, harden OS/systemd/network limits, and remove single-threaded or per-message bottlenecks from storage projectors. The first target is predictable behavior under high connection count and high frame rate; horizontal scaling comes after measurable single-node limits.
**Tech Stack:** Go 1.26, systemd, Linux TCP tuning, NATS JetStream, Kafka, Redis, TDengine, MySQL, Prometheus-style `/metrics`.
---
## Current Baseline
Observed on 2026-07-03 from ECS `115.29.187.205`:
| Item | Current Observation |
| --- | --- |
| Gateway active connections | JT808 about `242`, GB32960 about `2` |
| ECS load | about `0.09 / 0.15 / 0.21` |
| Memory | about `7.5 GB total`, `6.6 GB available` |
| Disk | `/` about `37%` used |
| NATS bridge ack pending | `0` |
| NATS bridge pending | about `26` |
| Kafka lag | `0` across history/realtime/stat sampled metrics |
| Redis projector latency | about `0 ms` gauge |
| MySQL projector latency | about `1-2 ms` gauge |
| Gateway file limit | `1048576` |
| Gateway max processes | `30123` |
| Kernel `net.core.somaxconn` | `128` |
| Gateway listen backlog | Go `net.Listen` default, bounded by `somaxconn` |
| Gateway default max connections | env default adjusted to `TCP_MAX_CONNECTIONS=120000` |
Key gap: production is healthy at current load, but it has not proved 100K connection or high burst behavior. The current OS accept backlog and gateway default connection cap are below the target.
## Capacity Assumptions
Use these as starting targets. Revise after load test evidence.
| Scenario | Target |
| --- | --- |
| Connected vehicles | `100,000` total |
| Average frame interval | 10-30 seconds depending protocol |
| Average ingress FPS | `3,000-10,000 frames/s` |
| Burst ingress FPS | `20,000 frames/s` for short periods |
| Gateway p99 parse + enqueue | `< 50 ms` |
| Gateway publish queue drops | `0` under target load |
| NATS ack pending | sustained `0`, short spikes acceptable |
| Kafka lag | bounded and decreasing after burst |
| Redis current-state lag | `< 3 seconds` under normal load |
| TDengine history lag | bounded and observable |
| MySQL snapshot lag | can be lower priority and async; must not block Redis |
## File Map
| File | Responsibility |
| --- | --- |
| `docs/superpowers/plans/2026-07-03-100k-vehicle-ingest-hardening.md` | This implementation plan |
| `docs/ops/go-vehicle-ingest-memory.md` | Operational memory, update after each production change |
| `docs/ops/vehicle-ingest-runbook.md` | Production runbook, add capacity and tuning commands |
| `go/vehicle-gateway/cmd/load-sim/main.go` | New load simulator for JT808/GB32960 TCP and MQTT-like payload replay |
| `go/vehicle-gateway/internal/loadsim/*` | New reusable load simulation helpers |
| `go/vehicle-gateway/internal/gateway/tcp_server.go` | Gateway connection handling, metrics, connection admission |
| `go/vehicle-gateway/cmd/gateway/main.go` | Gateway env defaults for 100K mode |
| `go/vehicle-gateway/internal/eventbus/async_sink.go` | Async publish queue metrics/backpressure behavior |
| `go/vehicle-gateway/cmd/nats-fast-writer/main.go` | Fast writer batching and worker behavior |
| `go/vehicle-gateway/cmd/history-writer/main.go` | Kafka/TDengine consume/write batching |
| `go/vehicle-gateway/internal/history/writer.go` | TDengine write path; later batch insert |
| `go/vehicle-gateway/cmd/realtime-api/main.go` | Redis-first and MySQL async projector behavior |
| `deploy/systemd/*.conf` or `docs/ops/*.md` | Documented sysctl/systemd tuning until deployment templates exist |
## Task 1: Capacity Baseline and Production Guardrails
**Files:**
- Modify: `docs/ops/go-vehicle-ingest-memory.md`
- Modify: `docs/ops/vehicle-ingest-runbook.md`
- Create: `docs/ops/100k-capacity-baseline.md`
- [x] **Step 1: Create the capacity baseline document**
Create `docs/ops/100k-capacity-baseline.md` with:
```markdown
# 100K 车辆接入容量基线
更新时间2026-07-03
## 目标
支撑 100,000 台车辆接入,入口服务在高连接数和高帧率下保持可观测、可背压、可恢复。
## 初始容量假设
| 项 | 目标 |
| --- | --- |
| 连接车辆数 | 100,000 |
| 平均帧间隔 | 10-30 秒 |
| 平均入口 FPS | 3,000-10,000 |
| 短时 burst FPS | 20,000 |
| Gateway p99 parse + enqueue | < 50ms |
| Redis 当前态延迟 | < 3s |
| TDengine 历史写入 lag | 可观测且 burst 后下降 |
| MySQL 当前态 | 异步,不能阻塞 Redis |
## 当前生产观测
当前 ECS `115.29.187.205` 在低负载下健康:
- JT808 连接约 242。
- GB32960 连接约 2。
- Kafka lag 为 0。
- NATS ack pending 为 0。
- Redis 写入约 0ms。
- MySQL 投影约 1-2ms。
## 当前已知缺口
- `net.core.somaxconn=128`,无法作为 100K 连接生产基线。
- Gateway 默认 `TCP_MAX_CONNECTIONS` 已调整为 `120000`,生产仍可通过环境变量覆盖。
- 已新增可重复的 TCP 连接压测工具,后续需要用它跑 10K/50K/100K 阶段测试并记录结果。
- 仍缺按协议维度的连接生命周期、读超时、parse/publish p95/p99 直方图。
- TDengine history writer 当前逐消息插入,后续需要批量写入。
- Kafka topic 当前 12 分区100K 目标下需要结合实际 FPS 再评估分区数。
```
- [x] **Step 2: Update runbook with capacity baseline link**
Add this line near the top of `docs/ops/vehicle-ingest-runbook.md` after the production inventory link:
```markdown
10W 车辆容量目标、当前缺口和压测口径见 [100K 车辆接入容量基线](100k-capacity-baseline.md)。
```
- [x] **Step 3: Verify docs render**
Run:
```bash
rg -n "100K|10W|容量基线|somaxconn|TCP_MAX_CONNECTIONS" docs/ops docs/architecture
```
Expected: the new baseline and runbook link are found.
- [ ] **Step 4: Commit**
```bash
git add docs/ops/100k-capacity-baseline.md docs/ops/vehicle-ingest-runbook.md docs/ops/go-vehicle-ingest-memory.md
git commit -m "docs: add 100k ingest capacity baseline"
```
## Task 2: Load Simulator Skeleton
**Files:**
- Create: `go/vehicle-gateway/internal/loadsim/config.go`
- Create: `go/vehicle-gateway/internal/loadsim/config_test.go`
- Create: `go/vehicle-gateway/internal/loadsim/templates.go`
- Create: `go/vehicle-gateway/internal/loadsim/template_test.go`
- Create: `go/vehicle-gateway/internal/loadsim/schedule.go`
- Create: `go/vehicle-gateway/internal/loadsim/schedule_test.go`
- Create: `go/vehicle-gateway/internal/loadsim/runner.go`
- Create: `go/vehicle-gateway/internal/loadsim/runner_test.go`
- Create: `go/vehicle-gateway/cmd/load-sim/main.go`
- Create: `go/vehicle-gateway/cmd/load-sim/main_test.go`
- [x] **Step 1: Write failing tests for loadsim config, templates, schedule, runner and CLI stats**
Implemented tests:
- `TestConfigFromFlagsParsesCapacityKnobs`
- `TestConfigFromFlagsRejectsUnsafeValues`
- `TestFrameTemplateReturnsParsableJT808LocationFrame`
- `TestFrameTemplateReturnsParsableGB32960RealtimeFrame`
- `TestConnectionBatchesSpreadsConnectionsByRate`
- `TestRunnerDialsTargetConnectionsAndWritesFrames`
- `TestFormatStatsIncludesCapacityCounters`
- [x] **Step 2: Run tests to verify RED**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/loadsim
```
Expected: FAIL because package or symbols do not exist.
- [x] **Step 3: Implement minimal load simulator**
Implemented:
- flag config validation
- parsable JT808 0x0200 and GB32960 realtime frame templates
- connection batch scheduling
- TCP runner with dial injection for tests
- `cmd/load-sim` CLI
- [x] **Step 4: Run test to verify GREEN**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/loadsim
```
Expected: PASS.
- [x] **Step 5: Add CLI skeleton**
Implemented `go/vehicle-gateway/cmd/load-sim/main.go`.
- [x] **Step 6: Verify CLI builds**
Run:
```bash
cd go/vehicle-gateway
go test ./cmd/load-sim ./internal/loadsim
```
Expected: PASS.
- [ ] **Step 7: Commit**
```bash
git add go/vehicle-gateway/internal/loadsim go/vehicle-gateway/cmd/load-sim
git commit -m "test(go): add vehicle ingest load simulator skeleton"
```
## Task 3: Gateway Capacity Configuration
**Files:**
- Modify: `go/vehicle-gateway/cmd/gateway/main.go`
- Modify: `docs/ops/100k-capacity-baseline.md`
- Modify: `docs/ops/vehicle-ingest-runbook.md`
- [x] **Step 1: Write test for 100K connection env default**
Added `TestGatewayDefaultsTo100KConnectionCeiling`.
- [x] **Step 2: Run test to verify RED before changing default**
Run:
```bash
cd go/vehicle-gateway
go test ./cmd/gateway
```
Expected: FAIL while call site still used `20000`.
- [x] **Step 3: Change gateway default**
In `go/vehicle-gateway/cmd/gateway/main.go`, change the TCP server config default:
```go
MaxConnections: envInt("TCP_MAX_CONNECTIONS", 120_000),
```
- [x] **Step 4: Document required sysctl**
Add to `docs/ops/100k-capacity-baseline.md`:
```markdown
## ECS OS 参数建议
100K 连接目标需要至少以下系统参数作为起点:
```bash
sysctl -w net.core.somaxconn=65535
sysctl -w net.ipv4.tcp_max_syn_backlog=65535
sysctl -w net.ipv4.ip_local_port_range="10000 65000"
sysctl -w net.ipv4.tcp_tw_reuse=1
```
systemd gateway 已配置 `LimitNOFILE=1048576`,需要持续保持。
```
```
- [x] **Step 5: Run tests**
Run:
```bash
cd go/vehicle-gateway
go test ./cmd/gateway ./internal/gateway
```
Expected: PASS.
- [ ] **Step 6: Commit**
```bash
git add go/vehicle-gateway/cmd/gateway/main.go go/vehicle-gateway/cmd/gateway/main_test.go docs/ops/100k-capacity-baseline.md docs/ops/vehicle-ingest-runbook.md
git commit -m "chore(go): raise gateway connection target"
```
## Task 4: Gateway Runtime Metrics for 100K Operations
**Files:**
- Modify: `go/vehicle-gateway/internal/gateway/tcp_server.go`
- Modify: `go/vehicle-gateway/internal/metrics/metrics.go`
- Test: `go/vehicle-gateway/internal/gateway/tcp_server_test.go`
- Test: `go/vehicle-gateway/internal/metrics/metrics_test.go`
- [x] **Step 1: Add metrics test for rejected connections**
Added `TestTCPServerRecordsConnectionRejectionMetric` in `go/vehicle-gateway/internal/gateway/tcp_server_test.go`.
- [x] **Step 2: Run RED or confirm helper coverage**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/gateway -run TestTCPServerRecordsConnectionRejectionMetric
```
Expected: FAIL before `recordConnectionRejection` exists.
- [x] **Step 3: Record max connection rejections**
In `go/vehicle-gateway/internal/gateway/tcp_server.go`, when the semaphore default case rejects a connection, add:
```go
server.recordConnectionRejection("max_connections")
```
Implemented `recordConnectionRejection`.
- [x] **Step 4: Run gateway tests**
Run:
```bash
cd go/vehicle-gateway
go test ./internal/gateway ./internal/metrics
```
Expected: PASS.
- [ ] **Step 5: Commit**
```bash
git add go/vehicle-gateway/internal/gateway/tcp_server.go go/vehicle-gateway/internal/gateway/tcp_server_test.go go/vehicle-gateway/internal/metrics/metrics.go go/vehicle-gateway/internal/metrics/metrics_test.go
git commit -m "feat(go): expose gateway connection rejection metrics"
```
## Task 5: Redis-First Realtime Projector Safety
**Files:**
- Modify: `go/vehicle-gateway/cmd/realtime-api/main.go`
- Test: `go/vehicle-gateway/cmd/realtime-api/main_test.go`
- [x] **Step 1: Add test proving MySQL queue drops do not fail Redis update**
Add a test around `asyncSecondaryRealtimeUpdater` in `go/vehicle-gateway/cmd/realtime-api/main_test.go` using existing test patterns. The test should:
1. Create an updater with queue size `1`.
2. Fill the queue.
3. Call `Update` with a new envelope.
4. Assert `Update` returns `nil`.
5. Assert metric `vehicle_realtime_async_queue_total{status="dropped"}` increments.
- [x] **Step 2: Run test**
Run:
```bash
cd go/vehicle-gateway
go test ./cmd/realtime-api -run TestAsyncSecondaryQueueDropDoesNotFailPrimaryUpdate
```
Actual: PASS. The behavior already existed; this test locks it as regression coverage.
- [x] **Step 3: Ensure queue drop is observable and non-fatal**
In `go/vehicle-gateway/cmd/realtime-api/main.go`, preserve current Redis-first behavior:
- Redis update remains synchronous and primary.
- MySQL update remains async secondary.
- Full secondary queue increments dropped metric and does not return error to Kafka processing.
- [x] **Step 4: Run tests**
```bash
cd go/vehicle-gateway
go test ./cmd/realtime-api ./internal/realtime
```
- [ ] **Step 5: Commit**
```bash
git add go/vehicle-gateway/cmd/realtime-api/main.go go/vehicle-gateway/cmd/realtime-api/main_test.go
git commit -m "test(go): lock redis-first realtime projector behavior"
```
## Task 6: TDengine History Writer Batching Design
**Files:**
- Create: `docs/architecture/tdengine-batch-writer-design.md`
- Later Modify: `go/vehicle-gateway/cmd/history-writer/main.go`
- Later Modify: `go/vehicle-gateway/internal/history/writer.go`
- [x] **Step 1: Document the batch writer decision**
Created `docs/architecture/tdengine-batch-writer-design.md`.
- [ ] **Step 2: Commit design before code**
```bash
git add docs/architecture/tdengine-batch-writer-design.md
git commit -m "docs: design tdengine batch writer"
```
## Task 7: Production Sysctl and Service Guardrail Rollout
**Files:**
- Modify: `docs/ops/vehicle-ingest-runbook.md`
- Modify: `docs/ops/go-vehicle-ingest-memory.md`
- [x] **Step 1: Apply sysctl changes on ECS**
Run on ECS:
```bash
cat >/etc/sysctl.d/zz-lingniu-vehicle-ingest.conf <<'EOF'
net.core.somaxconn = 65535
net.ipv4.tcp_max_syn_backlog = 65535
net.ipv4.ip_local_port_range = 10000 65000
net.ipv4.tcp_tw_reuse = 1
EOF
sysctl --system
```
- [x] **Step 2: Verify sysctl**
Run:
```bash
sysctl net.core.somaxconn net.ipv4.tcp_max_syn_backlog net.ipv4.ip_local_port_range net.ipv4.tcp_tw_reuse
```
Expected:
```text
net.core.somaxconn = 65535
net.ipv4.tcp_max_syn_backlog = 65535
net.ipv4.ip_local_port_range = 10000 65000
net.ipv4.tcp_tw_reuse = 1
```
- [x] **Step 3: Restart gateway**
Run:
```bash
systemctl restart lingniu-go-gateway.service
curl -fsS http://127.0.0.1:20211/readyz
ss -ltnp | grep -E ':(808|32960|20211) '
```
- [x] **Step 4: Document rollout**
Add the applied sysctl values and timestamp to `docs/ops/go-vehicle-ingest-memory.md`.
Actual release: `/opt/lingniu-go-native/releases/100k-hardening-20260703175730`.
- [ ] **Step 5: Commit**
```bash
git add docs/ops/go-vehicle-ingest-memory.md docs/ops/vehicle-ingest-runbook.md
git commit -m "ops: document 100k gateway sysctl rollout"
```
## Task 8: First 10K Connection Test
**Files:**
- Modify: `docs/ops/100k-capacity-baseline.md`
- Modify: `go/vehicle-gateway/cmd/load-sim/main.go`
- Modify: `go/vehicle-gateway/internal/loadsim/*`
- [x] **Step 1: Implement JT808 connection hold mode**
Extended load-sim with `-send=false` to open N TCP connections and keep them open for `duration`, without sending frames.
Expected CLI:
```bash
go run ./cmd/load-sim -protocol jt808 -addr 115.29.187.205:808 -connections 10000 -send-interval 30s -duration 5m -send=false
```
- [x] **Step 2: Run from non-production client**
Run a small smoke first:
```bash
go run ./cmd/load-sim -protocol jt808 -addr 115.29.187.205:808 -connections 100 -duration 30s -send=false
```
Actual: ran on ECS loopback `127.0.0.1:808` to avoid external network limits and avoid business frame pollution.
- [x] **Step 3: Capture metrics during test**
On ECS:
```bash
curl -fsS http://127.0.0.1:20211/metrics | grep vehicle_gateway_active_connections
uptime
free -m
ss -tn sport = :808 | wc -l
journalctl -u lingniu-go-gateway.service --since '10 minutes ago' --no-pager | grep -Ei 'error|failed|panic|fatal' || true
```
- [x] **Step 4: Update capacity baseline**
Record:
- max active connections reached
- CPU/load
- memory
- connection rejection count
- errors
Actual: recorded 100 / 1,000 / 10,000 / 50,000 / 100,000 total hold-only results in `docs/ops/100k-capacity-baseline.md`.
- [ ] **Step 5: Commit**
```bash
git add docs/ops/100k-capacity-baseline.md go/vehicle-gateway/cmd/load-sim go/vehicle-gateway/internal/loadsim
git commit -m "test(go): record 10k gateway connection baseline"
```
## Self-Review
Spec coverage:
- 10W production objective: covered by capacity assumptions and phased tasks.
- Gateway connection limit and OS tuning: Tasks 3, 4, 7, 8.
- NATS/Kafka stability: baseline metrics and future batching notes; bridge-specific tuning remains a follow-up after first load test.
- Redis/MySQL projector stability: Task 5.
- TDengine write bottleneck: Task 6 sets design; implementation should follow once load evidence shows the threshold.
- ECS operations memory: Tasks 1 and 7.
Known intentional gaps:
- Full 100K test is not first. The first safe milestone is 10K connection test, then 50K, then 100K.
- Multi-ECS horizontal sharding is not implemented in this plan. It should follow after single-node evidence.
- Kafka partition changes are not included yet; changing partitions without measured FPS and consumer bottlenecks would be premature.

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -1,481 +0,0 @@
> **Superseded:** This 2026-06-23 DuckDB-based design is historical context.
> Use `docs/target-architecture.md` and
> `docs/superpowers/specs/2026-06-29-vehicle-ingest-redesign.md` for the
> current production architecture: TDengine is the default hot history store,
> `event-file-store` and Xinda Push have been removed from the production codebase.
# GB32960 Production Readiness Design
## Goal
Make the GB32960 ingestion, history, and analytics services production-ready:
- receive and parse GB32960 traffic quickly and accurately;
- maximize VIN-to-platform/vendor-profile association through platform login and local mappings;
- store complete RAW frames and queryable telemetry history with DuckDB;
- support full frame replay from persisted RAW bytes;
- compute real-time daily vehicle metrics and alarm timelines into MySQL;
- reduce service coupling and improve cohesion inside each module.
## Non-Goals
- Do not put database credentials in code, YAML files, tests, docs, or commits.
- Do not make MySQL the source for full telemetry history. MySQL is for derived daily metrics and alarm timelines.
- Do not require every private GB32960 extension to be perfect before storing data. RAW bytes must always be kept when a valid frame is received.
## Production Architecture
```mermaid
flowchart LR
tcp["GB32960 TCP:32960"] --> ingest["gb32960-ingest-app"]
mapping["local vin-platform-profile.jsonl"] --> ingest
ingest --> kafkaEvent["Kafka vehicle.event.gb32960.v1"]
ingest --> kafkaRaw["Kafka vehicle.raw.gb32960.v1"]
kafkaEvent --> history["vehicle-history-app"]
kafkaRaw --> history
history --> duckdb["DuckDB hot history"]
history --> raw["RAW frame archive"]
kafkaEvent --> analytics["vehicle-analytics-app"]
analytics --> mysql["MySQL daily stats + alarms"]
```
## Service Boundaries
### 1. `gb32960-ingest-app`
Responsibilities:
- Netty TCP accept, frame splitting, GB32960 decode, auth, ACK, and Kafka production.
- Platform login handling and connection-local `platformAccount`.
- VIN-to-platform/vendor-profile resolution.
- Produce structured telemetry, session, alarm, diagnostics, and RAW envelope messages.
It must not:
- write DuckDB, MySQL, or local archive files;
- run historical queries;
- run daily statistics.
ACK policy:
- login/report-like GB32960 commands use durable ACK: ACK only after Kafka dispatch succeeds;
- malformed bytes that cannot form a valid frame are dropped with diagnostics;
- valid frames with partial private-block parse failures still go to Kafka with `parseStatus=PARTIAL`.
VIN/profile resolution priority:
1. active platform login on the current TCP channel;
2. exact VIN entry from local mapping file;
3. VIN prefix entry from local mapping file;
4. configured platform extension rule;
5. standard GB32960 profile.
Local mapping file format:
```jsonl
{"vin":"LNXNEGRR2SR321390","platformAccount":"Hyundai","vendorProfile":"guangdong-fc","enabled":true}
{"vinPrefix":"LNXNEGRR","platformAccount":"Hyundai","vendorProfile":"guangdong-fc","enabled":true}
```
The mapping file is loaded at startup and can be reloaded by an actuator endpoint or file watcher in a later phase.
### 2. `vehicle-history-app`
Responsibilities:
- Consume Kafka event and RAW topics.
- Persist RAW bytes to an append-only archive.
- Persist query indexes and compact telemetry points to DuckDB.
- Serve high-performance query APIs:
- full frame list by `vin + eventTime`;
- telemetry fields by `vin + eventTime`;
- one-frame replay by `rawArchiveUri`;
- parse diagnostics and missing-profile troubleshooting.
It must not:
- listen on GB32960 TCP;
- run MySQL statistics;
- mutate daily business metrics.
### 3. `vehicle-analytics-app`
Responsibilities:
- Consume Kafka event topic.
- Maintain real-time daily stats in MySQL.
- Maintain full alarm timelines in MySQL.
- Expose metrics query APIs and operational health.
It must not:
- parse RAW frames for normal operation;
- write DuckDB history;
- own GB32960 TCP connectivity.
## Kafka Contract
Keep Protobuf `VehicleEnvelope`, but make the producer contract stricter:
- `eventId`: stable idempotency key, deterministic from protocol, VIN, command, event time, raw checksum, and sequence when available.
- `vin`: required for vehicle frames; platform-only commands use `_platform` or empty plus `platformAccount`.
- `eventTimeMs`: GB32960 data collection time when available.
- `ingestTimeMs`: service receive time.
- `metadata.platformAccount`: resolved platform account.
- `metadata.vendorProfile`: selected parser profile.
- `metadata.parseStatus`: `OK`, `PARTIAL`, or `FAILED`.
- `metadata.parseErrorCode`: stable code for failures.
- `raw_archive.data`: present on RAW topic only; event topic keeps `rawArchiveUri` pointer.
- `telemetry_snapshot`: standard field projection for downstream services.
Consumer rule:
- History consumes both event and RAW topics.
- Analytics consumes only event topic.
- Consumer offset is committed only after the target store commit succeeds.
- Reprocessing must be safe through deterministic ids and database upsert/ignore semantics.
## DuckDB History Design
Use DuckDB as the hot historical database, not a Parquet-rewrite sidecar.
DuckDB official guidance says row-by-row insert loops are inefficient for bulk loading, and Appender/batched writes are the intended high-throughput path. Therefore history writes use a single writer worker with bounded batches and one DuckDB connection.
### Tables
`gb32960_frame_index`
```sql
CREATE TABLE IF NOT EXISTS gb32960_frame_index (
event_id VARCHAR PRIMARY KEY,
vin VARCHAR NOT NULL,
platform_account VARCHAR,
vendor_profile VARCHAR,
command VARCHAR NOT NULL,
command_code INTEGER NOT NULL,
event_time TIMESTAMP NOT NULL,
event_time_ms BIGINT NOT NULL,
ingest_time TIMESTAMP NOT NULL,
ingest_time_ms BIGINT NOT NULL,
partition_date DATE NOT NULL,
raw_archive_uri VARCHAR NOT NULL,
raw_checksum VARCHAR,
raw_size_bytes BIGINT NOT NULL,
parse_status VARCHAR NOT NULL,
parse_error_code VARCHAR,
block_count INTEGER NOT NULL,
metadata_json VARCHAR NOT NULL
);
CREATE INDEX IF NOT EXISTS gb32960_frame_vin_time_idx
ON gb32960_frame_index(vin, event_time_ms);
CREATE INDEX IF NOT EXISTS gb32960_frame_raw_uri_idx
ON gb32960_frame_index(raw_archive_uri);
CREATE INDEX IF NOT EXISTS gb32960_frame_partition_time_idx
ON gb32960_frame_index(partition_date, event_time_ms);
```
`gb32960_telemetry_point`
```sql
CREATE TABLE IF NOT EXISTS gb32960_telemetry_point (
event_id VARCHAR NOT NULL,
vin VARCHAR NOT NULL,
event_time TIMESTAMP NOT NULL,
event_time_ms BIGINT NOT NULL,
ingest_time TIMESTAMP NOT NULL,
platform_account VARCHAR,
vendor_profile VARCHAR,
field_key VARCHAR NOT NULL,
value_type VARCHAR NOT NULL,
value_text VARCHAR NOT NULL,
value_num DOUBLE,
unit VARCHAR,
quality VARCHAR NOT NULL,
raw_archive_uri VARCHAR NOT NULL,
PRIMARY KEY(event_id, field_key)
);
CREATE INDEX IF NOT EXISTS gb32960_point_vin_field_time_idx
ON gb32960_telemetry_point(vin, field_key, event_time_ms);
```
`gb32960_alarm_frame`
```sql
CREATE TABLE IF NOT EXISTS gb32960_alarm_frame (
event_id VARCHAR PRIMARY KEY,
vin VARCHAR NOT NULL,
event_time TIMESTAMP NOT NULL,
event_time_ms BIGINT NOT NULL,
level VARCHAR NOT NULL,
active_bits_json VARCHAR NOT NULL,
fault_codes_json VARCHAR NOT NULL,
raw_archive_uri VARCHAR NOT NULL
);
CREATE INDEX IF NOT EXISTS gb32960_alarm_vin_time_idx
ON gb32960_alarm_frame(vin, event_time_ms);
```
### RAW Archive
RAW archive remains file/object based:
```text
archive://yyyy/MM/dd/GB32960/{vin}/{event_id}.bin
```
The archive writer verifies:
- stored byte length equals envelope `raw_archive.size_bytes`;
- checksum matches;
- URI is unique for the deterministic event id.
Frame replay path:
1. query `gb32960_frame_index` by `rawArchiveUri`;
2. read RAW bytes from archive;
3. resolve `platformAccount/vendorProfile` from index metadata;
4. decode using current parser;
5. return decoded frame plus original indexed parse diagnostics.
## Query APIs
Keep existing endpoints but back them by the new DuckDB tables:
- `GET /api/event-history/gb32960/frames`
- full frame query by VIN/time.
- `GET /api/event-history/gb32960/frame`
- one RAW frame replay by `rawArchiveUri`.
- `GET /api/event-history/gb32960/telemetry-snapshots`
- compact telemetry snapshot by VIN/time.
- `GET /api/event-history/gb32960/telemetry-fields`
- field projection optimized by `gb32960_telemetry_point`.
- `GET /api/event-history/gb32960/diagnostics`
- parse status and missing-profile investigation by VIN/time.
Time semantics:
- `eventTime` is business query time.
- `ingestTime` is kept for operational replay, late-arrival detection, and Kafka lag investigation.
- Dates without timezone are interpreted in `Asia/Shanghai`.
## MySQL Analytics Design
Credentials are supplied only by environment variables:
```text
MYSQL_HOST
MYSQL_PORT
MYSQL_DATABASE
MYSQL_USERNAME
MYSQL_PASSWORD
```
Daily derived metrics are stored in the common metric table, not in a
protocol-specific daily-mileage table.
```sql
CREATE TABLE IF NOT EXISTS vehicle_stat_metric (
vin VARCHAR(64) NOT NULL,
stat_date DATE NOT NULL,
metric_key VARCHAR(64) NOT NULL,
metric_value DECIMAL(18,6) NULL,
metric_unit VARCHAR(16) NOT NULL,
calculation_method VARCHAR(64) NOT NULL,
created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP,
updated_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP,
PRIMARY KEY (vin, stat_date, metric_key)
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4;
```
`vehicle_alarm_timeline`
```sql
CREATE TABLE IF NOT EXISTS vehicle_alarm_timeline (
alarm_id VARCHAR(128) NOT NULL,
vin VARCHAR(32) NOT NULL,
alarm_key VARCHAR(128) NOT NULL,
level VARCHAR(32) NOT NULL,
start_time DATETIME(3) NOT NULL,
end_time DATETIME(3),
last_seen_time DATETIME(3) NOT NULL,
start_raw_archive_uri VARCHAR(512),
last_raw_archive_uri VARCHAR(512),
status VARCHAR(16) NOT NULL,
details_json JSON,
updated_at DATETIME(3) NOT NULL,
PRIMARY KEY (alarm_id),
KEY vehicle_alarm_vin_time_idx (vin, start_time),
KEY vehicle_alarm_open_idx (vin, status, alarm_key)
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4;
```
`vehicle_stat_event_dedup`
```sql
CREATE TABLE IF NOT EXISTS vehicle_stat_event_dedup (
event_id VARCHAR(128) NOT NULL,
consumer_group VARCHAR(128) NOT NULL,
processed_at DATETIME(3) NOT NULL,
PRIMARY KEY (event_id, consumer_group)
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4;
```
Writes use `INSERT ... ON DUPLICATE KEY UPDATE` so replayed Kafka messages update the same daily row instead of creating duplicates.
## Daily Metric Algorithms
All daily statistics are keyed by `eventTime` in `Asia/Shanghai`.
### Total Mileage
Use the latest valid `totalMileageKm` seen for the day.
Validation:
- reject negative values;
- reject impossible jumps based on configurable max speed and elapsed time;
- keep the sample as diagnostic if rejected.
### Daily Mileage
JT808 daily mileage uses the GPS total mileage reported in location additional
information. The local-day minimum and maximum GPS total mileage values rewrite
the same metric row, so ordered streaming and historical replay use one
calculation path.
```text
dailyMileage = today.maxTotalMileage - today.minTotalMileage
calculationMethod = JT808_TOTAL_MILEAGE_DIFF
```
The result is stored in `vehicle_stat_metric` with
`metric_key=daily_mileage_km`. There is no JT808-specific daily mileage table,
no Redis mileage state, and no previous-day baseline calculation.
### Daily Power Consumption
Preferred:
- vendor cumulative power-consumption field if available.
Fallback:
```text
deltaKwh = sum(max(0, totalVoltageV * totalCurrentA) * deltaSeconds / 3600000)
```
The sign convention is configurable per vendor profile because GB32960 vehicle current may be positive or negative depending on charging/discharging representation.
### Daily Hydrogen Consumption
Use a tiered market-practical algorithm instead of a single fixed formula:
1. `DIRECT_COUNTER`: vendor cumulative hydrogen consumption or remaining hydrogen mass delta.
2. `PVT_MASS_DELTA`: pressure/temperature/tank-volume mass estimate when storage data is available.
3. `FUEL_CELL_ENERGY_MODEL`: integrate fuel-cell electric output and divide by calibrated efficiency.
4. `UNKNOWN`: keep null when required signals are missing.
The PVT approach follows the industry pattern behind compressed hydrogen consumption tests, where pressure, temperature, and tank volume can be used to estimate hydrogen mass. The implementation must expose calibration constants per vehicle model/profile:
```json
{
"vinPrefix": "LNXNEGRR",
"tankVolumeLiter": 140,
"hydrogenModel": "PVT_MASS_DELTA",
"fuelCellEfficiency": 0.52,
"currentDischargeSign": "POSITIVE"
}
```
### Daily Hydrogen Consumption Rate
```text
dailyHydrogenKgPer100km = dailyHydrogenKg / dailyMileageKm * 100
```
If mileage is missing or less than the configured minimum distance, keep the rate null and set quality to `INSUFFICIENT_DISTANCE`.
### Alarm Timeline
For each telemetry frame:
- derive active alarm keys from alarm bits and fault code arrays;
- open a timeline row when a key first appears;
- update `last_seen_time` while it remains active;
- close open alarms when the key disappears after a configurable grace window;
- keep start/end raw archive URIs for replay.
## Reliability and Observability
Required metrics:
- TCP active connections;
- frames received per command;
- decode failures by code;
- missing platform login rejections;
- vendor profile resolution source;
- Kafka send latency and failures;
- history write batch size, latency, failure count;
- DuckDB query latency by endpoint;
- MySQL upsert latency and failure count;
- consumer lag by group/topic/partition;
- late sample count by VIN/date.
Required logs:
- one structured line for every dropped or rejected valid frame;
- no per-frame INFO spam for successful reports after first frame per VIN/profile unless debug is enabled;
- include `vin`, `platformAccount`, `command`, `eventId`, `reasonCode`, and `peer`.
## Migration Plan
Phase 1: History foundation
- Replace Parquet rewrite store with DuckDB hot tables and batch writer.
- Keep RAW bytes archive.
- Move existing history endpoints onto the new store.
- Add diagnostic endpoint for missing VIN/profile cases.
Phase 2: Ingest association and diagnostics
- Add local `vin-platform-profile.jsonl` loader.
- Use resolved profile before private block parsing.
- Add structured parse/reject diagnostics to Kafka metadata and logs.
- Add tests for platform login, mapping fallback, and partial private parsing.
Phase 3: MySQL analytics
- Add MySQL repository and migrations.
- Implement daily stat accumulator with idempotent upsert.
- Implement alarm timeline open/update/close logic.
- Add integration tests using Testcontainers or a local MySQL profile.
Phase 4: Production hardening
- Add backfill/replay commands from DuckDB RAW archive to analytics.
- Add retention/export from DuckDB to Parquet for cold storage.
- Add dashboards and operational runbook.
## Acceptance Criteria
- A valid GB32960 frame received on port 32960 appears in Kafka with RAW bytes or RAW URI.
- A valid frame is queryable by `vin + eventTime` from history.
- A returned frame can be replayed from `rawArchiveUri` and decoded with the stored profile.
- A VIN with only local mapping, and no current platform login, can still use the correct private parser profile.
- A private block parse failure does not drop the frame.
- Daily MySQL rows update in real time for mileage, total mileage, power consumption, hydrogen consumption, hydrogen rate, and sample metadata.
- Alarm timeline rows preserve full alarm start, update, and close history.
- Replaying the same Kafka event does not duplicate daily stats or alarm rows.
- No secret appears in committed files.
## References
- DuckDB Appender: https://duckdb.org/docs/current/data/appender
- DuckDB INSERT performance guidance: https://duckdb.org/docs/current/data/insert
- MySQL `INSERT ... ON DUPLICATE KEY UPDATE`: https://dev.mysql.com/doc/refman/9.7/en/insert-on-duplicate.html
- SAE J2572 overview: https://h2tools.org/fuel-cell-codes-and-standards/sae-j2572-measuring-exhaust-emissions-energy-consumption-and-range

View File

@@ -1,411 +0,0 @@
> **Superseded:** This 2026-06-23 split design is historical context.
> Use `docs/target-architecture.md` and
> `docs/superpowers/specs/2026-06-29-vehicle-ingest-redesign.md` for the
> current production architecture: TDengine is the default hot history store,
> `sink-archive` owns raw bytes, and `event-file-store` is optional
> compatibility only.
# GB32960 Service Split Design
Date: 2026-06-23
## Goal
Split the current all-in-one vehicle ingest runtime into lower-coupled, higher-cohesion services. The first production slice focuses on GB/T 32960 because it is the active high-pressure path: persistent TCP connections, protocol ACKs, raw frame volume, history queries, and downstream analytics are currently assembled in one `bootstrap-all` process.
The target outcome is:
- Keep GB32960 ingest reliable and fast.
- Move disk writes and history queries out of the protocol process.
- Move statistics and analysis out of the protocol process.
- Use Kafka as the durable boundary between ingest, history, and analytics.
- Preserve `bootstrap-all` during migration as a local development and rollback entry point.
## Decision
Use ACK semantics A:
> A GB32960 frame is ACKed successfully after the protocol service parses/accepts it and writes the required Kafka message(s) successfully.
The protocol service must not wait for raw archive, DuckDB/Parquet writes, snapshot generation, or statistics computation before sending the GB32960 response frame.
## Service Boundaries
### `gb32960-ingest-app`
Purpose: own GB32960 connection handling and Kafka production.
Responsibilities:
- Listen on the GB32960 TCP port.
- Handle platform login, VIN authorization, TLS if enabled, and session state required for protocol responses.
- Decode GB32960 frames and apply vendor extension parsing.
- Produce raw frame records and normalized event records to Kafka.
- ACK successful reports only after Kafka production succeeds.
- Publish malformed or rejected frames to a DLQ when possible.
- Expose operational health and metrics for connections, decode rate, Kafka latency, ACK latency, and DLQ count.
Non-responsibilities:
- No local raw archive writes.
- No DuckDB or Parquet history writes.
- No snapshot query API.
- No vehicle statistics or analysis.
- No long-running analytical calculations.
Initial module dependencies:
- `ingest-api`
- `ingest-core`
- `ingest-codec-common`
- `session-core`
- `protocol-gb32960`
- `sink-kafka`
- `observability`
### `vehicle-history-app`
Purpose: own raw storage, historical index storage, and history query APIs.
Responsibilities:
- Consume GB32960 raw records and normalized event records from Kafka.
- Write raw `.bin` payloads through `sink-archive`.
- Write queryable event indexes through `event-file-store`.
- Provide APIs for raw frame lookup, decoded frame lookup, history query, and snapshot/source-frame lookup.
- Rebuild or backfill the DuckDB sidecar index from stored Parquet files when needed.
- Track consumer lag, write latency, archive failures, index failures, and query latency.
Non-responsibilities:
- No TCP protocol listener.
- No GB32960 ACK decisions.
- No online statistics computation.
- No direct coupling to `gb32960-ingest-app` internals.
Initial module dependencies:
- `ingest-api`
- `sink-kafka`
- `sink-archive`
- `event-file-store`
- `event-history-service`
- `protocol-gb32960` only for raw-frame decode/query support
- `observability`
### `vehicle-analytics-app`
Purpose: own vehicle state, daily statistics, alarms, and later analytics.
Responsibilities:
- Consume normalized vehicle event records from Kafka.
- Maintain latest vehicle state if enabled.
- Compute daily vehicle statistics and alarm-derived metrics.
- Persist analytical outputs to the selected backend.
- Support reprocessing from Kafka offsets when analytical logic changes.
- Track consumer lag, per-VIN processing latency, aggregation failures, and output write latency.
Non-responsibilities:
- No protocol listener.
- No GB32960 ACK decisions.
- No raw archive writes.
- No normal-path dependency on raw archive or history APIs.
Initial module dependencies:
- `ingest-api`
- `sink-kafka`
- `vehicle-state-service`
- `vehicle-stat-service`
- `observability`
## Kafka Contract
Kafka is the service boundary. Messages must be stable enough that `vehicle-history-app` and `vehicle-analytics-app` do not depend on protocol service internals.
### Topics
Use versioned topics:
- `vehicle.raw.gb32960.v1`
- `vehicle.event.gb32960.v1`
- `vehicle.dlq.gb32960.v1`
The current topic names under `lingniu.ingest.sink.kafka.topics` can remain temporarily for compatibility, but the split apps should converge on versioned topic names.
### Keys
Use VIN as the Kafka key whenever a VIN is available.
Benefits:
- Preserves per-vehicle ordering within a partition.
- Lets history and analytics scale by partition.
- Keeps stateful analytics simpler.
When VIN is unavailable, use a stable fallback key such as platform account plus peer address plus receive time bucket.
### Raw Record
Topic: `vehicle.raw.gb32960.v1`
Minimum fields:
- `schemaVersion`
- `protocol = GB32960`
- `eventId`
- `receiveTime`
- `eventTime` if decoded
- `peer`
- `platformAccount`
- `vin` if decoded
- `command`
- `protocolVersion`
- `rawBytes`
- `checksumStatus`
- `decodeStatus`
- `metadata`
Consumers:
- `vehicle-history-app` writes archive `.bin` files and raw-frame query indexes.
- Future replay tools can regenerate normalized events from raw records.
### Normalized Event Record
Topic: `vehicle.event.gb32960.v1`
Minimum fields:
- `schemaVersion`
- `protocol = GB32960`
- `eventId`
- `sourceRawEventId`
- `receiveTime`
- `eventTime`
- `platformAccount`
- `vin`
- `command`
- `eventType`
- `normalizedFields`
- `vendorFields`
- `alarmFields`
- `metadata`
Consumers:
- `vehicle-history-app` writes queryable history records.
- `vehicle-analytics-app` updates state and statistics.
### DLQ Record
Topic: `vehicle.dlq.gb32960.v1`
Minimum fields:
- `schemaVersion`
- `stage`
- `errorCode`
- `errorMessage`
- `receiveTime`
- `peer`
- `platformAccount` if known
- `vin` if known
- `rawBytes` when available
- `metadata`
Typical stages:
- `FRAME_DECODE`
- `BODY_PARSE`
- `AUTH`
- `KAFKA_PRODUCE`
- `ACK_WRITE`
## ACK and Failure Semantics
### Success Path
1. Receive TCP bytes.
2. Decode a complete GB32960 frame.
3. Validate checksum/auth/session rules.
4. Build raw and normalized Kafka records.
5. Produce required Kafka records successfully.
6. Send GB32960 success ACK.
### Kafka Produce Failure
If required Kafka production fails, the service must not send a success ACK.
Allowed behavior:
- Retry within a bounded timeout.
- Return a GB32960 failure response when the protocol allows it.
- Close the channel after repeated produce failures.
- Emit local error logs and metrics.
Do not silently ACK frames that have not crossed the Kafka durability boundary.
### History or Analytics Failure
Failures in `vehicle-history-app` or `vehicle-analytics-app` must not affect GB32960 ACKs. They are handled by Kafka offset retry, DLQ, operational alerting, and backfill/replay.
## Current Coupling to Remove
`bootstrap-all` currently assembles protocol modules, Kafka producer/consumer, archive sink, event file store, history API, vehicle state, vehicle statistics, command gateway, and multiple inbound protocols in one application.
The split should remove these production couplings:
- Protocol runtime directly containing archive and event-file-store writers.
- Protocol runtime directly containing statistics processors.
- History query API sharing the same process as TCP ingest.
- Kafka consumer workers living in the same all-in-one app as protocol listeners.
- Operational scaling tied to one JVM for ingest, storage, queries, and analytics.
## Migration Plan
### Phase 1: Add Split App Entrypoints
Add new app modules:
- `modules/apps/gb32960-ingest-app`
- `modules/apps/vehicle-history-app`
- `modules/apps/vehicle-analytics-app`
Keep `modules/apps/bootstrap-all` unchanged as the fallback runtime.
### Phase 2: GB32960 Ingest App
Create a production profile for `gb32960-ingest-app`:
- Enable `protocol-gb32960`.
- Enable Kafka producer.
- Disable local archive.
- Disable event-file-store.
- Disable history API.
- Disable vehicle state/stat services.
- Disable unrelated protocols.
Acceptance:
- GB32960 TCP port starts.
- Platform login works.
- Realtime report frames are parsed.
- Kafka records are produced with VIN keys.
- ACK is sent only after Kafka success.
### Phase 3: History App
Create `vehicle-history-app`:
- Enable Kafka consumer.
- Enable archive sink.
- Enable event-file-store.
- Enable event-history HTTP API.
- Disable protocol listeners.
- Disable analytics processors.
Acceptance:
- Consumes `vehicle.raw.gb32960.v1`.
- Writes raw `.bin` files.
- Consumes `vehicle.event.gb32960.v1`.
- Writes Parquet/DuckDB indexes.
- Queries can locate raw frames and decoded snapshots.
### Phase 4: Analytics App
Create `vehicle-analytics-app`:
- Enable Kafka consumer.
- Enable vehicle state if a state backend is configured.
- Enable vehicle statistics.
- Disable protocol listeners.
- Disable archive and event-file-store.
Acceptance:
- Consumes `vehicle.event.gb32960.v1`.
- Maintains expected per-VIN state/stat outputs.
- Can be restarted and resume from Kafka offsets.
- Does not affect ingest ACK latency.
### Phase 5: Parallel Run and Cutover
Run old and new paths in parallel for a bounded validation window.
Compare:
- Received frame count.
- Kafka produced record count.
- Raw archive file count.
- Unique VIN count.
- History query count and sample query correctness.
- Analytics output count and sample values.
- GB32960 ACK latency before and after split.
Cut over only after counts and sample queries match within the agreed tolerance.
## Operational Guidance
Scale services independently:
- Scale `gb32960-ingest-app` by connection count and Kafka produce latency.
- Scale `vehicle-history-app` by disk throughput, consumer lag, and query latency.
- Scale `vehicle-analytics-app` by consumer lag and computation latency.
Monitor:
- Kafka producer error rate and p99 produce latency.
- ACK p99 latency.
- TCP active connections.
- Decode failures and DLQ counts.
- Consumer lag per group.
- Archive write latency and failure rate.
- Event-file-store flush latency and failure rate.
- Analytics processing latency.
## Risks and Mitigations
Risk: Kafka outage blocks GB32960 ACKs.
Mitigation: bounded producer retry, clear failure ACK/close behavior, Kafka cluster monitoring, and optional local emergency spool only as a later explicit design.
Risk: Two Kafka topics for raw and normalized events can diverge.
Mitigation: include `eventId` and `sourceRawEventId`, produce records transactionally if required, or publish a single envelope containing both raw and normalized sections in the first implementation if transaction support is not ready.
Risk: History app reprocessing duplicates archive/index records.
Mitigation: deterministic archive keys and idempotent index writes keyed by `eventId` or `rawArchiveUri`.
Risk: Analytics logic changes require backfill.
Mitigation: keep raw and normalized topics with sufficient retention; allow analytics consumers to reset offsets or run a backfill group.
Risk: Query APIs accidentally depend on ingest internals.
Mitigation: history APIs should depend on `event-file-store`, `sink-archive`, and decode libraries only, not `gb32960-ingest-app`.
## Open Decisions
These are intentionally left for implementation planning:
- Whether Kafka production uses two independent sends or a transaction for raw + normalized records.
- Exact protobuf/JSON schema shape for `vehicle.raw.gb32960.v1` and `vehicle.event.gb32960.v1`.
- Whether latest snapshot state belongs only to `vehicle-analytics-app` or is also materialized by `vehicle-history-app` for query convenience.
- Initial Kafka partition count and retention duration.
- Whether command downlink stays with `command-gateway` or gets a separate command service later.
## Completion Criteria
The split is complete when:
- `gb32960-ingest-app`, `vehicle-history-app`, and `vehicle-analytics-app` are separate runnable app modules.
- `gb32960-ingest-app` can receive live GB32960 data and ACK after Kafka success.
- `vehicle-history-app` can consume Kafka and provide raw/history/snapshot query APIs.
- `vehicle-analytics-app` can consume Kafka and compute state/stat outputs.
- `bootstrap-all` remains available for development and rollback.
- Verification covers build, app startup, Kafka production/consumption, raw archive writes, history queries, analytics outputs, and ACK latency.

View File

@@ -1,621 +0,0 @@
# Vehicle Ingest Redesign Design
## Background
Current GB32960 and JT808 ingestion can receive, decode, publish, and query data, but the responsibilities are split across several partially overlapping paths:
- Netty handlers produce `RawFrame` and the dispatcher emits both raw archive events and normalized vehicle events.
- Kafka envelopes carry raw archive references, while raw bytes are stored separately by the archive sink.
- GB32960 historical queries use raw archive records plus file replay decoding.
- JT808 location history is materialized separately into TDengine.
- DuckDB/Parquet, local raw archive, Kafka, and TDengine each own part of the story, so query and durability semantics are hard to reason about.
The new design intentionally keeps useful protocol parsers and app deployment knowledge, but replaces the ingestion, durability, storage, and query boundaries with a first-principles model.
## Goals
1. Store every original inbound protocol frame, including malformed frames when bytes are available.
2. Query historical uploaded data for GB32960 and JT808 by vehicle, protocol, message type, and time range.
3. Extend to new fields, protocol messages, and vendor profiles without rewriting the core pipeline.
4. Sustain production load for 10,000 vehicles and 1,000 concurrent historical query requests per second.
5. Make ACK and durability rules explicit enough that an acknowledged frame is not silently lost.
## Non-Goals
- Do not make Kafka the long-term historical query database.
- Do not store large raw byte arrays directly in Kafka topics.
- Do not keep DuckDB/Parquet as the production hot query store for GB32960 and JT808.
- Do not build one generic API that hides protocol-specific semantics. Shared internals are preferred; external APIs may remain protocol-aware.
## Recommended Architecture
Use `Raw Archive + Kafka + TDengine` as the core architecture.
```mermaid
flowchart LR
GB["GB32960 TCP"] --> Edge["Protocol Edge"]
JT["JT808 TCP"] --> Edge
Edge --> RawStore["Raw Archive"]
RawStore --> RawTopic["Kafka vehicle.raw-frame.v1"]
Edge --> Decode["Protocol Decoder"]
Decode --> FactTopic["Kafka vehicle.decoded-fact.v1"]
RawTopic --> Writer["History Writer"]
FactTopic --> Writer
Writer --> TD["TDengine"]
Writer --> Redis["Realtime State"]
API["History API"] --> TD
API --> RawStore
RealtimeAPI["Realtime API"] --> Redis
```
The architecture has three facts:
- `RawFrameFact`: one row per inbound frame, whether parsing succeeds or fails.
- `DecodedFact`: zero or more normalized facts derived from a raw frame.
- `RawArchiveObject`: immutable raw bytes addressed by `raw_uri`.
The raw archive is the forensic source of truth. TDengine is the query source of truth. Kafka is the decoupling and replay pipe.
## Module Boundaries
### Protocol Edge Apps
Apps:
- `gb32960-ingest-app`
- `jt808-ingest-app`
Responsibilities:
- Accept TCP connections.
- Split frames.
- Perform protocol-level validation, authentication, session binding, and ACK.
- Create `RawFrameFact`.
- Persist raw bytes through `RawArchiveWriter`.
- Publish raw frame fact to Kafka.
- Decode frames and publish decoded facts.
They do not query history, export data, or write business tables directly.
### Ingest Core
New module:
- `modules/core/ingest-facts`
Core types:
```java
public record RawFrameFact(
String frameId,
ProtocolId protocol,
String vehicleKey,
String vin,
String phone,
int messageId,
int subType,
Instant eventTime,
Instant receivedAt,
String peer,
String rawUri,
String checksum,
long rawSizeBytes,
ParseStatus parseStatus,
String parseError,
Map<String, String> metadata
) {}
```
```java
public sealed interface DecodedFact permits
DecodedFact.Location,
DecodedFact.Realtime,
DecodedFact.Alarm,
DecodedFact.Session,
DecodedFact.Register,
DecodedFact.Extension {
String factId();
String frameId();
ProtocolId protocol();
String vehicleKey();
String vin();
Instant eventTime();
Instant receivedAt();
String rawUri();
Map<String, String> metadata();
}
```
`vehicleKey` is the stable partition key:
- GB32960: VIN.
- JT808: resolved VIN when known, otherwise `jt808:<phone>`.
- Unknown/malformed: `unknown:<protocol>:<hash>`.
### Raw Archive
New or refactored module:
- `modules/sinks/raw-archive-store`
Interface:
```java
public interface RawArchiveWriter {
RawArchiveReceipt write(RawArchiveWriteRequest request) throws IOException;
}
```
Receipt:
```java
public record RawArchiveReceipt(
String rawUri,
String checksum,
long sizeBytes
) {}
```
Rules:
- Raw bytes are immutable once written.
- Write path is content-addressed or deterministic by `protocol/date/vehicleKey/frameId.bin`.
- Write must be fsync-capable for local storage and pluggable for object storage.
- `rawUri` must be resolvable by history API for single-frame replay.
- Malformed frames are written with `parseStatus=FAILED` when bytes are available.
Deployment constraints:
- Raw archive, TDengine, and the history service may run on different ECS instances, but should use private-network access in the same VPC and availability zone.
- High-frequency historical queries only read TDengine and must not read raw archive.
- Raw archive is only used by single-frame detail, troubleshooting, replay, and asynchronous export paths.
- `rawUri` must be globally resolvable by services and must not be a private local absolute path on one ECS instance.
- The first local raw archive implementation must stay behind a replaceable interface so it can later move to OSS, MinIO, or another object store.
### Kafka Topics
Use stable versioned topics:
- `vehicle.raw-frame.v1`
- `vehicle.decoded-fact.v1`
- `vehicle.dlq.v1`
Kafka key:
```text
<protocol>:<vehicleKey>
```
`vehicle.raw-frame.v1` payload contains `RawFrameFact` without raw bytes. It carries `rawUri`, checksum, byte size, message id, parse status, and metadata.
`vehicle.decoded-fact.v1` payload contains one `DecodedFact`. It always references `frameId` and `rawUri`.
`vehicle.dlq.v1` carries failed store, publish, decode, and write attempts with enough metadata to replay from raw archive when possible.
### History Writer
New app or refactored app:
- `vehicle-history-app`
Responsibilities:
- Consume `vehicle.raw-frame.v1`.
- Consume `vehicle.decoded-fact.v1`.
- Batch write TDengine.
- Update realtime state store for latest vehicle state.
- Write DLQ on invalid payloads or storage failures.
The writer is idempotent by `(frame_id)` for raw facts and `(fact_id)` for decoded facts.
## TDengine Model
TDengine is the production historical query store.
### Raw Frames Super Table
```sql
CREATE STABLE IF NOT EXISTS raw_frames (
ts TIMESTAMP,
frame_id NCHAR(64),
received_at TIMESTAMP,
message_id INT,
sub_type INT,
event_time TIMESTAMP,
raw_uri NCHAR(512),
checksum NCHAR(128),
raw_size_bytes BIGINT,
parse_status NCHAR(16),
parse_error NCHAR(512),
peer NCHAR(128),
metadata_json NCHAR(4096)
) TAGS (
protocol NCHAR(16),
vehicle_key NCHAR(128),
vin NCHAR(64),
phone NCHAR(32)
);
```
Child table:
```text
raw_<protocol>_<hash(vehicleKey)>
```
Primary query patterns:
- Raw frames by vehicle and time.
- Raw frames by protocol, message id, parse status, and time.
- Single raw frame by `frame_id` or `raw_uri`.
### Location Super Table
```sql
CREATE STABLE IF NOT EXISTS vehicle_locations (
ts TIMESTAMP,
fact_id NCHAR(64),
frame_id NCHAR(64),
received_at TIMESTAMP,
longitude DOUBLE,
latitude DOUBLE,
altitude_m DOUBLE,
speed_kmh DOUBLE,
direction_deg DOUBLE,
alarm_flag BIGINT,
status_flag BIGINT,
total_mileage_km DOUBLE,
raw_uri NCHAR(512),
metadata_json NCHAR(4096)
) TAGS (
protocol NCHAR(16),
vehicle_key NCHAR(128),
vin NCHAR(64),
phone NCHAR(32)
);
```
### Realtime Super Table
```sql
CREATE STABLE IF NOT EXISTS vehicle_realtime (
ts TIMESTAMP,
fact_id NCHAR(64),
frame_id NCHAR(64),
received_at TIMESTAMP,
speed_kmh DOUBLE,
total_mileage_km DOUBLE,
battery_soc DOUBLE,
total_voltage_v DOUBLE,
total_current_a DOUBLE,
running_mode NCHAR(64),
vehicle_state NCHAR(64),
charging_state NCHAR(64),
raw_uri NCHAR(512),
fields_json NCHAR(8192),
metadata_json NCHAR(4096)
) TAGS (
protocol NCHAR(16),
vehicle_key NCHAR(128),
vin NCHAR(64),
phone NCHAR(32)
);
```
`fields_json` stores sparse full-field values for low-frequency inspection. High-frequency queried fields should be promoted to explicit columns or a dedicated extension table.
### Alarm Super Table
```sql
CREATE STABLE IF NOT EXISTS vehicle_alarms (
ts TIMESTAMP,
fact_id NCHAR(64),
frame_id NCHAR(64),
received_at TIMESTAMP,
alarm_level NCHAR(32),
alarm_code BIGINT,
alarm_name NCHAR(128),
longitude DOUBLE,
latitude DOUBLE,
raw_uri NCHAR(512),
fields_json NCHAR(8192),
metadata_json NCHAR(4096)
) TAGS (
protocol NCHAR(16),
vehicle_key NCHAR(128),
vin NCHAR(64),
phone NCHAR(32)
);
```
### Session and Registration Tables
```sql
CREATE STABLE IF NOT EXISTS vehicle_sessions (
ts TIMESTAMP,
fact_id NCHAR(64),
frame_id NCHAR(64),
received_at TIMESTAMP,
session_type NCHAR(32),
result_code INT,
raw_uri NCHAR(512),
fields_json NCHAR(8192),
metadata_json NCHAR(4096)
) TAGS (
protocol NCHAR(16),
vehicle_key NCHAR(128),
vin NCHAR(64),
phone NCHAR(32)
);
```
JT808 registration details go into `vehicle_sessions` first and may later be projected into MySQL identity bindings for durable identity lookup.
## ACK and Durability Rules
For frames that require a protocol ACK:
1. Frame bytes are accepted from Netty.
2. Raw bytes are written to raw archive.
3. `RawFrameFact` is published to Kafka.
4. Protocol-specific ACK is sent.
5. Decoding and `DecodedFact` publication may run before or after ACK, but failures must emit a DLQ event and update raw frame parse status.
For GB32960 realtime/report frames, this prevents acknowledging a frame whose original bytes cannot be recovered.
For JT808 register/auth/heartbeat, ACK follows the same raw archive and raw fact boundary. Register ACK can still include token generation before raw fact publish, but must not be flushed to the channel until the durability boundary succeeds.
If raw archive write fails, close or keep the channel according to protocol tolerance, but do not ACK success.
If Kafka raw fact publish fails, do not ACK success for durable-ACK frames. Non-durable frames go to local DLQ when configured.
## Protocol Decoding and Extension
Use a plugin contract:
```java
public interface ProtocolFrameDecoder {
ProtocolId protocol();
DecodeResult decode(RawFrameFact rawFact, byte[] rawBytes);
}
```
```java
public record DecodeResult(
ParseStatus status,
List<DecodedFact> facts,
String errorMessage,
Map<String, String> metadata
) {}
```
GB32960:
- Frame command maps to session, realtime, location, alarm, and extension facts.
- Vendor extensions are selected by platform account, VIN, or configured profile.
- Full raw replay remains possible by reading `rawUri` and running the current decoder.
JT808:
- Message id maps to register, auth, heartbeat, location, batch location, media metadata, and extension facts.
- Identity resolution updates `vehicleKey`.
- Registration facts carry province, city, maker, device id, plate, plate color, and auth token metadata.
New protocol messages do not require TDengine schema changes unless they become high-frequency query dimensions. Unknown fields are stored as extension facts with `fields_json`.
## Query APIs
Keep API surfaces explicit.
Raw frame APIs:
- `GET /api/history/raw-frames`
- Filters: `protocol`, `vin`, `phone`, `vehicleKey`, `messageId`, `parseStatus`, `dateFrom`, `dateTo`, `pageSize`, `cursor`.
- Returns frame metadata, raw URI, checksum, parse status, and compact decoded summary when available.
Single raw frame API:
- `GET /api/history/raw-frames/{frameId}`
- Returns raw metadata and decoded detail.
- Supports `includeRawHex=true` with a strict size limit.
Location APIs:
- `GET /api/history/locations`
- Filters: `protocol`, `vin`, `phone`, `vehicleKey`, time range, cursor.
Realtime APIs:
- `GET /api/history/realtime`
- Historical time-series query.
Realtime state APIs:
- `GET /api/realtime/vehicles/{vehicleKey}`
- Reads Redis latest state, not TDengine.
Export:
- Export is asynchronous.
- Request creates an export job.
- Worker streams TDengine query results to object storage or local export files.
- API returns job status and download URI.
## Pagination
Use cursor pagination, not deep offset, for high-QPS history queries.
Cursor fields:
```text
ts, received_at, fact_id
```
Descending query predicate:
```sql
WHERE (
ts < :cursorTs
OR (ts = :cursorTs AND received_at < :cursorReceivedAt)
OR (ts = :cursorTs AND received_at = :cursorReceivedAt AND fact_id < :cursorFactId)
)
```
Page size defaults to 100 and is capped at 1000.
## Capacity Design
Assumptions:
- 10,000 vehicles.
- Normal report interval: 5 to 10 seconds.
- Expected writes: 1,000 to 2,000 frames per second.
- Peak burst factor: 3x.
- Historical query target: 1,000 requests per second.
Write path:
- Netty worker threads stay CPU-light.
- Raw archive writes use bounded async workers.
- Kafka producer uses compression, batching, idempotence, and vehicle-key partitioning.
- History writer batches TDengine inserts by super table and child table.
- TDengine child tables are partitioned by vehicle key to make single-vehicle queries cheap.
Read path:
- Most high-QPS queries require vehicle key and time range.
- Unbounded cross-vehicle queries are admin-only and capped.
- Realtime latest state is served from Redis latest-state cache.
- Raw detail replay reads one raw object and decodes on demand.
- Common dictionary and metadata responses are cached in memory.
Backpressure:
- Raw archive worker queue has a fixed bound.
- Kafka publish futures are tracked.
- If durability boundary cannot complete within timeout, durable ACK fails and the channel is closed or throttled.
- History writer lag is observable by Kafka consumer lag and TDengine batch latency.
## Migration Plan
Phase 1: Build new fact model beside existing code.
- Add `ingest-facts`.
- Add raw archive receipt contract.
- Add serializers for raw frame and decoded fact Kafka messages.
- Add tests for stable ids, vehicle key, and raw URI generation.
Phase 2: Harden raw durability boundary.
- Refactor GB32960 and JT808 handlers so ACK waits for raw archive and raw fact publish.
- Keep existing parsers.
- Emit decoded facts to the new Kafka topic.
Phase 3: Build TDengine writer.
- Create TDengine schema manager.
- Write raw frame facts.
- Write location, realtime, alarm, session facts.
- Add idempotent insert behavior by stable ids.
Phase 4: Build query APIs.
- Raw frame query.
- Single frame replay.
- Location history query.
- Realtime history query.
- Latest realtime state query.
Phase 5: Cut over and retire old hot paths.
- Stop using DuckDB/Parquet as production hot history for GB32960/JT808.
- Keep raw replay compatibility while historical data migrates.
- Remove protocol-specific ad hoc history storage once TDengine coverage is verified.
## Testing Strategy
Unit tests:
- `RawFrameFact` validation and vehicle key derivation.
- Raw archive path generation and checksum.
- GB32960 decoder facts from golden frames.
- JT808 decoder facts from sample frames.
- ACK boundary success and failure cases.
- TDengine SQL generation.
Integration tests:
- Start app with embedded or mocked Kafka producer and fake raw archive.
- Send GB32960 sample frame and verify raw fact, decoded fact, raw archive receipt.
- Send JT808 register/auth/location frames and verify registration/session/location facts.
- Simulate raw archive failure and verify no success ACK.
- Simulate Kafka publish failure and verify durable ACK failure.
Storage tests:
- TDengine schema initialization.
- Batch insert raw frames and decoded facts.
- Query by vehicle key and time range.
- Cursor pagination correctness.
- Single raw frame replay by `rawUri`.
Load tests:
- 2,000 frames/s sustained write for at least 30 minutes.
- 6,000 frames/s burst for 5 minutes.
- 1,000 QPS location/realtime history queries with bounded p95 latency.
- Verify no raw frame count gap between raw archive receipts, Kafka raw facts, and TDengine raw frame rows.
## Operational Requirements
Metrics:
- TCP connections.
- Frame receive rate.
- Raw archive write latency and failure count.
- Kafka publish latency and failure count.
- Decoder success/failure count by protocol and message id.
- TDengine batch size, latency, and failure count.
- Query QPS and latency by endpoint.
- Kafka consumer lag.
Logs:
- One structured log per failed frame with `frameId`, protocol, vehicle key, message id, raw URI, and error.
- No full raw bytes in normal logs.
- Raw hex only in explicit debug tools with size limits.
Alerts:
- Raw archive write failure.
- Durable ACK timeout.
- Kafka producer failure.
- History writer lag.
- TDengine write failure.
- Raw archive count and TDengine raw frame row count divergence.
## Acceptance Criteria
The redesign is complete only when all of the following are true:
1. Every accepted GB32960 and JT808 frame creates a raw archive object and a TDengine `raw_frames` row.
2. Malformed frames with bytes are queryable from raw history with parse status and error reason.
3. GB32960 and JT808 location history can be queried by vehicle and time range with cursor pagination.
4. GB32960 realtime history can be queried by vehicle and time range.
5. JT808 registration/session data is queryable and can update identity binding.
6. Single-frame detail can reload raw bytes by `rawUri` and decode with the current decoder.
7. ACK-required frames do not receive success ACK before raw archive and raw fact publication succeed.
8. New protocol fields can be added by implementing a decoder/projector and, only when needed, a TDengine table migration.
9. Load tests demonstrate sustained 2,000 frames/s writes and 1,000 QPS bounded historical queries.
10. Old DuckDB/Parquet hot-history paths are no longer required for GB32960/JT808 production queries.
## Open Decisions
The following choices are intentionally fixed for the first implementation:
- Use local raw archive first, with the interface shaped for object storage later.
- Use TDengine as the only production hot historical query store for GB32960/JT808.
- Use explicit protocol-aware query APIs instead of a single generic query API.
- Use cursor pagination for all high-QPS historical queries.
- Keep Kafka payloads byte-light by sending raw references instead of raw bytes.

View File

@@ -1,624 +0,0 @@
# 车辆接入重设计方案
## 背景
当前 GB32960 和 JT808 链路已经能够完成接收、解析、发布和查询,但职责边界被拆散在多条路径里:
- Netty Handler 生成 `RawFrame`Dispatcher 同时发出原始归档事件和标准化车辆事件。
- Kafka Envelope 只携带 raw archive 引用,原始 bytes 由 archive sink 另行存储。
- GB32960 历史查询依赖 raw archive 记录,再回读文件重新解析。
- JT808 位置历史又单独物化到 TDengine。
- DuckDB/Parquet、本地 raw archive、Kafka、TDengine 分别承担一部分职责,导致查询语义和持久化边界不好推理。
新设计保留已有协议解析器和部署经验,但重新定义接入、持久化、存储和查询边界。
## 目标
1. 保存每一条入站协议原始帧,包括可拿到 bytes 的异常帧。
2. 支持按车辆、协议、消息类型、时间范围查询 GB32960 和 JT808 历史上报数据。
3. 新增字段、协议消息、厂商扩展时,不需要重写核心管线。
4. 支撑 10,000 辆车生产写入,以及 1,000 QPS 历史查询。
5. 明确 ACK 和持久化规则,避免“已经 ACK 但数据悄悄丢失”。
## 非目标
- 不把 Kafka 作为长期历史查询数据库。
- 不把大块原始 bytes 直接塞进 Kafka topic。
- 不继续把 DuckDB/Parquet 作为 GB32960/JT808 的生产热查询库。
- 不做一个掩盖协议差异的万能接口。内部可以统一,外部 API 保持协议语义清晰。
## 推荐架构
采用 `Raw Archive + Kafka + TDengine` 作为核心架构。
```mermaid
flowchart LR
GB["GB32960 TCP"] --> Edge["协议接入边界"]
JT["JT808 TCP"] --> Edge
Edge --> RawStore["Raw Archive"]
RawStore --> RawTopic["Kafka vehicle.raw-frame.v1"]
Edge --> Decode["协议解析器"]
Decode --> FactTopic["Kafka vehicle.decoded-fact.v1"]
RawTopic --> Writer["History Writer"]
FactTopic --> Writer
Writer --> TD["TDengine"]
Writer --> Redis["实时状态"]
API["历史 API"] --> TD
API --> RawStore
RealtimeAPI["实时 API"] --> Redis
```
架构里只有三类事实:
- `RawFrameFact`:每一条入站帧一条记录,不管解析成功还是失败。
- `DecodedFact`:从 raw frame 派生出来的零条或多条标准事实。
- `RawArchiveObject`:不可变原始 bytes通过 `raw_uri` 寻址。
raw archive 是排障和回放的事实源。TDengine 是查询事实源。Kafka 是解耦和重放管道。
## 模块边界
### 协议接入 APP
APP
- `gb32960-ingest-app`
- `jt808-ingest-app`
职责:
- 接收 TCP 连接。
- 完成帧切分。
- 执行协议层校验、鉴权、会话绑定和 ACK。
- 创建 `RawFrameFact`
- 通过 `RawArchiveWriter` 持久化原始 bytes。
- 发布 raw frame fact 到 Kafka。
- 解析协议帧并发布 decoded facts。
接入 APP 不负责历史查询、导出,也不直接写业务查询表。
### Ingest Core
新增模块:
- `modules/core/ingest-facts`
核心类型:
```java
public record RawFrameFact(
String frameId,
ProtocolId protocol,
String vehicleKey,
String vin,
String phone,
int messageId,
int subType,
Instant eventTime,
Instant receivedAt,
String peer,
String rawUri,
String checksum,
long rawSizeBytes,
ParseStatus parseStatus,
String parseError,
Map<String, String> metadata
) {}
```
```java
public sealed interface DecodedFact permits
DecodedFact.Location,
DecodedFact.Realtime,
DecodedFact.Alarm,
DecodedFact.Session,
DecodedFact.Register,
DecodedFact.Extension {
String factId();
String frameId();
ProtocolId protocol();
String vehicleKey();
String vin();
Instant eventTime();
Instant receivedAt();
String rawUri();
Map<String, String> metadata();
}
```
`vehicleKey` 是稳定分区键:
- GB32960VIN。
- JT808已解析到 VIN 时使用 VIN否则使用 `jt808:<phone>`
- 未知或异常帧:使用 `unknown:<protocol>:<hash>`
### Raw Archive
新增或重构模块:
- `modules/sinks/raw-archive-store`
接口:
```java
public interface RawArchiveWriter {
RawArchiveReceipt write(RawArchiveWriteRequest request) throws IOException;
}
```
写入回执:
```java
public record RawArchiveReceipt(
String rawUri,
String checksum,
long sizeBytes
) {}
```
规则:
- 原始 bytes 一旦写入,不允许原地修改。
- 写入路径可以按内容寻址,也可以按 `protocol/date/vehicleKey/frameId.bin` 生成。
- 本地存储需要支持 fsync 能力,接口要预留对象存储实现。
- `rawUri` 必须能被历史 API 解析,用于单帧回放。
- 异常帧只要有 bytes也必须写入并记录 `parseStatus=FAILED`
部署约束:
- Raw archive、TDengine、history 服务可以不在同一台 ECS但应在同一 VPC 和同一可用区内通过内网访问。
- 高频历史查询只访问 TDengine不访问 raw archive。
- Raw archive 只进入单帧详情、问题排查、回放和异步导出路径。
- `rawUri` 必须是全局可解析地址,不能是某台 ECS 私有的本地绝对路径。
- 第一版本地 raw archive 需要通过可替换接口封装,后续可以切换到 OSS、MinIO 或其它对象存储。
### Kafka Topic
使用稳定的版本化 topic
- `vehicle.raw-frame.v1`
- `vehicle.decoded-fact.v1`
- `vehicle.dlq.v1`
Kafka key
```text
<protocol>:<vehicleKey>
```
`vehicle.raw-frame.v1` 的 payload 是不含 raw bytes 的 `RawFrameFact`,携带 `rawUri`、checksum、byte size、message id、parse status 和 metadata。
`vehicle.decoded-fact.v1` 的 payload 是单条 `DecodedFact`,必须引用 `frameId``rawUri`
`vehicle.dlq.v1` 保存存储失败、发布失败、解析失败和写入失败事件,并携带足够元数据,尽量支持从 raw archive 重放。
### History Writer
新增或重构 APP
- `vehicle-history-app`
职责:
- 消费 `vehicle.raw-frame.v1`
- 消费 `vehicle.decoded-fact.v1`
- 批量写 TDengine。
- 更新实时状态存储。
- 对非法 payload 或存储失败写 DLQ。
Writer 写入需要幂等:
- raw fact 以 `(frame_id)` 幂等。
- decoded fact 以 `(fact_id)` 幂等。
## TDengine 模型
TDengine 是生产历史查询主库。
### 原始帧超级表
```sql
CREATE STABLE IF NOT EXISTS raw_frames (
ts TIMESTAMP,
frame_id NCHAR(64),
received_at TIMESTAMP,
message_id INT,
sub_type INT,
event_time TIMESTAMP,
raw_uri NCHAR(512),
checksum NCHAR(128),
raw_size_bytes BIGINT,
parse_status NCHAR(16),
parse_error NCHAR(512),
peer NCHAR(128),
metadata_json NCHAR(4096)
) TAGS (
protocol NCHAR(16),
vehicle_key NCHAR(128),
vin NCHAR(64),
phone NCHAR(32)
);
```
子表命名:
```text
raw_<protocol>_<hash(vehicleKey)>
```
主要查询模式:
- 按车辆和时间查询原始帧。
- 按协议、消息 ID、解析状态和时间查询原始帧。
-`frame_id``raw_uri` 查询单帧。
### 位置超级表
```sql
CREATE STABLE IF NOT EXISTS vehicle_locations (
ts TIMESTAMP,
fact_id NCHAR(64),
frame_id NCHAR(64),
received_at TIMESTAMP,
longitude DOUBLE,
latitude DOUBLE,
altitude_m DOUBLE,
speed_kmh DOUBLE,
direction_deg DOUBLE,
alarm_flag BIGINT,
status_flag BIGINT,
total_mileage_km DOUBLE,
raw_uri NCHAR(512),
metadata_json NCHAR(4096)
) TAGS (
protocol NCHAR(16),
vehicle_key NCHAR(128),
vin NCHAR(64),
phone NCHAR(32)
);
```
### 实时数据超级表
```sql
CREATE STABLE IF NOT EXISTS vehicle_realtime (
ts TIMESTAMP,
fact_id NCHAR(64),
frame_id NCHAR(64),
received_at TIMESTAMP,
speed_kmh DOUBLE,
total_mileage_km DOUBLE,
battery_soc DOUBLE,
total_voltage_v DOUBLE,
total_current_a DOUBLE,
running_mode NCHAR(64),
vehicle_state NCHAR(64),
charging_state NCHAR(64),
raw_uri NCHAR(512),
fields_json NCHAR(8192),
metadata_json NCHAR(4096)
) TAGS (
protocol NCHAR(16),
vehicle_key NCHAR(128),
vin NCHAR(64),
phone NCHAR(32)
);
```
`fields_json` 存放稀疏全字段值,用于低频排查。高频查询字段应提升为明确列,或放入专用扩展表。
### 报警超级表
```sql
CREATE STABLE IF NOT EXISTS vehicle_alarms (
ts TIMESTAMP,
fact_id NCHAR(64),
frame_id NCHAR(64),
received_at TIMESTAMP,
alarm_level NCHAR(32),
alarm_code BIGINT,
alarm_name NCHAR(128),
longitude DOUBLE,
latitude DOUBLE,
raw_uri NCHAR(512),
fields_json NCHAR(8192),
metadata_json NCHAR(4096)
) TAGS (
protocol NCHAR(16),
vehicle_key NCHAR(128),
vin NCHAR(64),
phone NCHAR(32)
);
```
### 会话和注册表
```sql
CREATE STABLE IF NOT EXISTS vehicle_sessions (
ts TIMESTAMP,
fact_id NCHAR(64),
frame_id NCHAR(64),
received_at TIMESTAMP,
session_type NCHAR(32),
result_code INT,
raw_uri NCHAR(512),
fields_json NCHAR(8192),
metadata_json NCHAR(4096)
) TAGS (
protocol NCHAR(16),
vehicle_key NCHAR(128),
vin NCHAR(64),
phone NCHAR(32)
);
```
JT808 注册详情先进入 `vehicle_sessions`,后续可以投影到 MySQL 车辆身份绑定表,用于持久身份解析。
## ACK 和持久化规则
对需要协议 ACK 的帧:
1. Netty 接收到 frame bytes。
2. 原始 bytes 写入 raw archive。
3. `RawFrameFact` 发布到 Kafka。
4. 发送协议 ACK。
5. 解码和 `DecodedFact` 发布可以在 ACK 前或 ACK 后执行,但失败必须发 DLQ并更新 raw frame 解析状态。
对 GB32960 实时上报和补发帧,这条规则保证不会 ACK 一条无法恢复原始 bytes 的帧。
对 JT808 注册、鉴权、心跳,也遵循相同 raw archive 和 raw fact 边界。注册 ACK 可以先生成 token但在 durability boundary 成功前不能 flush 到 channel。
如果 raw archive 写入失败,不返回成功 ACK。具体是关闭连接还是保留连接由协议容忍度决定。
如果 Kafka raw fact 发布失败durable ACK 帧不返回成功 ACK。非 durable 帧可以写本地 DLQ。
## 协议解析和扩展
使用插件契约:
```java
public interface ProtocolFrameDecoder {
ProtocolId protocol();
DecodeResult decode(RawFrameFact rawFact, byte[] rawBytes);
}
```
```java
public record DecodeResult(
ParseStatus status,
List<DecodedFact> facts,
String errorMessage,
Map<String, String> metadata
) {}
```
GB32960
- 命令帧映射为会话、实时、位置、报警和扩展事实。
- 厂商扩展按平台账号、VIN 或配置 profile 选择。
- 全量 raw replay 仍通过读取 `rawUri` 并运行当前解析器完成。
JT808
- 消息 ID 映射为注册、鉴权、心跳、位置、批量位置、多媒体元数据和扩展事实。
- 身份解析会更新 `vehicleKey`
- 注册事实携带省、市、厂商、设备 ID、车牌、车牌颜色和 auth token metadata。
新增协议消息通常不需要改 TDengine schema除非它成为高频查询维度。未知字段先作为 extension fact 存入 `fields_json`
## 查询 API
API 保持明确,不做一个含糊的统一接口。
原始帧查询:
- `GET /api/history/raw-frames`
- 过滤条件:`protocol``vin``phone``vehicleKey``messageId``parseStatus``dateFrom``dateTo``pageSize``cursor`
- 返回 frame metadata、raw URI、checksum、parse status以及可选的紧凑解析摘要。
单帧详情:
- `GET /api/history/raw-frames/{frameId}`
- 返回 raw metadata 和解码详情。
- 支持 `includeRawHex=true`,但必须有严格大小限制。
位置历史:
- `GET /api/history/locations`
- 过滤条件:`protocol``vin``phone``vehicleKey`、时间范围、cursor。
实时历史:
- `GET /api/history/realtime`
- 查询历史时序数据。
实时状态:
- `GET /api/realtime/vehicles/{vehicleKey}`
- 从 Redis 或内存 latest state 读取,不扫 TDengine。
导出:
- 导出必须异步。
- 请求创建 export job。
- Worker 将 TDengine 查询结果流式写入对象存储或本地导出文件。
- API 返回 job 状态和下载 URI。
## 分页
高 QPS 历史查询使用 cursor pagination不使用深 offset。
Cursor 字段:
```text
ts, received_at, fact_id
```
倒序查询谓词:
```sql
WHERE (
ts < :cursorTs
OR (ts = :cursorTs AND received_at < :cursorReceivedAt)
OR (ts = :cursorTs AND received_at = :cursorReceivedAt AND fact_id < :cursorFactId)
)
```
默认 page size 为 100上限为 1000。
## 容量设计
假设:
- 10,000 辆车。
- 常规上报间隔 5 到 10 秒。
- 预计写入 1,000 到 2,000 frames/s。
- 峰值突发系数 3 倍。
- 历史查询目标 1,000 requests/s。
写路径:
- Netty worker 保持 CPU-light。
- Raw archive 写入使用有界异步 worker。
- Kafka producer 开启压缩、批量、幂等,并按 vehicle key 分区。
- History writer 按超级表和子表批量写 TDengine。
- TDengine 子表按 vehicle key 分区,使单车查询足够便宜。
读路径:
- 高频查询必须带 vehicle key 和时间范围。
- 无边界跨车查询仅限管理接口,并强制限流和限制返回量。
- 实时 latest state 从 Redis 或内存状态读取。
- Raw 详情只读取单个 raw object 并按需解析。
- 字典和元数据接口使用内存缓存。
背压:
- Raw archive worker queue 固定上限。
- Kafka publish future 必须被跟踪。
- Durability boundary 超时时durable ACK 失败channel 关闭或被限流。
- History writer lag 通过 Kafka consumer lag 和 TDengine batch latency 观察。
## 迁移计划
阶段 1在现有代码旁边建立新的 fact model。
- 新增 `ingest-facts`
- 新增 raw archive receipt contract。
- 新增 raw frame 和 decoded fact 的 Kafka 序列化。
- 为 stable id、vehicle key、raw URI 生成规则加测试。
阶段 2强化 raw durability boundary。
- 重构 GB32960 和 JT808 handler让 ACK 等待 raw archive 和 raw fact publish。
- 保留现有解析器。
- 向新 Kafka topic 发 decoded facts。
阶段 3构建 TDengine writer。
- 创建 TDengine schema manager。
- 写入 raw frame facts。
- 写入 location、realtime、alarm、session facts。
- 按 stable id 实现幂等写入。
阶段 4构建查询 API。
- Raw frame 查询。
- 单帧回放。
- 位置历史查询。
- 实时历史查询。
- 最新实时状态查询。
阶段 5切换并下线旧热路径。
- GB32960/JT808 生产热历史不再依赖 DuckDB/Parquet。
- 历史迁移期间保留 raw replay 兼容。
- TDengine 覆盖验证完成后,移除协议特定的临时历史存储。
## 测试策略
单元测试:
- `RawFrameFact` 校验和 vehicle key 派生。
- Raw archive 路径生成和 checksum。
- GB32960 golden frame 到 decoded facts。
- JT808 sample frame 到 decoded facts。
- ACK boundary 成功和失败场景。
- TDengine SQL 生成。
集成测试:
- 使用 mocked Kafka producer 和 fake raw archive 启动 APP。
- 发送 GB32960 sample frame验证 raw fact、decoded fact 和 raw archive receipt。
- 发送 JT808 注册、鉴权、位置帧,验证 registration、session、location facts。
- 模拟 raw archive 失败,验证不发送成功 ACK。
- 模拟 Kafka publish 失败,验证 durable ACK 失败。
存储测试:
- TDengine schema 初始化。
- 批量写入 raw frames 和 decoded facts。
- 按 vehicle key 和时间范围查询。
- Cursor 分页正确性。
-`rawUri` 单帧回放。
压测:
- 持续 30 分钟 2,000 frames/s 写入。
- 持续 5 分钟 6,000 frames/s 突发写入。
- 1,000 QPS location/realtime 历史查询p95 延迟受控。
- 校验 raw archive receipt、Kafka raw fact、TDengine raw frame row 三者数量无缺口。
## 运维要求
指标:
- TCP 连接数。
- Frame 接收速率。
- Raw archive 写入延迟和失败数。
- Kafka 发布延迟和失败数。
- 按协议和 message id 统计 decoder 成功/失败数。
- TDengine batch size、延迟和失败数。
- 各查询 endpoint 的 QPS 和延迟。
- Kafka consumer lag。
日志:
- 每个失败帧输出一条结构化日志,包含 `frameId`、protocol、vehicle key、message id、raw URI 和 error。
- 正常日志不输出完整 raw bytes。
- Raw hex 只允许在显式 debug 工具里输出,并限制大小。
告警:
- Raw archive 写入失败。
- Durable ACK 超时。
- Kafka producer 失败。
- History writer lag。
- TDengine 写入失败。
- Raw archive 数量和 TDengine raw frame row 数量不一致。
## 验收标准
重设计只有满足以下条件才算完成:
1. 每一条被接收的 GB32960 和 JT808 帧都会创建 raw archive object 和 TDengine `raw_frames` 行。
2. 有 bytes 的异常帧可以从 raw history 查询到,包含 parse status 和错误原因。
3. GB32960 和 JT808 位置历史可以按车辆和时间范围 cursor 分页查询。
4. GB32960 实时历史可以按车辆和时间范围查询。
5. JT808 注册和会话数据可查询,并且可以更新 identity binding。
6. 单帧详情可以通过 `rawUri` 重新读取 raw bytes并使用当前解析器解码。
7. 需要 ACK 的帧,在 raw archive 和 raw fact publish 成功前,不会发送成功 ACK。
8. 新协议字段可以通过新增 decoder/projector 扩展;只有成为高频查询字段时才需要 TDengine 表迁移。
9. 压测证明可以持续 2,000 frames/s 写入,并支持 1,000 QPS 有边界历史查询。
10. GB32960/JT808 生产查询不再依赖旧 DuckDB/Parquet 热历史路径。
## 已固定决策
第一版实现固定以下选择:
- 先使用本地 raw archive实现接口时预留对象存储。
- TDengine 是 GB32960/JT808 唯一生产热历史查询库。
- 使用协议明确的查询 API不做一个泛化万能查询 API。
- 高频历史查询统一使用 cursor pagination。
- Kafka payload 保持轻量,只传 raw 引用,不传 raw bytes。

View File

@@ -1,498 +0,0 @@
# Go 车辆接入重构设计规格
## 目标
使用 Go 重新设计车辆数据接入运行面,让系统回到第一性原则:
- 接入层只负责可靠收包、协议解析、应答、身份解析和投递。
- Kafka 是唯一在线消息通道。
- TDengine 保存高吞吐时序明细和完整 RAW 解析 JSON。
- MySQL 保存低频配置、身份映射和每日统计窄表。
- Redis 保存可重建的准实时状态。
- 32960、JT808、宇通 MQTT 使用同一套数据层接口,避免每个协议各自落库。
第一阶段只做生产链路的骨架和核心数据闭环接收、解析、Kafka、TDengine、MySQL 统计、Redis 实时查询、ECS 验证。历史查询 API 和业务侧复杂分析放在后续阶段。
## 参考原则
本设计只吸收开源项目的边界思想,不复制外部实现代码。
- GB32960 参考 `DarkInno/gb32960-go-sdk`连接管理、Handler、Forwarder、鉴权接口分离。
- JT808 参考 `cuteLittleDevil/go-jt808`:高并发 Go 原生 TCP、协议核心小而可扩展、事件/适配器机制。
- JT808 参考 `fakeyanss/jt808-server-go``FramePayload -> PacketData -> JT808Msg -> Reply` 的分层处理、Gateway 模式、保活和版本兼容。
- TDengine 参考官方数据模型:按数据采集点建立 supertable静态维度做 tags明细数据按子表写入Go 连接优先使用 WebSocket 驱动。
## 非目标
- 第一阶段不继续扩展 Java 服务。
- 第一阶段不做统一大查询 API。
- 第一阶段不恢复信达 Push。
- 第一阶段不把所有协议字段展开成一张超宽 TDengine 表。
- Redis 不作为历史事实源。
- MySQL 不保存高频原始明细。
## 目标目录
新的 Go 运行面放在 `go/vehicle-gateway`,不继续沿用实验性的多 module 分散结构。现有 `go/ingest-edge``go/vehicle-state` 可以作为迁移素材,最终归并或删除。
```text
go/vehicle-gateway/
cmd/
gateway/ # 协议接入进程32960 TCP、JT808 TCP、宇通 MQTT
history-writer/ # Kafka -> TDengine RAW/location/mileage points
stat-writer/ # Kafka -> MySQL 每日统计窄表
realtime-api/ # Redis 准实时查询 API
internal/
config/ # 环境变量、配置文件、校验
gateway/ # TCP/MQTT 生命周期、连接限制、优雅停机
protocol/
gb32960/ # 32960 编解码
jt808/ # 808 编解码
yutongmqtt/ # 宇通 MQTT payload 解析
identity/ # VIN、phone、device_id、plate 绑定
envelope/ # 统一事件模型
eventbus/ # Kafka producer/consumer
history/ # TDengine adapter
stats/ # 每日里程聚合
realtime/ # Redis adapter 与快照合并
observability/ # 日志、metrics、健康检查
```
## 协议接入设计
### GB32960
职责:
- 监听 TCP `32960`
- 支持车辆登入、实时信息上报、补发信息上报、车辆登出、平台登入/登出、心跳、校时。
- 正确返回协议应答。
- 完整解析 2016 数据单元:
- 整车数据
- 驱动电机数据
- 燃料电池数据
- 发动机数据
- 车辆位置数据
- 极值数据
- 报警数据
- 可充电储能装置电压数据
- 可充电储能装置温度数据
- RAW 中保存完整 parsed JSON。
- 关键字段标准化为统一 field
- `speed_kmh`
- `total_mileage_km`
- `soc_percent`
- `longitude`
- `latitude`
- `vehicle_status`
- `charge_status`
### JT808
职责:
- 监听 TCP `808`
- 支持 2011/2013 基础兼容,预留 2019 版本标记。
- 分层处理:
- Frame`0x7e` 包边界、转义、校验。
- Packet消息头、手机号 BCD、流水号、分包。
- Message注册、鉴权、心跳、注销、位置、批量位置、未知上行。
- Reply注册应答、平台通用应答。
- 终端手机号按协议头 BCD 解析,内部同时保留原始 BCD、规范化 phone、去前导零 phone。
- 注册帧和鉴权帧写入 808 registration 表。
- 如果设备直接上报位置帧,也要走 identity 解析,尝试用 phone、device_id、plate 找到 VIN。
- 0200 附加信息必须解析表 27
- `0x01` GPS 总里程DWORD单位 0.1 km。
- 其他已知附加项结构化放入 `parsed_json.additional`
- 标准化字段:
- `speed_kmh`
- `total_mileage_km`
- `longitude`
- `latitude`
- `altitude_m`
- `direction_deg`
- `alarm_flag`
- `status_flag`
### 宇通 MQTT
职责:
- 使用正式 MQTT 配置订阅生产 topic。
- 接收 payload 后解析为统一 envelope。
- 按 payload 中的车辆标识解析 VIN、车牌、设备号。
- 保存完整 payload 和 parsed JSON。
- 标准化字段向 32960/808 对齐:
- `speed_kmh`
- `total_mileage_km`
- `longitude`
- `latitude`
- `soc_percent`
- `vehicle_status`
## 统一 Envelope
所有协议接入后输出同一结构。
```json
{
"event_id": "string",
"trace_id": "string",
"protocol": "GB32960|JT808|YUTONG_MQTT",
"message_id": "string",
"vin": "string",
"vehicle_key": "string",
"phone": "string",
"device_id": "string",
"plate": "string",
"source_endpoint": "ip:port or mqtt://broker/topic",
"event_time_ms": 0,
"received_at_ms": 0,
"raw_hex": "string",
"raw_text": "string",
"parsed": {},
"fields": {
"speed_kmh": 0,
"total_mileage_km": 0,
"longitude": 0,
"latitude": 0
},
"parse_status": "OK|PARTIAL|BAD_FRAME",
"parse_error": ""
}
```
规则:
- `event_id` 使用协议、设备标识、消息流水、事件时间、RAW hash 生成,保证幂等。
- Kafka partition key 优先使用 VIN没有 VIN 时使用 `protocol:phone/device_id`
- `parsed` 保存协议完整结构化字段。
- `fields` 只保存跨协议核心字段,用于统计、位置和实时快照。
## Kafka Topic
生产只支持 Kafka。
| Topic | 内容 | Key |
|---|---|---|
| `vehicle.raw.gb32960.v1` | 32960 完整 RAW envelope | VIN 或 vehicle_key |
| `vehicle.raw.jt808.v1` | 808 完整 RAW envelope | VIN 或 phone |
| `vehicle.raw.yutong-mqtt.v1` | 宇通 MQTT 完整 RAW envelope | VIN 或 vehicle_key |
| `vehicle.event.unified.v1` | 统一轻量事件,用于业务消费 | VIN 或 vehicle_key |
接入层必须先写 RAW topic再写 unified event。后续消费者只依赖 Kafka不直接依赖接入进程内存。
发布可靠性:
- Kafka 生产端必须开启同步写入RAW 写成功后才允许写 unified event。
- Kafka 写入必须支持可配置重试、单次写超时和短退避,默认值为 `KAFKA_PUBLISH_ATTEMPTS=3``KAFKA_PUBLISH_TIMEOUT_MS=3000``KAFKA_PUBLISH_BACKOFF_MS=100`
- gateway 支持本地磁盘 spool/WAL配置 `KAFKA_SPOOL_DIR` 后启用Kafka 长时间不可用时,发布失败的 envelope 先原子写入本地 JSON 文件。
- spool 补发按文件名顺序执行,补发成功后删除文件;默认补发间隔由 `KAFKA_SPOOL_REPLAY_INTERVAL_MS=1000` 控制。
- 如果 RAW 写 Kafka 失败并已落盘,同一个 event 的 unified event 不允许抢先写 Kafka也必须进入 spool确保恢复后按 RAW -> unified 顺序补发。
- Kafka topic 不允许自动创建topic 和分区数由部署脚本或运维初始化,避免生产拼写错误造成隐性分流。
## TDengine 数据库设计
如果现有 TDengine 库不符合目标,可以新建库并迁移。建议库名:
```sql
CREATE DATABASE IF NOT EXISTS lingniu_vehicle_ts KEEP 7300 DURATION 10 BUFFER 256;
```
### raw_frames
保存完整 RAW 和完整 parsed JSON。
```sql
CREATE STABLE IF NOT EXISTS raw_frames (
ts TIMESTAMP,
frame_id NCHAR(64),
event_id NCHAR(64),
message_id INT,
event_time TIMESTAMP,
received_at TIMESTAMP,
raw_size_bytes INT,
raw_hex BINARY(16374),
raw_text BINARY(16374),
parsed_json BINARY(16374),
fields_json BINARY(4096),
parse_status NCHAR(16),
parse_error BINARY(1024),
source_endpoint NCHAR(128)
) TAGS (
protocol NCHAR(32),
vehicle_key NCHAR(64),
vin NCHAR(32),
phone NCHAR(32),
device_id NCHAR(64)
);
```
说明:
- 子表按 `protocol + vehicle_key hash` 创建。
- RAW 查询优先按 tags 和时间范围过滤。
- `parsed_json` 是完整协议字段;查询 API 可选择是否返回,避免默认大字段拖慢。
### vehicle_locations
只保存位置查询核心字段。
```sql
CREATE STABLE IF NOT EXISTS vehicle_locations (
ts TIMESTAMP,
event_id NCHAR(64),
frame_id NCHAR(64),
received_at TIMESTAMP,
longitude DOUBLE,
latitude DOUBLE,
altitude_m DOUBLE,
speed_kmh DOUBLE,
direction_deg INT,
alarm_flag BIGINT,
status_flag BIGINT,
total_mileage_km DOUBLE
) TAGS (
protocol NCHAR(32),
vehicle_key NCHAR(64),
vin NCHAR(32),
phone NCHAR(32),
device_id NCHAR(64)
);
```
### vehicle_mileage_points
保存总里程采样点,用于回放重算和查询。
```sql
CREATE STABLE IF NOT EXISTS vehicle_mileage_points (
ts TIMESTAMP,
event_id NCHAR(64),
frame_id NCHAR(64),
received_at TIMESTAMP,
total_mileage_km DOUBLE,
speed_kmh DOUBLE,
longitude DOUBLE,
latitude DOUBLE
) TAGS (
protocol NCHAR(32),
vehicle_key NCHAR(64),
vin NCHAR(32),
phone NCHAR(32),
device_id NCHAR(64)
);
```
## MySQL 数据库设计
MySQL 只保存配置、身份映射、808 注册鉴权和每日统计窄表。
### vehicle_identity_binding
用于把外部标识定位到 VIN。
```sql
CREATE TABLE vehicle_identity_binding (
id BIGINT PRIMARY KEY AUTO_INCREMENT,
vin VARCHAR(32) NOT NULL,
plate VARCHAR(32) NULL,
phone VARCHAR(32) NULL,
device_id VARCHAR(64) NULL,
remark VARCHAR(255) NULL,
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
updated_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP,
UNIQUE KEY uk_vin (vin),
KEY idx_plate (plate),
KEY idx_phone (phone),
KEY idx_device_id (device_id)
);
```
### jt808_registration
只记录 808 独有的注册、鉴权和在线身份信息。
```sql
CREATE TABLE jt808_registration (
phone VARCHAR(32) PRIMARY KEY,
vin VARCHAR(32) NULL,
plate VARCHAR(32) NULL,
device_id VARCHAR(64) NULL,
manufacturer_id VARCHAR(32) NULL,
terminal_model VARCHAR(64) NULL,
terminal_id VARCHAR(64) NULL,
auth_code VARCHAR(128) NULL,
source_endpoint VARCHAR(128) NULL,
first_registered_at DATETIME NULL,
latest_registered_at DATETIME NULL,
latest_authed_at DATETIME NULL,
latest_seen_at DATETIME NULL,
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
updated_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP,
KEY idx_vin (vin),
KEY idx_plate (plate),
KEY idx_device_id (device_id)
);
```
### vehicle_daily_metric
32960 和 808 的每日里程、每日总里程都进入同一张窄表。
```sql
CREATE TABLE vehicle_daily_metric (
id BIGINT PRIMARY KEY AUTO_INCREMENT,
vin VARCHAR(32) NOT NULL,
stat_date DATE NOT NULL,
protocol VARCHAR(32) NOT NULL,
metric_key VARCHAR(64) NOT NULL,
metric_value DECIMAL(18,3) NOT NULL,
metric_unit VARCHAR(16) NOT NULL,
first_total_mileage_km DECIMAL(18,3) NULL,
latest_total_mileage_km DECIMAL(18,3) NULL,
sample_count BIGINT NOT NULL DEFAULT 0,
calculation_method VARCHAR(64) NOT NULL,
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
updated_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP,
UNIQUE KEY uk_daily_metric (vin, stat_date, protocol, metric_key),
KEY idx_stat_date (stat_date),
KEY idx_protocol_metric (protocol, metric_key)
);
```
指标:
- `daily_mileage_km = max(total_mileage_km) - min(total_mileage_km)`
- `daily_total_mileage_km = max(total_mileage_km)`
统计消费者可幂等重放。每个样本 upsert 时只维护当天 `min/max total_mileage_km`,不依赖进程内状态。
## Redis 实时数据设计
Redis 保存可从 Kafka 重建的数据。
| Key | 内容 |
|---|---|
| `vehicle:online:{vin}` | 在线状态、最后活跃时间、协议列表 |
| `vehicle:latest:{vin}` | 合并后的 VIN 最新快照 |
| `vehicle:latest:{vin}:{protocol}` | 单协议最新快照 |
| `vehicle:last_seen` | VIN 到最后接收时间的 sorted set |
合并规则:
- 位置以最新 event_time 为准。
- 总里程取最新上报值,不做递增假设。
- 32960 多帧字段按 field timestamp 合并。
- 808 和 MQTT 不覆盖其他协议独有字段。
- Redis TTL 用于在线判断,不作为数据删除依据。
## 数据流
```mermaid
flowchart LR
source["车辆 / 平台"] --> gateway["Go gateway"]
gateway --> parser["protocol parser"]
parser --> identity["identity resolver"]
identity --> kafkaRaw["Kafka RAW topics"]
identity --> kafkaEvent["Kafka unified event"]
kafkaRaw --> history["history-writer"]
kafkaRaw --> stats["stat-writer"]
kafkaEvent --> realtime["realtime consumer"]
history --> td["TDengine raw_frames / locations / mileage_points"]
stats --> mysql["MySQL vehicle_daily_metric"]
realtime --> redis["Redis latest state"]
```
## 错误处理
- 协议坏帧仍写 RAW topic`parse_status=BAD_FRAME`
- 可部分解析的帧写 `parse_status=PARTIAL`,保留 parse error。
- Kafka 写失败时接入层先按配置重试;重试耗尽后不写 unified event必须打错误日志和 metrics。
- 启用 `KAFKA_SPOOL_DIR`Kafka 重试耗尽会落本地 spool如果 spool 写失败,才视为接入层最终失败。
- TDengine 写失败不提交 Kafka offset。
- MySQL 统计写失败不提交 Kafka offset。
- Redis 写失败不影响历史和统计,但要通过 metrics 暴露。
- 808 相同 phone 新连接时,旧连接应被替换并记录 latest_seen。
## 部署设计
生产 ECS 运行 Go 二进制或容器:
- `vehicle-gateway`
- `vehicle-history-writer`
- `vehicle-stat-writer`
- `vehicle-realtime-api`
配置来源:
- 环境变量优先。
- Nacos 可作为后续配置中心,但 Go 第一阶段必须能只靠环境变量启动,便于 Portainer 部署和故障恢复。
TDengine 连接:
- 优先 WebSocket `taosWS`
- 若当前 ECS 只开放原生端口,可短期兼容 `taosSql`,但代码配置必须显式标记为兼容模式。
## 验证计划
### 本地验证
- 用样例 808 报文验证:
- phone BCD 解析。
- 0200 经纬度、速度、方向、时间。
- 表 27 `0x01` GPS 总里程。
- 用真实 32960 报文验证:
- 登录应答正确。
- 实时帧和补发帧都能写 RAW。
- 位置和总里程字段进入统一 fields。
- 用宇通 MQTT payload 验证:
- payload 完整保存。
- 核心字段归一化。
- 不连 Kafka 时可输出 JSON log。
- 连接 Kafka 时 RAW topic 可消费到 envelope。
### ECS 验证
- 部署前先旁路接收或短窗口切流。
- 验证端口:
- `32960`
- `808`
- MQTT 订阅连接。
- 验证 Kafka topic 有持续写入。
- 验证 TDengine
- `raw_frames` 有三种协议数据。
- `vehicle_locations` 能按 VIN、协议、时间分页查询。
- `vehicle_mileage_points` 有 32960 和 808 总里程采样。
- 验证 MySQL
- `vehicle_daily_metric` 有 32960 和 808 的 `daily_mileage_km``daily_total_mileage_km`
- 验证 Redis
- 查询 VIN 在线。
- 查询 VIN 合并实时快照。
- 查询 VIN 分协议实时快照。
## 验收标准
- Go gateway 能在 ECS 上接收真实 32960、808、宇通 MQTT 数据。
- 三种协议 RAW 都进入 Kafka。
- Kafka 短暂写失败时gateway 按配置重试;单元测试覆盖首次失败后成功和重试耗尽返回错误。
- Kafka 长时间不可用时gateway 可把 RAW/unified 写入本地 spool单元测试覆盖 RAW 已落盘时 unified 也落盘、恢复后按顺序补发并删除文件。
- 三种协议 RAW 和 parsed JSON 都进入 TDengine。
- 32960 和 808 的位置进入 `vehicle_locations`
- 32960 和 808 的总里程采样进入 `vehicle_mileage_points`
- 32960 和 808 的每日里程、每日总里程进入 MySQL 窄表。
- Redis 可以查询 VIN 在线和实时数据。
- 任一消费者宕机后可通过 Kafka offset 继续恢复。
- 查询和统计验证使用 ECS 上的真实数据完成。
## 实施顺序
1. 创建 `go/vehicle-gateway` 单 module迁移现有 Go 原型中可用的解析、Kafka、TDengine、MySQL、Redis 代码。
2. 先写协议解析单元测试,覆盖 808 样例报文和 32960 核心字段。
3. 完成统一 envelope 和 Kafka sink。
4. 完成 TDengine schema bootstrap 与 history writer。
5. 完成 MySQL schema bootstrap 与 stat writer。
6. 完成 Redis realtime writer/API。
7. 本地启动 gateway用 ECS 转发流量验证。
8. 构建 Linux 镜像,部署 ECS。
9. 按验收标准逐项验证并记录证据。

View File

@@ -0,0 +1,552 @@
# 车辆数据管理中台设计 Spec
## 目标
建设一套独立的车辆数据管理中台,放在新项目目录 `vehicle-data-platform`。中台包含前端和后端,前端使用 Semi UI整体风格高级、简洁、接近大厂数据控制台后端使用 Go BFF 聚合现有车辆接入链路的数据能力。系统一期目标是让业务和运营人员可以稳定完成查车、看实时状态、查历史、追溯 RAW、看里程和定位数据质量问题同时给平台运维提供只读链路健康态势。
## 产品定位更新:一个车辆服务
中台的第一性原则是“一个车辆服务”。GB32960、JT808、宇通 MQTT 是同一辆车的数据来源证据,不是三个并列的业务产品。所有主页面都应先回答车辆层面的问题,再在需要排障、对账、追溯时下钻到协议来源。
用户输入 VIN、车牌、手机号或来源标识后系统需要尽量解析到同一个车辆上下文并在实时监控、轨迹回放、历史查询、告警事件、统计分析之间保持这个上下文。协议只作为来源标签、可信度解释和 RAW 证据入口出现。
## 一期定位
一期优先服务运营、业务和数据质量人员,核心问题是:
- 车辆是否在线。
- 车辆在哪里。
- 最新状态是什么。
- 最近是否断链。
- 每日和区间里程是否可信。
- 原始报文和解析字段是否可追溯。
- 告警事件是否能追到影响车辆和证据。
- VIN、车牌、手机号、OEM 绑定是否正确。
链路运维能力作为辅助模块内嵌,展示 Gateway、NATS、Kafka、Redis、TDengine、MySQL 和 capacity-check 的关键健康状态,但一期不做复杂监控平台和告警工单闭环。
## 非目标
一期不做以下内容:
- 复杂多租户权限体系。
- 自定义大屏编排。
- 复杂工单闭环。
- 任意字段计算平台。
- 完整 BI 报表系统。
- 复杂三维轨迹或大屏动画。
- 对现有 Go gateway、NATS bridge、history writer、stat writer、realtime API 做破坏性改造。
## 项目结构
新建项目目录:
```text
vehicle-data-platform/
apps/
web/
api/
docs/
product-spec.md
api-contract.md
deployment.md
deploy/
systemd/
nginx/
scripts/
```
### 前端技术栈
- React
- TypeScript
- Vite
- Semi UI
- ECharts
- 地图组件,一期可先用轻量地图封装,后续再替换为正式地图供应商
### 后端技术栈
- Go
- 标准库 HTTP 或轻量路由
- MySQL driver
- Redis client
- TDengine driver
- HTTP client 聚合现有 health、metrics、capacity-check 输出
后端作为 BFF不让前端直接连接 MySQL、Redis、TDengine 或现有 realtime API。
## 信息架构
左侧导航围绕车辆服务组织,一期可以先保留现有 7 个一级模块,但命名和内容要向以下目标靠拢。
### 1. 运营总览
路径:`/dashboard`
能力:
- 在线车辆数。
- 今日活跃车辆。
- 今日上报量。
- 异常车辆数。
- Kafka lag。
- 链路健康状态。
- 车辆服务状态分布:健康、降级、离线、身份缺失、暂无来源。
- 来源覆盖分布:单源车辆、多源车辆、暂无来源车辆。
- 最新车辆分布地图小窗。
- 最近异常列表:断链、未绑定、里程异常、解析失败、跨来源不一致。
### 2. 车辆服务
路径:`/vehicles`
能力:
- 表格展示车牌、VIN、手机号、OEM、来源证据、在线来源数、车辆服务状态、最后上报、最近位置、绑定完整度。
- 支持 VIN、车牌、手机号、OEM 搜索。
- 支持服务状态、来源覆盖、在线状态、绑定完整度筛选。
- 支持查看详情、导出。
- 支持导入绑定入口的一期展示:提供导入说明和模板下载按钮,不提交数据;正式写入流程不纳入一期验收。
### 3. 实时监控
路径:`/realtime`
能力:
- 表格视图和地图视图切换。
- 展示 VIN、车牌、来源证据、速度、SOC、总里程、最后上报时间、位置。
- 地图按车辆服务状态和在线状态分色,协议只用于点位详情里的来源证据。
- 支持只看在线车辆、异常车辆、断链车辆和多源不一致车辆。
- 点击车辆后打开右侧抽屉展示最新实时字段摘要、来源新鲜度和跳转入口。
### 4. 车辆详情
路径:`/vehicles/:vin`
能力:
- 顶部身份卡车牌、VIN、OEM、协议、在线状态。
- Tabs
- 最新状态。
- 历史位置。
- RAW 报文。
- 里程统计。
- 数据质量。
- 多协议对比:同一车辆在 GB32960、JT808、YUTONG_MQTT 下的最新时间、字段完整度、里程差异。
### 5. 轨迹回放
路径:`/history`
能力:
- 查询条件VIN/车牌/手机号、时间范围、来源筛选。
- 地图轨迹播放,支持起点、终点、当前播放点和速度。
- 表格与地图联动:点击点位能看到该点时间、位置、速度、里程和来源。
- 可跳转 RAW 证据、统计分析和告警事件。
### 6. 历史查询
路径:`/history`
能力:
- Tab历史位置、RAW 报文、解析字段。
- 查询条件VIN、协议、时间范围、消息类型、字段筛选。
- RAW 支持分页、字段裁剪、是否包含 payload、JSON 详情查看。
- 解析字段使用扁平化字段名,支持选择字段缩小响应体。
### 7. 告警事件
路径:`/quality`,后续可独立为 `/alerts`
能力:
- 展示来源断链、未绑定、无 VIN、无车牌、无手机号、RAW 解析失败、里程突变、跨来源不一致。
- 每条告警必须能跳回车辆详情、轨迹回放、历史 RAW 或统计证据。
- 一期支持人工确认和状态展示,通知发送作为后续阶段。
### 8. 统计分析
路径:`/mileage`
能力:
- 日里程查询。
- 区间里程查询。
- 协议来源对比。
- 异常标记:总里程倒退、突增、缺失、跨协议不一致。
- 在线率、数据完整率和来源一致性作为后续统计卡片。
### 9. 运维质量
路径:`/quality`
能力:
- 断链车辆。
- 未绑定车辆。
- 无 VIN、无车牌、无手机号车辆。
- RAW 解析失败。
- capacity-check 结果。
- Kafka/NATS/TDengine/Redis/MySQL 关键状态。
## 高德地图设计
地图供应商使用高德地图。用户提供了 Web JS 和 API 服务两类 key系统实现时需要按用途分开
- Web JS key 只通过前端运行时配置注入,例如 `window.__LINGNIU_PLATFORM_CONFIG__.amap.webJsKey`,不写死在 TypeScript 源码。
- 安全密钥或代理服务地址同样通过运行时配置注入。
- API 服务 key 放在后端或 ECS 环境变量中,只用于后端地理编码、行政区划或后续服务端地图能力。
- 前端没有拿到可用 key 时,`VehicleMap` 使用坐标预览降级,不阻塞查车、查轨迹和查告警。
地图一期覆盖三类业务场景:
- 实时监控:多车点位、在线状态、异常状态、点位详情。
- 轨迹回放:折线、起终点、当前播放点、表格联动。
- 告警事件:告警车辆定位和影响范围提示。
## 跨页面车辆上下文
全局搜索解析到车辆后,需要把 VIN、车牌、手机号、来源和服务状态作为页面上下文
- 车辆服务行可以跳转车辆档案。
- 实时监控车辆可以跳转轨迹回放、历史查询、统计分析和告警事件。
- 轨迹点可以跳转同日里程统计和 RAW 证据。
- 里程异常可以跳转对应日期轨迹和 RAW 证据。
- 告警事件可以跳转车辆档案、历史证据和统计证据。
这一点是产品一致性的硬约束。页面之间不应只靠用户手动复制 VIN。
## 视觉与交互设计
整体视觉使用 Semi UI 原生组件为基础,避免大屏炫技风。目标是云控制台、数据控制台、飞书后台一类的专业感。
### 设计原则
- 灰白背景,深色正文,蓝青强调色。
- 红橙只用于异常、告警、风险。
- 表格优先,不用卡片堆满页面。
- KPI 卡片少而准。
- 右侧抽屉承载详情,不频繁跳整页。
- RAW 和 parsed JSON 使用代码查看器或结构化 JSON Tree。
- 地图是业务辅助,不作为首页背景。
- 所有列表页都有稳定筛选区、刷新时间和导出入口。
### 页面骨架
- 顶部栏:系统名、全局车辆搜索、环境状态、刷新时间、用户入口。
- 左侧栏7 个一级导航。
- 主体区:页面标题、筛选/操作区、数据区。
- 右侧抽屉车辆摘要、RAW 详情、异常解释。
## 后端 BFF 设计
后端统一暴露 `/api/*`,并托管前端静态文件。
### API 响应格式
成功:
```json
{
"data": {},
"traceId": "string",
"timestamp": 1783080000000
}
```
分页:
```json
{
"data": {
"items": [],
"total": 100,
"limit": 50,
"offset": 0
},
"traceId": "string",
"timestamp": 1783080000000
}
```
错误:
```json
{
"error": {
"code": "QUERY_FAILED",
"message": "查询失败",
"detail": "具体错误"
},
"traceId": "string",
"timestamp": 1783080000000
}
```
### 后端模块
#### vehicle
职责:
- 车辆台账。
- 身份绑定。
- VIN、车牌、手机号、OEM 查询。
数据源:
- MySQL `vehicle_identity_binding`
- MySQL `vehicle_realtime_snapshot``vehicle_realtime_location` 用于补充在线状态和最新上报。
#### realtime
职责:
- 实时位置。
- 实时快照。
- 单车当前状态。
- 单车协议 realtime raw 摘要。
数据源:
- MySQL `vehicle_realtime_snapshot`
- MySQL `vehicle_realtime_location`
- Redis `vehicle:realtime-raw:{protocol}:{vin}`
- Redis online key 族。
#### history
职责:
- 历史位置。
- RAW frame 查询。
- RAW 字段过滤。
数据源:
- TDengine `vehicle_locations`
- TDengine `raw_frames`
- TDengine `raw_frame_payload_chunks`
#### mileage
职责:
- 日里程。
- 区间里程。
- 里程异常解释。
数据源:
- MySQL `vehicle_daily_mileage`
#### quality
职责:
- 无 VIN。
- 无车牌。
- 无手机号。
- 长时间未上报。
- 里程突变。
- 解析失败。
数据源:
- MySQL identity 和 realtime 表。
- TDengine raw_frames。
- Redis online。
#### ops
职责:
- capacity-check。
- 服务 readyz。
- Kafka/NATS/Redis/TDengine/MySQL 状态摘要。
数据源:
- 现有本机 health endpoint。
- 现有 metrics endpoint。
- `/opt/lingniu-go-native/current/capacity-check` 输出。
## 一期 API
### Dashboard
- `GET /api/dashboard/summary`
返回:
- 车辆在线数。
- 今日活跃数。
- 今日上报量。
- 协议分布。
- 异常计数。
- 链路健康摘要。
### Vehicles
- `GET /api/vehicles`
- `GET /api/vehicles/:vin`
- `GET /api/vehicles/:vin/realtime`
- `GET /api/vehicles/:vin/protocols`
### Realtime
- `GET /api/realtime/locations`
- `GET /api/realtime/snapshots`
- `GET /api/realtime/raw/:protocol/:vin`
### History
- `GET /api/history/locations`
- `POST /api/history/raw-frames/query`
RAW query 使用 POST避免字段过滤数组导致 URL 过长。
### Mileage
- `GET /api/mileage/daily`
- `GET /api/mileage/range`
### Quality
- `GET /api/quality/issues`
- `GET /api/quality/unbound-vehicles`
- `GET /api/quality/offline-vehicles`
### Ops
- `GET /api/ops/health`
- `GET /api/ops/capacity`
## 数据源映射
### MySQL
- `vehicle_identity_binding`:车辆身份事实表。
- `vehicle_realtime_snapshot`:每协议每 VIN 最新轻量快照。
- `vehicle_realtime_location`:每协议每 VIN 最新位置。
- `vehicle_daily_mileage`:每日里程统计。
- `jt808_registration`JT808 注册与鉴权辅助信息。
### Redis
- `vehicle:realtime-raw:{protocol}:{vin}`:最新完整协议 parsed 状态。
- `vehicle:online:*`:在线状态。
- `vehicle:last_seen`:最近活跃排序。
- `vehicle:rt-kv:*`:扁平化实时字段。
### TDengine
- `raw_frames`RAW 证据层。
- `raw_frame_payload_chunks`:超长 payload 分片。
- `vehicle_locations`:历史位置。
## 权限设计
一期做轻量鉴权:
- 登录页。
- 单一管理员账号或配置文件 token。
- 前端请求带 bearer token。
- 后端校验 token。
后续再扩展 RBAC
- 业务只读。
- 运维只读。
- 管理员。
## 部署设计
ECS 部署:
- 后端端口:`20300`
- 后端托管前端静态文件。
- `/api/*` 是后端接口。
- `/` 和前端路由返回静态入口。
- systemd unit`lingniu-vehicle-platform.service`
访问地址:
```text
http://115.29.187.205:20300
```
## 环境变量
后端需要:
- `HTTP_ADDR=:20300`
- `MYSQL_DSN`
- `REDIS_ADDR`
- `REDIS_USERNAME`
- `REDIS_PASSWORD`
- `REDIS_DB`
- `TDENGINE_DSN`
- `TDENGINE_DATABASE=lingniu_vehicle_ts`
- `CAPACITY_CHECK_BIN=/opt/lingniu-go-native/current/capacity-check`
- `AUTH_TOKEN`
## 验收标准
一期完成时必须满足:
1. 新项目目录 `vehicle-data-platform` 存在。
2. 前端使用 Semi UI。
3. UI 有完整导航、统一布局、统一视觉规范。
4. 可以查看车辆台账。
5. 可以查询实时位置和实时状态。
6. 可以进入单车详情。
7. 可以查询历史位置。
8. 可以查询 RAW frame并支持字段裁剪。
9. 可以查看日里程和区间里程。
10. 可以查看数据质量问题。
11. 可以查看链路健康状态。
12. 后端 API 返回统一响应结构。
13. 前端核心列表、详情和查询来自真实后端 API。
14. ECS 部署后可通过 `http://115.29.187.205:20300` 访问。
15. 后端 systemd 服务健康。
16. 浏览器验证主要页面可访问、筛选可用、详情抽屉可用。
## 测试策略
### 后端
- API handler 单元测试。
- query 参数解析测试。
- response envelope 测试。
- 数据源 repository 测试。
- 健康检查测试。
### 前端
- TypeScript build。
- 页面 smoke test。
- API client 测试。
- 主要页面空态、加载态、错误态检查。
### 部署
- 本地前端 build。
- 后端 go test。
- ECS systemd 启动。
- `/api/ops/health` 返回正常。
- 浏览器打开首页、车辆台账、实时监控、历史查询。
## 后续演进
- RBAC 权限。
- 车辆绑定导入写入流程。
- 告警通知。
- 轨迹回放。
- BI 报表。
- 自定义字段分析。
- 运维监控大盘。

View File

@@ -0,0 +1,161 @@
# Multi-Source Mileage Statistics Design
## Goal
Vehicle mileage statistics must keep every protocol source that reports data, while still exposing one reliable default result for BI and API consumers. The design separates source facts from elected business results so multi-source JT808, GB32960, and Yutong MQTT data can be audited, corrected, and re-elected without losing raw source evidence.
## Current Problem
The current `vehicle_daily_mileage` table stores one row per `vin + stat_date + protocol`. That prevents duplicate counting, but it also forces source election too early. When multiple source platforms report the same VIN, mixing them can create false mileage jumps; selecting only one source can undercount when a reliable alternative exists.
The observed example was `LA9GG64L7PBAF4001` on `2026-07-08` for JT808. Two source IPs reported different total mileage ranges. The correct model is to retain each source as its own candidate and then elect the recommended candidate for default queries.
## Tables
### `vehicle_data_source`
Stores source platform metadata discovered from incoming data and later maintained by operators.
Identity:
- `protocol`
- `source_ip`
Fields:
- `id`
- `protocol`
- `source_ip`
- `latest_source_endpoint`
- `platform_name`
- `trust_priority`
- `enabled`
- `first_seen_at`
- `latest_seen_at`
- `remark`
- `created_at`
- `updated_at`
Rules:
- `protocol + source_ip` is unique.
- Ports are not part of source identity because sender ports change often.
- Runtime may update `latest_source_endpoint`, `first_seen_at`, and `latest_seen_at`.
- Runtime must not overwrite manually maintained `platform_name`, `trust_priority`, `enabled`, or `remark`.
### `vehicle_daily_mileage_source`
Stores one daily candidate per source.
Identity:
- `vin`
- `stat_date`
- `protocol`
- `source_key`
Fields:
- `vin`
- `stat_date`
- `protocol`
- `source_key`
- `source_ip`
- `source_endpoint`
- `phone`
- `device_id`
- `platform_name`
- `first_total_mileage_km`
- `latest_total_mileage_km`
- `daily_mileage_km`
- `sample_count`
- `first_event_time`
- `latest_event_time`
- `quality_status`
- `quality_reason`
- `is_selected`
- `created_at`
- `updated_at`
Rules:
- Candidate rows are facts by source. They should not be deleted just because they are not selected.
- `source_key` is normalized as protocol-specific device identity plus source IP. For JT808 that is usually `phone@source_ip`.
- A candidate can be marked invalid without being removed.
- Candidate statistics never mix different `source_key` values.
### `vehicle_daily_mileage`
Keeps the existing default query surface for BI and APIs.
Rules:
- One row per `vin + stat_date + protocol`.
- The row is a projection from one selected `vehicle_daily_mileage_source` row.
- It stores selected source metadata for fast default queries.
## Election
The default selected source is chosen from candidate rows. Selection priority:
1. Source must be enabled in `vehicle_data_source`.
2. Prefer lower `trust_priority` when configured.
3. Prefer candidates with a valid same-source previous-day baseline.
4. Reject impossible deltas, such as negative mileage or mileage above the configured daily maximum.
5. Prefer more samples when priority and quality are tied.
6. Prefer later `latest_event_time` when still tied.
If trusted source A is missing on a day, source B can be selected only if B has its own previous-day baseline. The system must not calculate `B today - A yesterday`.
## Data Flow
1. Gateway/history/realtime parsing continues to emit protocol fields without non-protocol synthetic fields.
2. Stats writer extracts protocol-specific total mileage from the existing fields event.
3. Stats writer upserts `vehicle_data_source` by `protocol + source_ip`.
4. Stats writer upserts `vehicle_daily_mileage_source` by `vin + stat_date + protocol + source_key`.
5. Election updates `vehicle_daily_mileage` from candidates.
6. History backfill can rebuild candidates from TDengine raw frames and then rebuild final projections.
## API Behavior
Default daily mileage API continues to read `vehicle_daily_mileage`.
Future query extension:
- `includeCandidates=true` returns all candidates for each selected row.
- Candidate rows include `platform_name`, `source_ip`, `source_endpoint`, `quality_status`, and `quality_reason`.
## Edge Cases
- Multiple ports from the same platform source are collapsed by source IP.
- A new source with no previous-day baseline is stored as a candidate with `NO_PREVIOUS_BASELINE` and is not selected by default.
- A source that reports a total-mileage reset is stored as a candidate with an invalid quality status until a clear policy is applied.
- If all candidates are invalid, no final row is projected for that VIN/day/protocol.
- Manual platform names and priority changes should allow re-election without reparsing raw data.
## Performance
- Real-time writes are small upserts keyed by indexed dimensions.
- Default BI/API reads stay fast because they read the final projection table.
- Candidate/audit queries are opt-in.
- Backfill aggregates by source and date instead of writing every raw frame to MySQL.
## Testing
Automated tests should cover:
- Source IP extraction and endpoint normalization.
- Source table upsert preserving manually maintained fields.
- Candidate stats never mixing different source keys.
- Election choosing the continuous source for `LA9GG64L7PBAF4001`-style data.
- Election skipping `B today - A yesterday`.
- Backfill rebuilding candidates and final projection.
## Rollout
1. Add schemas and repository helpers.
2. Write candidate rows in stats writer and backfill.
3. Add election projection into `vehicle_daily_mileage`.
4. Rebuild `2026-07-08` JT808 statistics and verify `LA9GG64L7PBAF4001`.
5. Deploy to ECS and monitor stat-writer lag and write errors.

View File

@@ -1,202 +0,0 @@
# Target Architecture
This document records the production target for `lingniu-vehicle-ingest`.
The project should stay small at runtime: protocol apps ingest and publish,
history consumes and indexes, analytics derives metrics, and business systems
read through explicit APIs or Kafka.
Only supported message backbone: Kafka.
## Active Scope
Active production protocols:
| Protocol | Runtime app | Event topic | Raw topic |
|---|---|---|---|
| GB/T 32960 | `gb32960-ingest-app` | `vehicle.event.gb32960.v1` | `vehicle.raw.gb32960.v1` |
| JT/T 808 | `jt808-ingest-app` | `vehicle.event.jt808.v1` | `vehicle.raw.jt808.v1` |
| Yutong MQTT | `yutong-mqtt-app` | `vehicle.event.mqtt-yutong.v1` | `vehicle.raw.mqtt-yutong.v1` |
Xinda Push source, Maven profile, deployment bindings, and history consumers
are removed. New optimization work should target the active protocols above.
JSATL12 attachment upload is also outside the default production reactor. Keep
it available through the `optional-attachments` profile until an attachment app
is explicitly deployed.
## Goals
- Keep protocol IO, raw archive, history, latest-state, and statistics as
separate responsibilities.
- Publish normalized Kafka envelopes with VIN as the partition key whenever a
VIN is known.
- Store complete raw payload metadata and parsed JSON in TDengine `raw_frames`.
- Store query-friendly location rows separately from raw payload JSON.
- Keep derived statistics in metric tables, not protocol-specific daily tables.
- Use Redis only for optional latest-state APIs that can be rebuilt from Kafka.
- Keep optional compatibility modules outside the production hot path.
## Non-Goals
- Ingest apps do not write business tables.
- Ingest apps do not serve history query APIs.
- Redis is not the long-term historical store.
- File-based event indexes are not part of the current build surface.
- `telemetry_fields` parsing and field-trend APIs are owned by a separate
field parsing service. This project stores RAW parsed JSON plus compact
location rows and does not expose a telemetry-fields history API.
## Runtime Responsibilities
### Protocol Apps
Apps:
- `gb32960-ingest-app`
- `jt808-ingest-app`
- `yutong-mqtt-app`
Responsibilities:
- Accept protocol traffic.
- Decode, authenticate, acknowledge, and maintain sessions where the protocol
requires it.
- Archive raw bytes through `sink-archive` when bytes are available.
- Publish parsed event envelopes to Kafka.
- Publish raw envelopes to Kafka.
- Resolve vehicle identity through the shared MySQL identity binding table.
Protocol apps must stay independent from TDengine history readers, Redis state
repositories, and statistic calculators.
### Event Contract
The Kafka sink owns the protobuf envelope and consumer plumbing.
Consumers should use `EnvelopeConsumerProcessor` so invalid protobuf, skipped
envelopes, and downstream failures are converted to structured results and DLQ
records instead of blocking a partition.
Core identity fields:
| Field | Meaning |
|---|---|
| `event_id` | Parsed event idempotency key |
| `trace_id` | Cross-service trace id |
| `vin` | Vehicle id and Kafka partition key |
| `source` | Source protocol, for example `GB32960`, `JT808`, `YUTONG_MQTT` |
| `event_time_ms` | Device event time |
| `ingest_time_ms` | Platform receive time |
| `raw_uri` | `archive://...` reference to raw bytes when available |
Full-field telemetry uses `TelemetrySnapshot` fields with stable keys such as
`total_mileage_km`, `speed_kmh`, `longitude`, and `latitude`.
### History App
App: `vehicle-history-app`
Responsibilities:
- Consume active protocol event topics.
- Consume active protocol raw topics.
- Write raw frame rows into TDengine `raw_frames`.
- Write compact location rows into TDengine location tables.
- Keep `payloadJson.parsed` on raw rows for full raw inspection.
- Expose paged history APIs for raw frames and locations.
- Keep specialized APIs off by default unless explicitly enabled.
Production default:
- `TDENGINE_HISTORY_ENABLED=true` in deployment.
- `EVENT_FILE_STORE_ENABLED=false`.
File-based event indexes have been removed; this app should keep history
queries on TDengine and raw replay on `archive://...` references.
### Analytics App
App: `vehicle-analytics-app`
Responsibilities:
- Consume JT808 event envelopes.
- Calculate 808 daily mileage from the reported GPS total mileage only:
```text
daily_mileage_km = max_total_mileage_km - min_total_mileage_km
calculation_method = JT808_TOTAL_MILEAGE_DIFF
```
- Store the metric in `vehicle_stat_metric` with
`metric_key = daily_mileage_km`; the local-day minimum and maximum GPS total
mileage used for the subtraction stay on the same metric row as calculation
source columns.
Runtime state: none outside `vehicle_stat_metric`. There is no JT808-specific
daily mileage table. Restart recovery reads the same metric row.
### Latest State
Module: `vehicle-state-service`
This module is outside the default production reactor. Build it with
`-Poptional-latest-state` when Redis latest-state APIs are explicitly needed.
Responsibilities:
- Consume normalized Kafka envelopes.
- Maintain latest state, location, and safety snapshots in Redis.
- Serve latest-state APIs for operational screens.
Redis state must be rebuildable from Kafka replay and should not be used as the
source of truth for historical queries.
## Data Flow
Default production flow:
```mermaid
flowchart LR
vehicle["Vehicles / Platforms"] --> ingest["Protocol apps"]
ingest --> archive["sink-archive<br/>raw bytes"]
ingest --> kafka["Kafka<br/>event + raw envelopes"]
kafka --> history["vehicle-history-app"]
kafka --> analytics["vehicle-analytics-app"]
history --> tdengine["TDengine<br/>raw_frames + locations"]
analytics --> mysql["MySQL<br/>vehicle_stat_metric"]
```
Optional latest-state flow, enabled only with `-Poptional-latest-state`, consumes
the same Kafka envelopes and rebuilds Redis snapshots outside the default
production deployment.
## Failure Handling
- Protocol apps continue accepting traffic when history, analytics, or Redis
consumers are unavailable.
- Kafka producer failures are handled by sink retry/circuit-breaker behavior and
DLQ topics where configured.
- History consumers should commit offsets only after TDengine writes succeed.
- Analytics daily mileage state can recover from `vehicle_stat_metric` or Kafka
replay.
- Raw archive failures must be observable, but event publication and history
indexing remain separate concerns.
## Acceptance Criteria
- Active production deployment contains GB32960, JT808, Yutong MQTT,
vehicle-history, and vehicle-analytics apps.
- Xinda Push source and deployment bindings are removed.
- JSATL12 is absent from the default Maven reactor and available through
`optional-attachments`.
- vehicle-state-service is absent from the default Maven reactor and available
through `optional-latest-state`.
- raw-archive-store prototype has been removed; raw bytes are written through
`sink-archive`.
- History APIs read from TDengine; file-based event indexes are not maintained.
- Raw frame queries can return complete parsed JSON through `payloadJson.parsed`.
- Location queries page over compact rows and reference raw records instead of
duplicating full raw JSON.
- JT808 daily mileage is stored only in `vehicle_stat_metric`.
- Runtime state needed for 808 mileage is also in `vehicle_stat_metric`.
- Build validation covers the active modules and their composition tests.

View File

@@ -0,0 +1,231 @@
# 车辆数据中台一期现状分析与总体方案
> 结论日期2026-07-14
> 范围:只完成现状分析和方案设计,不进入 V2 功能开发或 ECS 部署。
> 代码依据:`/Users/lingniu/project/ai-coding/ln-bi`、`vehicle-data-platform`、`go/vehicle-gateway`,以及 `docs/architecture`、`docs/ops` 中的生产说明。
## 1. 结论摘要
当前数据接入链路已经具备可用的生产底座GB32960、JT808、YUTONG_MQTT 经 NATS/Kafka 分流后,分别形成 Redis 当前态、TDengine RAW/位置历史、MySQL 身份/实时/日里程投影;现有平台 BFF 也已提供车辆、实时位置、轨迹点、RAW、里程、质量和运维健康等只读接口。
但现有前端不是可直接扩展的一期产品:生产入口加载的是约 6000 行的 `PrototypeApp.tsx`,仍混有 mock 数据和“接口缺失时展示能力边界”的原型逻辑;另一个模块化 `App.tsx` 并非当前入口。现有地图每次最多创建 500 个普通 Marker没有聚合、MassMarks/Canvas 图层和视口查询,无法满足 1 万辆级监控。所谓告警接口实际上是质量问题接口的别名,没有规则、事件、处理记录和通知持久化。
建议丢弃现有原型页面的产品实现,但保留经验证的数据适配、领域工具、测试、运行时配置和部署脚手架,建立独立 V2。前端沿用 React + TypeScript + Vite吸收 `ln-bi` 的视觉语言、壳层、鉴权思想和基础组件,但采用正式路由、查询缓存、虚拟表格和独立地图 SDK 层。后端保留现有接入链路,通过平台 BFF 和新增的告警/导出 worker 补齐业务能力,不让浏览器直接访问 Redis、TDengine 或 MySQL。
## 2. 当前前端技术栈与目录
### 2.1 `ln-bi` 参考项目
| 项目 | 当前实现 | 评价 |
| --- | --- | --- |
| 框架 | React 19、React DOM 19、TypeScript 5.8、Vite 6.2 | 可作为 V2 目标栈参考;不建议仅为对齐版本立刻升级已有平台依赖 |
| 样式 | Tailwind CSS 4、少量 CSS 变量 | 视觉语言清晰,但业务组件 class 较分散,需要提炼 token 和组件配方 |
| 图标/动效 | lucide-react、Motion | 可复用设计语言;监控页面动效应克制并支持 reduced motion |
| 图表 | Recharts 3.8 | 适合常规 BI百万点、多 Y 轴时序更推荐现平台已有 ECharts 5 |
| 地图 | `@amap/amap-jsapi-loader` + AMap JS API 2.0 | 加载和实例生命周期可借鉴;目前只是热力图业务组件,不是通用地图 SDK |
| API | 原生 `fetch` + JWT 注入 | 可借鉴鉴权流程;错误类型、超时、重试、取消、缓存仍不足 |
| 服务端 | Hono、MySQL/Postgres、JWT、XLSX | 属于 `ln-bi` 自身 BFF不应直接搬入车辆接入 Go 数据面 |
| 路由 | pathname + hash + `replaceState` 手写 | 适合少量模块,不适合可分享查询、详情层级和权限路由 |
| 状态 | React 本地状态/Context | 未使用全局状态库或服务端查询缓存 |
主要目录:
```text
ln-bi/src/
├── App.tsx # 模块注册、权限门禁、懒加载
├── auth/ # jumpToken -> JWT、fetch token 注入
├── components/
│ ├── Shell.tsx # 桌面侧栏、移动底栏、模块切换、水印
│ ├── SearchSelect.tsx
│ ├── MultiSearchSelect.tsx
│ └── ui/surface.tsx # PageFrame、SurfaceCard、MetricTile、状态组件
├── modules/
│ ├── vehicle-heatmap/ # AMap 热力图和筛选/详情面板
│ ├── hydrogen-heatmap/
│ ├── mileage/ # 表格、详情弹窗、XLSX 导出
│ └── ...
├── server/ # Hono BFF、认证、DB 查询
├── shared/auth/roles.ts # 角色判断
└── index.css # Tailwind、主题变量、地图控件样式
```
### 2.2 当前车辆平台前端
目录为 `vehicle-data-platform/apps/web`,使用 React 18.3、TypeScript 5.7、Vite 6、Semi UI 2.71、ECharts 5.6。当前入口 `src/main.tsx` 加载 `PrototypeApp`,不是 `src/App.tsx`
```text
vehicle-data-platform/apps/web/src/
├── main.tsx # 当前生产入口 -> PrototypeApp
├── PrototypeApp.tsx # 大型原型单体,真实接口与 mock 逻辑混合
├── App.tsx # 未作为入口的模块化旧实现
├── api/ # 统一响应信封、类型和调用方法
├── components/ # VehicleMap、状态标签、空状态等
├── config/ # API/高德运行时配置
├── domain/ # 路由、导出、车辆查询等纯函数
├── integrations/amap.ts # AMap loader、坐标校验、逆地理编码
├── layout/AppShell.tsx
├── pages/ # 旧模块化页面
├── prototype/ # mock、真实数据适配、view model、字段映射
└── styles/ # token、全局样式、原型样式
```
V2 的原则是“页面重建、能力甄别复用”:不复制 `PrototypeApp.tsx``prototype.css`;保留可独立测试的 `api``domain``integrations`、真实数据归一化逻辑及其测试,再按新接口契约重构。
## 3. 可复用能力清单
| 能力 | 来源 | 复用决策 | 说明 |
| --- | --- | --- | --- |
| 80px 深色侧栏、顶部面包屑、水印、移动底栏 | `ln-bi/components/Shell.tsx` | 设计复用、代码重构 | V2 桌面优先,应改为正式路由和可折叠侧栏 |
| PageFrame、SurfaceCard、MetricTile、SegmentedNav | `ln-bi/components/ui/surface.tsx` | 可迁移后组件化 | 统一 token、尺寸和可访问性后复用 |
| SearchSelect、MultiSearchSelect | `ln-bi/components` | 交互参考 | 当前仅接收字符串数组缺少远程搜索、虚拟列表、label/value 和键盘完整性 |
| 加载、空、错误状态 | 两项目 | 合并重构 | 建立统一 `AsyncState`,禁止页面各自拼接 |
| JWT/jumpToken 门禁、角色判断 | `ln-bi/auth``shared/auth` | 复用认证协议思想 | token 仍应保存在 session权限改为路由/操作级声明,后端必须二次校验 |
| AMap loader、实例销毁、运行时密钥 | 两项目 | 合并为通用 SDK | 统一 loader promise、插件注册、安全代理和错误状态 |
| 热力图数据归一化 | `ln-bi/vehicle-heatmap` | 可复用算法 | `log1p/sqrt` 强度变换适合密度层,不等于车辆状态点图层 |
| 表格/详情弹窗/XLSX | `ln-bi/mileage` | 交互与导出格式参考 | 一期大数据导出必须后端异步,不能沿用浏览器全量 XLSX |
| ECharts | 当前车辆平台 | 保留 | 更适合多 Y 轴、dataZoom、断点、阶梯线和大数据采样 |
| `api/client.ts` 响应信封和 traceId | 当前车辆平台 | 扩展复用 | 增加鉴权、AbortSignal、超时、错误码、幂等键和查询缓存 |
| 车辆 lookup、字段归一化、CSV 纯函数及测试 | 当前车辆平台 | 审核后复用 | 去掉 mock fallback统一到 V2 DTO/领域模型 |
| Semi UI 页面组件 | 当前车辆平台 | 暂不作为 V2 默认 | 与 `ln-bi` Tailwind 视觉体系并存会形成双设计系统V2 启动前只选一套基础组件策略 |
不存在可直接复用的“通用图表组件”或“通用高性能表格组件”。`ln-bi` 的 Recharts 与页面耦合,当前平台的 ECharts 也需要建立 `TimeSeriesChart`、轴/单位/空值策略和采样契约;表格需新增服务端分页、列配置、固定列和虚拟滚动封装。
## 4. 高德地图当前封装方式
### 4.1 `ln-bi`
`AmapHeatmapCanvas.tsx` 动态导入 `@amap/amap-jsapi-loader`,写入 `_AMapSecurityConfig.securityJsCode`,加载 AMap 2.0 的 HeatMap、ToolBar、Scale 插件;创建白色地图和 HeatMap点击地图回传经纬度数据变化时调用 `setDataSet`,焦点变化时计算 bounds卸载时销毁实例。
优点是 loader 官方、生命周期完整、地图与侧面板分屏清楚。限制是车辆和加氢模块各有近似实现,密钥直接下发前端,未封装 Marker/聚合/信息窗/轨迹/播放,也没有实例共享和视口事件节流。
### 4.2 当前车辆平台
`integrations/amap.ts` 自行加载 `https://webapi.amap.com/loader.js`,按插件集合缓存 Promise支持 Scale、Geocoder、坐标合法性检查、URI Marker 链接和浏览器逆地理编码;`appConfig.ts``window.__LINGNIU_APP_CONFIG__` 或 Vite 环境变量读取 Web JS key、安全代码/代理和 API base URL。
`VehicleMap.tsx` 在组件内创建 Map、Marker、Polyline重绘时清除 overlay最多截取 500 个点,并在未配置地图时提供坐标预览。它适合验证和小规模页面,不适合一期:每点 HTML button、没有聚合、状态图层、InfoWindow 管理、车辆朝向、轨迹抽稀/播放或视口查询。
### 4.3 V2 建议封装
建立 `features/map-sdk`,分为:
- `AMapProvider/useAMap`:唯一 loader、插件按需加载、实例注册、resize/destroy。
- `BaseMap`:纯地图容器,只接收中心、缩放、样式和事件。
- `VehiclePointLayer`:小规模用 Marker规模化优先 MarkerCluster/LabelsLayer/MassMarks 或 Canvas/WebGL 能力;详情卡只保留一个 InfoWindow。
- `TrackLayer`Polyline、起终点、停车/告警节点和移动标记;播放状态在业务 hook不绑入地图实例。
- `HeatmapLayer``GeocoderService``MapControls`:独立能力。
- 视口查询使用 `bounds + zoom + filterHash`;移动结束后防抖请求服务器聚合结果,前端不一次拉取全量 1 万点 DOM。
高德 Web JS key 可由运行时配置下发,但安全密钥优先使用服务端安全代理;逆地理编码优先走现有 `/api/map/reverse-geocode`,避免泄露 REST key并便于限流/缓存。
## 5. 当前车辆相关接口
现有平台 BFF 的有效能力如下;详细缺口见 `vehicle-data-platform-api-gap.md`
| 领域 | 接口 | 当前用途 |
| --- | --- | --- |
| 总览 | `GET /api/dashboard/summary` | 在线、今日活跃、帧数、问题数、协议/服务/链路统计 |
| 车辆 | `GET /api/vehicles``/api/vehicles/resolve``/api/vehicles/coverage``/summary` | 车辆列表、VIN/车牌/手机号解析、来源覆盖 |
| 单车 | `GET /api/vehicle-service``/summary``/overview``POST /overviews` | 聚合身份、实时、历史、RAW、里程、质量证据 |
| 实时 | `GET /api/realtime/vehicles``/api/realtime/locations` | VIN 聚合实时状态和每协议最新位置 |
| 历史 | `GET /api/history/locations``GET/POST /api/history/raw-frames` | TDengine 位置点和 RAW 证据分页 |
| 里程 | `GET /api/mileage/summary``/api/mileage/daily` | MySQL 日里程统计 |
| 在线 | `GET /api/statistics/online-summary``/online-vehicles` | 当前固定口径的在线统计和车辆状态 |
| 数据质量 | `GET /api/quality/summary``/issues``/notification-plan` | 由当前数据推导的质量信号和静态通知方案 |
| 告警兼容 | `GET /api/alert-events/summary``/alert-events``/notification-plan` | 实际是上述质量接口别名,不是业务告警事件 |
| 地图 | `GET /api/map/reverse-geocode` | 服务端高德逆地理编码 |
| 运维 | `GET /api/ops/health``/source-readiness` | 数据源可写性、Kafka lag、连接/Redis key、发布版本和接入准备度 |
同时Go `realtime-api` 还提供面向底层表的查询接口和 OpenAPI但 V2 浏览器应统一访问平台 BFF避免前端绑定多个端口与底层存储模型。
## 6. 数据职责概览
- TDengine `lingniu_vehicle_ts``raw_frames` 是原始接收与扁平解析证据,`raw_frame_payload_chunks` 保存超长 payload`vehicle_locations` 保存 VIN 级位置、速度、方向、告警标志和总里程历史。它不是车辆档案或告警业务库。
- Redis DB 50`vehicle:latest:*``vehicle:realtime-raw:*``vehicle:rt-kv:*``vehicle:online:*``vehicle:protocols:*``vehicle:last_seen` 提供可过期、可重建的当前态。它不能作为历史、告警或导出任务事实源。
- MySQL `lingniu_vehicle_data`:保存 `vehicle`、多标识映射、JT808 注册鉴权、每协议实时快照/位置、来源配置和日里程事实/结果。现有 `vehicle` 档案很轻,只含 VIN、车牌、OEM、启用状态尚不足以支撑车型、公司、车辆类型等产品字段。
统一模型及新增模型详见 `vehicle-data-platform-data-model.md`
## 7. 一期缺口结论
必须新增或升级的核心能力:
1. 完整车辆档案、组织/车型/协议/接入厂家字典与首次接入事实。
2. 地图聚合查询、组合筛选、状态统计和可配置在线口径。
3. 包含方向、SOC、告警标志、停车点、异常点过滤和抽稀元数据的轨迹接口。
4. 动态指标目录,以及多车、多指标、聚合/抽稀、分页的通用时序查询。
5. MySQL 持久化告警规则、告警事件、处理记录、站内通知和告警计算 worker。
6. MySQL 异步导出任务、对象/本地文件存储、worker、进度与下载授权。
7. 接入状态字段:首次/最近上报、事件时间与接收时间延迟、最近消息类型、最近错误、动态在线阈值。
8. 真正的认证、菜单/操作权限审计;现车辆平台 BFF 尚未呈现完整权限体系。
## 8. 推荐一期路由与菜单
```text
/
├── /monitor 全局监控(默认首页)
├── /vehicles 车辆查询
│ └── /vehicles/:vin 单车数字档案
├── /tracks 轨迹查询与回放
├── /history 历史数据分析
│ └── /history/exports 导出任务
├── /alerts
│ ├── /alerts/events 告警事件
│ ├── /alerts/rules 告警规则
│ └── /alerts/inbox 站内通知
├── /access 车辆接入状态
└── /operations 运维质量(管理员)
```
侧栏一级菜单保持 6 个业务入口:全局监控、车辆中心、轨迹回放、历史分析、告警中心、接入管理;导出任务挂在历史分析二级菜单,运维质量按权限置底。详情、告警、轨迹之间用带 `returnTo` 的上下文跳转,查询条件同步 URL浏览器返回可恢复筛选、时间和选中车辆。
## 9. 推荐前后端整体架构
```mermaid
flowchart LR
UI["V2 React Web"] --> BFF["Platform API / BFF"]
BFF --> RT["Redis 当前态"]
BFF --> TS["TDengine 时序与 RAW"]
BFF --> DB["MySQL 业务事实"]
BFF --> MAP["高德 REST / 安全代理"]
ING["Gateway + NATS + Kafka Writers"] --> RT
ING --> TS
ING --> DB
K["Kafka fields/raw"] --> AW["Alert evaluator"]
AW --> DB
BFF --> EQ["Export queue"]
EQ --> EW["Export worker"]
EW --> TS
EW --> DB
EW --> FS["文件/对象存储"]
```
前端按 `app/router``shared``entities/vehicle``features``pages` 分层。服务端状态使用查询缓存库统一取消、去重、失效和后台刷新;少量 UI 状态用 React Context/局部 store。地图、图表、表格均只接收领域 DTO不直接处理协议字段名。
平台 BFF 负责统一鉴权、参数校验、分页、在线状态计算、跨源聚合、字段目录、限流、traceId 和 DTO禁止让前端分别拼 Redis/MySQL/TDengine。已有接入 writer 保持不变,新告警 worker 消费可重放 fields/raw 流,异步导出 worker 读取 TDengine/MySQL 并把任务状态写 MySQL。
## 10. 运维与最终 ECS 部署边界
已查阅 `docs/ops/vehicle-ingest-runbook.md``docs/ops/go-vehicle-ingest-memory.md``docs/architecture/production-data-plane-inventory.md``vehicle-data-platform/docs/deployment.md`。当前接入数据面采用 systemd 裸机 release 目录与 `current` 软链,平台使用 `/opt/lingniu-vehicle-platform/{releases,current,env}`HTTP 端口 20300部署后应验证 `/api/ops/health`、车辆查询和运行时配置。
V2 最终仍部署到当前 ECS但不应在分析阶段执行。实施完成后的发布顺序应是数据库向前兼容迁移 -> 新 worker默认禁用规则-> BFF -> 静态 Web -> 小流量健康验证 -> 启用告警/导出 worker。每一步保留旧 release 和可回滚软链;密钥只进 ECS 环境文件,不进仓库或构建产物。完整发布验收纳入路线图最后阶段。
## 11. 风险与待确认事项
| 优先级 | 风险/待确认 | 影响与建议 |
| --- | --- | --- |
| P0 | `vehicle` 档案缺车型、类型、公司、接入厂家等 | 先确定主数据来源和维护责任,不要从实时 JSON 猜档案 |
| P0 | 现有告警接口名与真实语义不符 | V2 使用 `/api/v2/alerts/*` 新契约,旧别名标记 deprecated避免误把质量信号当已处理事件 |
| P0 | 动态指标没有目录、单位、类型和协议映射 | 先落指标元数据;否则图表、规则和导出会各自硬编码 |
| P0 | 用户/角色来源尚未确认 | 明确复用 `ln-bi` jumpToken/JWT还是接入统一 SSO后端权限不可只靠菜单隐藏 |
| P1 | 在线阈值是全局、协议级还是车辆级 | 建议全局默认 + 协议覆盖;一期暂不做单车覆盖 |
| P1 | 轨迹异常过滤可能隐藏原始证据 | API 同时返回 raw/filtered 计数和算法版本,允许关闭过滤 |
| P1 | TDengine `raw_frames.parsed_json` 是动态字段唯一历史来源 | 通用指标查询先从 JSON 提取会有成本;高频稳定指标应按使用量逐步物化,而非一期全列化 |
| P1 | 单 ECS 同时承担接入、查询、告警和导出 | 导出/重查询必须有并发、时间范围、行数和资源配额;压测后再定 worker 并发 |
| P1 | 地图 1 万辆的插件/授权与浏览器性能 | 先完成 1k/10k 基准,确定 MarkerCluster、MassMarks 或 Canvas 方案和高德配额 |
| P1 | 站内通知实时方式 | 一期可 SSE + 轮询降级;若 ECS 反代不适合长连接,先使用增量轮询 |
| P2 | Semi UI 与 Tailwind 双设计系统 | V2 启动前确认唯一基础组件策略;建议 Tailwind token + 无样式/轻量基础组件ECharts 保留 |
| P2 | 导出文件存储位置与保留期 | 确认 OSS 或 ECS 本地盘;推荐 OSS/兼容对象存储,任务和文件设过期清理策略 |
## 12. 本阶段完成定义
本阶段仅交付本分析、路线图、API 缺口和数据模型四份文档。V2 代码、数据库迁移、告警/导出 worker、ECS 发布均须在确认 P0 项后按路线图进入下一阶段。

View File

@@ -0,0 +1,153 @@
# 车辆数据中台一期 API 能力与缺口
## 1. 判定规则
- “已有”表示当前平台 BFF 已有路由和生产 Store 实现。
- “部分”表示可作为底层数据来源,但 DTO、筛选、性能或业务语义不足。
- “缺失”表示当前没有可支撑验收的持久化/API 链路。
- 新接口建议统一放 `/api/v2`;旧 `/api/alert-events*` 是质量接口别名,不得作为新告警契约继续扩展。
## 2. 现有接口清单
| 方法与路径 | 状态 | 数据源/说明 |
| --- | --- | --- |
| `GET /api/dashboard/summary` | 已有 | MySQL/运维探针聚合;口径偏现有服务健康 |
| `GET /api/vehicles` | 已有 | 车辆身份与实时快照集合,支持关键词/协议等基础过滤 |
| `GET /api/vehicles/resolve` | 已有 | 通过 VIN、车牌、手机号解析车辆 |
| `GET /api/vehicles/coverage``/summary` | 已有 | 多协议来源覆盖、绑定和在线来源数 |
| `GET /api/vehicle-service``/summary``/overview` | 已有 | 单车/车队证据聚合 |
| `POST /api/vehicle-service/overviews` | 已有 | 批量关键词概览 |
| `GET /api/realtime/vehicles` | 已有 | VIN 级聚合实时状态 |
| `GET /api/realtime/locations` | 已有 | MySQL 每协议最新位置 |
| `GET /api/history/locations` | 已有 | TDengine 位置历史;基础分页和时间过滤 |
| `GET /api/history/raw-frames` | 已有 | TDengine RAW默认 100 条,可选解析字段 |
| `POST /api/history/raw-frames/query` | 已有 | 复杂 RAW 查询 |
| `GET /api/mileage/summary``/daily` | 已有 | MySQL 日里程 |
| `GET /api/statistics/online-summary``/online-vehicles` | 已有 | 固定口径在线统计 |
| `GET /api/quality/summary``/issues``/notification-plan` | 已有 | 推导质量信号和静态方案,不持久化 |
| `GET /api/alert-events*` | 语义错误 | 上述质量接口别名,不是告警规则/事件/通知 |
| `GET /api/map/reverse-geocode` | 已有 | 服务端高德逆地理编码 |
| `GET /api/ops/health``/source-readiness` | 已有 | 链路、release、lag、连接、可写性和来源准备度 |
## 3. 按一期功能的缺口矩阵
| 功能 | 当前可复用 | 仍缺少 | 优先级 |
| --- | --- | --- | --- |
| 全局统计 | V2 monitor summary 已实现接入/在线/离线/行驶/静止/未知/今日上报,以及活跃业务告警 VIN 与同筛选车辆集合交集后的权威告警车辆数;页面仅在服务端明确可用时展示数值 | 更复杂的车型/企业/接入商组合字典可在车辆规模增长时从接入管理筛选下沉复用 | P0一期完成 |
| 地图车辆 | V2 monitor map 已实现 bounds/zoom、关键词/协议/状态筛选、低缩放聚合、高缩放轻量点、2,000 点自适应降级;前端独立缓存、防抖请求和 MassMarks列表上限 200 | 差量 cursor/SSE 作为车辆规模或刷新频率继续增长后的扩展项 | P0一期完成 |
| 车辆信息卡 | 选中车辆按需并发组合 realtime、vehicle service、daily mileage、服务端逆地理编码和 `status=active` 告警;展示当日里程、地址、全部真实来源、接入厂家、当前告警及三处深链 | 权威档案未提供接入厂家时明确显示“待补充”,不从协议猜测 | P0一期完成 |
| 单车档案 | vehicle service 已合并 `vehicle_profile`;管理员可单车维护或用 CSV/API 批量同步,外部同步具备来源/版本幂等、人工及异源保护、显式接管、dry-run、身份校验和版本审计 | 仍需为具体车厂/GPS 厂商配置定时连接器;首次接入/运行时长的自动投影仍待网关权威源 | P0部分完成 |
| 最新遥测 | `GET /api/v2/vehicles/:vin/telemetry/latest` 已服务端选择每个统一指标/源字段最新值,动态分组,保留厂家扩展,并返回中文名、单位、类型、设备/接收时间、帧、协议、端点、freshness/delay 与值级质量原因;页面独立缓存刷新且不再猜测 RAW 元数据 | 更多动态 RAW 指标进入统一目录仍需受控发现和管理端审计,不影响一期最新值取证 | P0一期完成 |
| 轨迹 | V2 playback 已有覆盖边界、最多 7 天校验、GPS 质量过滤、推断停车、活动分段、关键点保留抽稀、方向/SOC/告警同步字段和统计元数据 | ignition 可信怠速、持久 queryId/游标分段仍缺少 | P0部分完成 |
| 回放地址 | 暂停点按坐标缓存逆地理编码,播放时暂停解析,避免逐点请求 | 服务端批量地址目录仅在未来需要轨迹点地址表时再建设 | P1一期完成 |
| 历史列表/曲线 | V2 动态指标目录、多车多指标列表、服务端分页、按单位拆轴的时间序列聚合、空窗/覆盖率/异常证据和 60600 点受控曲线已上线 | 更多可聚合遥测指标需随统一指标目录逐项开放RAW 保持离散证据 | P0一期完成 |
| 导出 | 单 ECS 持久异步 CSV 队列已上线:进度/状态/完成时间/下载、单并发、31 天/5 车/32 指标/100 万行/30 分钟配额、游标流式写入、重启恢复和原子文件发布 | XLSX、取消和多实例对象存储属于扩容项不阻塞一期 CSV 验收 | P0一期完成 |
| 告警规则 | V2 MySQL 规则、版本审计、CRUD/启停/校验、统一指标目录动态校验、数值区间内外、布尔状态变化、协议/VIN/OEM/车型/企业范围、持续/恢复/重复抑制已实现 | 软删除、复杂范围组合预估和批量规则导入仍待补齐 | P0部分完成 |
| 告警事件 | V2 持久事件、事件时间持续候选、活跃指纹去重、恢复/处置时间线、通知、Kafka fields active consumer、规则副作用与 checkpoint 同事务、数据库权威重放抑制、迟到门禁、动态/墙钟所有权拆分、真实规则灰度、锁竞争及连续运行门禁均已上线 | 跨窗口历史重算属于后续分析能力;短信/邮件/企微供应商明确不在一期范围 | P0一期完成 |
| 站内通知 | V2 列表、未读数、批量已读和页面轮询已实现;外部通道明确为 reserved | SSE 可在通知规模增长时替换轮询;外部供应商不在一期范围 | P1一期完成 |
| 接入状态 | V2 summary/vehicles/thresholds 已实现动态阈值、事件/接收延迟、状态区分、协议/车辆厂家/车型/接入厂家/首次接入/最新上报筛选、同口径统计和版本审计;网关同一快照 upsert 已维护首次/前次/最新接收、连续间隔、样本数及回填边界;缺 VIN 的 JT808 终端已进入脱敏处置队列 | 仍需运维核对来源并维护权威 phone→VIN 绑定;历史首次接入只能由未来权威台账补齐,不能把上线回填当历史真值 | P0部分完成 |
| 权限审计 | ECS 强制 Bearer 鉴权viewer/operator/admin 累积权限、前端会话门禁、后端逐操作校验、规则/阈值/告警动作版本审计及越权拒绝均已验证 | 企业 SSO 与复杂多租户不在一期范围 | P0一期完成 |
## 4. 推荐 V2 API
### 4.1 元数据与车辆
| 方法与路径 | 用途 | 关键参数/响应 |
| --- | --- | --- |
| `GET /api/v2/meta/vehicle-filters` | 全局筛选字典 | OEM、车型、公司、协议、接入厂家、状态带版本/ETag |
| `GET /api/v2/metrics` | 动态指标目录 | `protocol, category, valueType, searchable, chartable, alertable` |
| `GET /api/v2/vehicles/search` | 远程车辆选择 | `q, cursor, limit`;返回 VIN、车牌、车辆编号 |
| `GET /api/v2/vehicles/:vin` | 单车数字档案 | 档案、实时摘要、接入摘要、当前告警摘要 |
| `GET /api/v2/vehicles/:vin/telemetry/latest` | 动态最新遥测 | 按 category 分组,值含时间、单位、质量、来源 |
### 4.2 全局监控
| 方法与路径 | 用途 | 关键参数/响应 |
| --- | --- | --- |
| `POST /api/v2/monitor/summary` | 与筛选完全一致的状态统计 | 组合过滤对象、`asOf`、在线阈值版本 |
| `POST /api/v2/monitor/map` | 视口聚合/车辆点 | `bounds, zoom, filters, cursor`;返回 `mode=clusters|points`、聚合数或轻量点 |
| `GET /api/v2/monitor/vehicles/:vin/card` | 点位详情卡 | 核心实时、当日里程、来源、告警、地址缓存 |
| `GET /api/v2/monitor/changes` | 可选差量刷新 | `since` 或 SSE返回变更车辆和统计版本 |
`monitor/map` 必须有最大点数和聚合降级;低缩放级别绝不返回全部明细。轻量点不携带完整 telemetry JSON。
### 4.3 轨迹
| 方法与路径 | 用途 | 关键参数/响应 |
| --- | --- | --- |
| `POST /api/v2/tracks/query` | 轨迹分段查询 | VIN、时间、协议、filter、targetPoints、cursor返回点、起终点、过滤/抽稀计数和算法版本 |
| `GET /api/v2/tracks/:queryId/segments` | 大查询后续分段 | segment/cursor支持取消/过期 |
| `GET /api/v2/tracks/:queryId/stops` | 停车点 | 起止时间、时长、坐标、地址缓存状态 |
轨迹点至少包含 `eventTime, receivedAt, lng, lat, speedKmh, directionDeg, socPercent, totalMileageKm, alarmFlag, statusFlag, state`。抽稀必须保留起终点、停车边界和告警点,并返回原始/有效/返回点数。
### 4.4 历史分析
| 方法与路径 | 用途 | 关键参数/响应 |
| --- | --- | --- |
| `POST /api/v2/timeseries/query` | 多车多指标曲线 | VINs、metricKeys、时间、granularity、aggregation、targetPoints、fillPolicy |
| `POST /api/v2/timeseries/rows` | 动态列列表 | 同上 + cursor、limit、sort、filters |
| `POST /api/v2/timeseries/estimate` | 查询/导出预估 | 估算点数、行数、扫描范围并给出建议粒度 |
服务端只允许指标目录白名单,不能把客户端 metricKey 直接拼入 SQL。响应带指标显示名、单位、类型、来源、采样方式和空值策略。
### 4.5 导出任务
| 方法与路径 | 用途 |
| --- | --- |
| `POST /api/v2/exports` | 创建任务携带查询快照、CSV/XLSX、时区使用幂等键 |
| `GET /api/v2/exports` | 当前用户任务分页 |
| `GET /api/v2/exports/:id` | 状态、进度、行数、完成/过期时间、错误 |
| `POST /api/v2/exports/:id/cancel` | 取消排队/运行任务 |
| `GET /api/v2/exports/:id/download` | 权限检查后短期下载或签名 URL |
状态:`queued, running, succeeded, failed, canceled, expired`。同步 API 不生成大文件。
### 4.6 告警与通知
| 方法与路径 | 用途 |
| --- | --- |
| `GET/POST /api/v2/alert-rules` | 规则列表/创建 |
| `GET/PUT/DELETE /api/v2/alert-rules/:id` | 详情/修改/软删除 |
| `POST /api/v2/alert-rules/:id/enable|disable` | 明确启停操作 |
| `POST /api/v2/alert-rules/validate` | 校验指标类型、操作符、阈值和范围 |
| `GET /api/v2/alert-events``/summary` | 事件列表和统计 |
| `GET /api/v2/alert-events/:id` | 事件详情与完整处理时间线 |
| `POST /api/v2/alert-events/:id/acknowledge` | 确认/进入处理中 |
| `POST /api/v2/alert-events/:id/close` | 关闭,要求备注 |
| `POST /api/v2/alert-events/:id/ignore` | 忽略,要求原因 |
| `GET /api/v2/notifications``/unread-count` | 站内通知/未读数 |
| `POST /api/v2/notifications/read` | 批量已读 |
所有变更接口记录操作者、时间、前后状态、备注和 traceId规则修改采用 `version` 乐观锁。
实现记录2026-07-14实际落地路由统一在 `/api/v2/alerts/*`,列表/汇总使用 POST 同构筛选体,处置统一为 `POST /events/:id/actions`,规则启停为版本化 `PUT /rules/:id/enabled`。迁移为 `002_alert_center.sql``003_alert_rule_advanced.sql``004_alert_repeat_index.sql`,独立 `alert-evaluator` systemd 服务以当前 MySQL realtime location 作为一期评估输入。旧 `/api/alert-events*` 继续只是质量投影兼容接口新页面不依赖它们。Kafka 可重放消费、迟到窗口和 traceId 写入通用审计仍是生产退出缺口。
指标目录实现记录2026-07-14`GET /api/v2/metrics` 已返回统一 key、中文名、单位、类别、值类型、协议及源字段映射和三类能力标记生产权威数据已落入 `vehicle_metric_definition/vehicle_metric_protocol_mapping``005_metric_catalog.sql` 仅补齐初始缺失项,不覆盖后续配置。告警编辑器只消费其中 `alertable=true` 且类型匹配的指标,规则写接口再次校验白名单、能力、值类型和 evaluator 支持。`GET /api/v2/vehicles/:vin/telemetry/latest` 现消费同一目录并为未入目录的厂家字段保留源字段、动态分类和质量证据;历史页仍保留 `/api/v2/history/metrics` 兼容目录。后续仅剩动态 RAW 指标受控发现与管理端版本审计。
### 4.7 接入管理
| 方法与路径 | 用途 |
| --- | --- |
| `POST /api/v2/access/summary` | 按同一筛选口径统计协议、厂家、在线率、今日上报、长离线、从未上报、延迟异常 |
| `POST /api/v2/access/vehicles` | 接入状态分页;支持阈值、时间、厂家/车型/协议过滤 |
| `POST /api/v2/access/unresolved-identities` | 无权威 VIN 的真实终端处置队列;只返回哈希 ID、脱敏标识和登记/上报证据 |
| `GET /api/v2/access/thresholds` | 获取全局默认和协议覆盖阈值 |
| `PUT /api/v2/access/thresholds` | 管理员更新阈值并记录版本/审计 |
接入行应返回 `firstSeenAt, latestEventAt, latestReceivedAt, reportIntervalSec, dataDelaySec, onlineState, thresholdSec, latestMessageType, latestError`,并区分 `online/offline/never_reported/unknown`
实现记录2026-07-14状态/阈值接口、同口径筛选、动态状态、阈值乐观锁和 MySQL 审计已落地gateway access projection 已维护 first/previous/latest received、latest error、样本数和连续间隔。`platform-v2-20260714080619` 新增身份待绑定接口和紧凑处置队列:生产当前识别 1 个真实 JT808 缺 VIN 终端响应和复制证据均仅含脱敏标识20 次顺序查询 p50 2.02ms、p95 2.44ms、最大 4.66ms。该队列用于推动权威绑定,不会把手机号猜成 VIN也不会让未绑定身份参与车辆告警。
## 5. 横切契约
- 时间统一使用 RFC 3339/UTC 传输UI 按 Asia/Shanghai 展示;同时保留事件时间和接收时间。
- 列表固定 `items + pageInfo`;大时序优先 cursor管理列表可 offset总数默认可选避免高成本 COUNT。
- 所有筛选/排序字段白名单化请求体大小、VIN 数、指标数、时间范围、返回点数均设上限。
- API 返回 `traceId`、数据 `asOf`、口径/算法版本;可取消请求使用 `AbortSignal`
- 查询 GET 可 ETag/短缓存;任务/规则 POST 使用幂等键;状态修改使用版本号。
- 前端只访问平台 BFF。Redis/TDengine/MySQL 连接与高德 REST key 均不暴露。
## 6. 兼容与下线
旧接口在 V2 联调期保留;新页面不依赖旧 `alert-events` 别名。完成迁移后先增加 deprecation 响应头和调用监控,再按版本窗口下线。现有 `/api/history/locations` 和 RAW 查询可继续作为运维证据接口,不强行替换为业务时序 DTO。

View File

@@ -0,0 +1,224 @@
# 车辆数据中台一期数据模型
## 1. 数据职责原则
| 存储 | 负责 | 不负责 |
| --- | --- | --- |
| Redis DB 50 | 最新状态、在线 TTL、字段级当前值、热查询缓存 | 历史事实、告警事件、规则、导出任务 |
| TDengine `lingniu_vehicle_ts` | 高频 RAW 证据、位置/稳定时序历史 | 车辆档案、权限、工作流状态 |
| MySQL `lingniu_vehicle_data` | 车辆主数据、身份映射、接入配置、当前业务投影、统计、规则、事件、通知、导出任务 | 每帧完整历史和无限增长的遥测明细 |
统一业务主键为 VIN。车牌、JT808 手机号、设备号、平台标识只是可变标识,通过映射解析到 VIN无法解析 VIN 的数据仍留 RAW/接入排障,不进入正式车辆统计。
## 2. 当前 TDengine 模型
### 2.1 `raw_frames` stable
列:`ts, frame_id, event_id, message_id, event_time, received_at, raw_size_bytes, raw_hex, raw_text, parsed_json, parse_status, parse_error, source_endpoint`
Tags`protocol, vehicle_key, vin, phone, device_id`
用途:证明收到什么、何时收到、解析成什么。`parsed_json` 物理列保存 Gateway 已一次扁平化的 `parsed_fields`,是动态协议字段的历史证据。新字段默认先进入这里,不为每个厂家字段立即新增物理列。
### 2.2 `raw_frame_payload_chunks` stable
列:事件/帧、接收时间、payload 类型、分片序号/总数、分片文本Tags 与 RAW 身份类似。
用途:保存超过主表列长度的 raw/parsed payload不参与常规业务查询。
### 2.3 `vehicle_locations` stable
列:`ts, event_id, received_at, longitude, latitude, altitude_m, speed_kmh, direction_deg, alarm_flag, status_flag, total_mileage_km`
Tags`protocol, vin`
用途VIN 非空车辆的高频位置、速度、方向、告警/状态位、总里程历史。当前平台 DTO 尚未完整透出方向、告警和状态位,也没有 SOC 列;轨迹一期需先补 DTO/查询SOC 可从 RAW 指标取值或按验证后的需求物化。
数据库当前配置 `KEEP 7300 DURATION 10 BUFFER 256`。保留期和磁盘容量应纳入运维监控,不在业务页面查询时临时改变。
## 3. 当前 Redis 模型
| Key | 值/用途 | 约束 |
| --- | --- | --- |
| `vehicle:latest:{vin}` | 跨协议最新核心字段 | 仅 VIN 非空;轻量快照 |
| `vehicle:latest:{vin}:{protocol}` | 单协议最新核心字段 | 可过期/重建 |
| `vehicle:realtime-raw:{protocol}:{vin}` | 单协议最新完整 parsed 状态 | Redis 中完整协议字段唯一副本 |
| `vehicle:rt-kv:{protocol}:{vin}:values` | 扁平字段当前值 | 与类型/时间/meta 配套 |
| `...:types` | 字段值类型 | 支持动态展示和规则类型校验的实时侧依据 |
| `...:times` | 每字段最新归一事件时间 | 防止乱序旧帧覆盖 |
| `...:meta` | 最新事件/接收时间、映射版本 | 用于新鲜度和追踪 |
| `vehicle:online:{protocol}:{vin}` | TTL 在线状态 | 在线事实的快速判断,不应永久保存 |
| `vehicle:online-state:{protocol}:{vin}` | 可分页的 Hash 副本 | 用于在线列表 |
| `vehicle:protocols:{vin}` | 最近出现协议集合 | 车辆来源发现 |
| `vehicle:last_seen` | 最近活跃车辆 ZSET | 分页/排序候选 |
生产只允许 `nats-fast-writer` 写 Redis 当前态Kafka 慢链路的 realtime writer 不应覆盖快路径。在线产品状态应由“最新时间 + 配置阈值”计算并返回阈值/依据,不能只信一个固定 Boolean。
## 4. 当前 MySQL 模型
### 4.1 身份与档案
| 表 | 主键/关键字段 | 用途 |
| --- | --- | --- |
| `vehicle` | `vin``plate, oem, enabled, updated_at` | 轻量车辆主实体;当前字段不足一期完整档案 |
| `vehicle_identifier` | `(protocol, source_code, identifier_type, identifier_value)`VIN/车牌/OEM | 多协议、多来源、多标识解析到 VIN |
| `vehicle_identity_binding` | VIN车牌、phone、OEM | 外部维护的兼容映射事实,运行期只读 |
| `jt808_registration` | `phone`device、plate、VIN、厂家、鉴权、来源、首次/最近注册鉴权/出现时间 | JT808 注册鉴权与绑定证据 |
### 4.2 当前态与来源
| 表 | 主键/关键字段 | 用途 |
| --- | --- | --- |
| `vehicle_realtime_snapshot` | `(protocol, vin)`plate、platform、peer、扁平 `parsed_json`、event/received time | 每协议每 VIN 最新合并遥测快照 |
| `vehicle_realtime_location` | `(protocol, vin)`经纬度、速度、总里程、SOC、海拔、方向、alarm/status、event/received time | 每协议每 VIN 最新位置业务缓存 |
| `vehicle_data_source` | 自增 ID唯一 `(protocol, source_ip)`source code/kind、platform、priority、enabled、first/latest seen | 数据来源配置和可信源选择 |
### 4.3 日里程
| 表 | 主键/关键字段 | 用途 |
| --- | --- | --- |
| `vehicle_daily_mileage_source` | `(vin, stat_date, protocol, source_key)`;首末里程、样本、质量、是否选中 | 多来源日里程事实和质量解释 |
| `vehicle_daily_mileage` | `(vin, stat_date, protocol)`source ID、日里程、最新总里程 | 对外查询结果投影 |
当前明确不存在/不应恢复的旧模型包括泛化 `vehicle_daily_metric`、TDengine `vehicle_mileage_points``raw_frames.fields_json` 重复列。
## 5. 一期统一领域模型
### 5.1 车辆标识
```text
VehicleId = VIN内部规范化大写、去空格
VehicleIdentifier = protocol + sourceCode + type + value -> VIN
Plate = 可变展示标识,不作为跨表唯一主键
Source = protocol + sourceCode/sourceId + endpoint
```
所有 API 都以 VIN 作为详情路径主键搜索接口可接受车牌、VIN、车辆编号/手机号并返回解析结果。多协议数据保留 `protocol``sourceId` 做归因,不能在合并时丢失来源。
### 5.2 时间模型
- `eventTime`:车端/协议事件时间,是轨迹、曲线和告警判断的主要时间。
- `receivedAt`:平台接收时间,用于延迟、迟到数据和链路健康。
- `storedAt/updatedAt`:投影或业务记录更新时间,不冒充车辆上报时间。
- `dataDelaySec = receivedAt - eventTime``silenceSec = now - max(receivedAt/eventTime 的已确认口径)`
### 5.3 在线状态
```text
never_reported: 无任何 latestReceivedAt
online: silenceSec <= thresholdSec
offline: silenceSec > thresholdSec
unknown: 时间非法、未来时间过大或来源状态无法判断
```
行驶/静止是独立 motion 状态,建议以速度阈值 + 最近更新时间判断;告警也是独立状态。不要把 `online/running/alarm` 压成一个互斥枚举UI 可基于多维状态决定主色和徽标。
### 5.4 指标值
```text
MetricValue {
vehicleId, metricKey, value, valueType, unit,
eventTime, receivedAt, protocol, sourceId,
quality, mappingVersion
}
```
`metricKey` 是协议无关的规范键;协议原字段通过映射表关联。厂家扩展字段允许命名空间,但必须有显示名、类型、分类和单位后才可进入图表/告警。
## 6. 建议新增 MySQL 模型
### 6.1 档案与组织
建议优先用关联表而非不断扩宽 `vehicle`
- `vehicle_profile(vin PK, vehicle_no, model_id, vehicle_type, company_id, operation_status, access_vendor_id, first_access_at, ...)`
- `vehicle_model(id, oem_id, code, name, vehicle_type, ...)`
- `organization(id, parent_id, type, code, name, enabled, ...)`
- `access_vendor(id, code, name, enabled, ...)`
外部主数据通过受控同步写入 `source_system/source_version/synced_at`。默认策略保护 `manual` 和其他外部来源;只有管理员显式选择 takeover 才能改变所有权。同一来源版本相同载荷幂等,不同载荷冲突,避免上游版本不可变性被破坏;每次创建/更新都进入 `vehicle_profile_audit`
### 6.2 指标目录
- `metric_definition(metric_key PK, display_name, category, value_type, unit, precision, chartable, alertable, enabled, sort_order, version)`
- `metric_protocol_mapping(metric_key, protocol, source_field, transform, mapping_version, enabled)`
`transform` 不能存任意可执行代码;使用有限转换类型或由版本化 Go 映射实现。目录服务可缓存,变更有版本和审计。
### 6.3 告警域
- `alert_rule(id, name, description, metric_key, operator, threshold_json, severity, duration_sec, recovery_json, repeat_interval_sec, enabled, version, created_by, updated_by, timestamps)`
- `alert_rule_scope(rule_id, scope_type, scope_value)`vehicle/model/oem/protocol/company一期可限制组合复杂度。
- `alert_event(id, rule_id, rule_version, vin, protocol, source_id, status, trigger_value_json, threshold_json, first_triggered_at, last_triggered_at, recovered_at, closed_at, location_json, dedupe_key, assignee, timestamps)`
- `alert_event_action(id, event_id, action, from_status, to_status, operator_id, note, created_at)`
- `notification(id, user_id, event_id, title, body, read_at, created_at)`
`alert_event.dedupe_key` 建唯一约束或等价幂等机制;阈值快照和 rule version 必须保存在事件上,避免规则修改后历史无法解释。
当前实现使用带业务前缀的实体表,完整 DDL 位于 `vehicle-data-platform/deploy/migrations/002_alert_center.sql`
- `vehicle_alert_rule` + `vehicle_alert_rule_audit`:当前规则与每版本不可变快照。
- `vehicle_alert_candidate``(rule_id, vin, protocol)` 持续命中计时,避免单帧达到阈值就误开事件。
- `vehicle_alert_stream_checkpoint``(consumer_group, topic, partition_id)` 保存数据库权威 `next_offset`、Kafka high watermark、处理/非法/迟到/回放计数、最近事件证据以及最近非法代码/时间。流 worker 先提交该事务,再提交 Kafka group offset崩溃后重复抓取的 offset 会被数据库 checkpoint 跳过。累计非法数用于审计,健康状态只对五分钟内仍在发生的非法消息告警。
- `vehicle_alert_event`:保存规则名/版本、触发阈值快照、证据时间、位置和乐观锁版本;`fingerprint + active status` 用于评估器幂等检查。
- `vehicle_alert_event_action`:触发、恢复和人工处置的不可变时间线。
- `vehicle_alert_notification`:站内通知真实已读状态;外部通道只有 `reserved`,没有配置供应商时绝不标记 `sent`
- `vehicle_alert_rule_state`:为 Boolean `changed` 规则保存每个 rule/VIN/protocol 的最近观测值;`003` 同时增加区间上限和 OEM 范围,`004` 为 fingerprint 重复间隔查询增加索引。
一期 evaluator 读取 `vehicle_realtime_location` 的当前快照并用 candidate 表证明持续时长。除 `freshness_sec` 明确按平台墙钟增长外,候选只允许由不同 `source_event_id` 且事件时间不回退的新观测推进;重复快照不能制造持续时长,迟到观测不能回退 candidate 或 Boolean 状态。规则行会在评估事务内加锁,停用规则、清理 candidate/Boolean 状态和写版本审计在同一事务提交,从而避免 evaluator 与管理员停用并发后留下幽灵候选。它仍不是 Kafka event-time 可重放评估器,因此跨当前快照的迟到数据重算、历史窗口规则和消费 offset 仍是阶段 8 的后续能力。
`alert-stream-evaluator` 已以 active 模式消费三类 canonical fields topic。它复用 Kafka consumer group 的分区顺序,校验 `event_kind/field_mapping/protocol/VIN/source_event_id/字段命名空间`,用接收时间识别超过可配置窗口的迟到观测。动态规则锁、当前批次范围内的 candidate/Boolean/活跃事件/重复窗口、事件/动作/通知副作用和 checkpoint 在同一个 MySQL 事务内完成,成功后才 commit Kafka数据库 ahead 时的重复 offset 由 checkpoint 跳过。超过迟到窗口的观测只允许驱动 `data_delay_sec`,不能开关速度/SOC/告警位事件。快照 evaluator 在 active 模式只负责按墙钟增长的 `freshness_sec`,不再读取动态规则。
### 6.4 导出域
- `export_job(id, user_id, type, format, query_json, status, progress, estimated_rows, exported_rows, file_uri, file_size, checksum, error_code, error_message, started_at, completed_at, expires_at, created_at)`
索引至少覆盖 `(user_id, created_at)``(status, created_at)` 和过期清理。`query_json` 保存规范化查询快照,不保存数据库密码或签名下载 URL。
### 6.5 在线阈值与审计
- `online_threshold(scope_type, scope_value, threshold_sec, version, updated_by, updated_at)`,一期建议 scope 仅 global/protocol。
- `audit_log(id, actor, action, resource_type, resource_id, before_json, after_json, trace_id, created_at)`
当前 V2 接入管理已先落地等价的窄表:
- `vehicle_access_threshold_config(id=1, version, default_threshold_sec, delay_threshold_sec, long_offline_sec, protocol_overrides_json, updated_by, updated_at)`
- `vehicle_access_threshold_audit(id, version, actor, summary, config_json, changed_at)`
表仅向前新增,不改变 gateway 的 `vehicle_realtime_snapshot(protocol, vin)` 主键与写入语义。后续统一审计域上线时可迁移到通用 `audit_log`,但必须保留版本历史。
## 7. 查询与物化策略
1. 实时页面优先 RedisMySQL realtime 作为业务查询/降级投影;返回值统一由 BFF 归一化。
2. 轨迹只查 TDengine `vehicle_locations`,必要的 SOC/告警补充应避免对每个点逐条回查 RAW先验证是否需要将 SOC 提升为位置列或建立专用指标 stable。
3. 通用历史指标首版可从 `raw_frames.parsed_json` 按白名单提取,但必须限制车辆数、指标数、时间范围并做基准;高频稳定指标按实际查询热度物化为专用时序模型。
4. 档案和字典在 BFF/Redis 做版本化短缓存,避免每次实时请求重复查不变数据。
5. 地图聚合可从 MySQL realtime/Redis 快照构建短周期服务端缓存;不要把地理历史查询压到实时表,也不要把全量点放浏览器聚合。
6. 告警 worker 使用 Kafka 可重放流和 eventTimeMySQL 保存工作流事实Redis 可保存短期去重/规则缓存但不是唯一事实。
## 8. 数据质量与治理要求
- 数值同时记录单位和精度;未知单位不得参与跨协议比较或规则判断。
- 原始值、规范值和转换版本可追踪;异常值标记不直接篡改 RAW。
- 未来时间、经纬度越界、速度突变、里程回退、重复事件均定义质量码。
- 轨迹过滤返回算法版本、原始/过滤/抽稀数量,可选择查看未过滤证据。
- 协议新增字段先进入扁平 RAW + Redis KV经目录登记后才进入 UI形成稳定高频查询后再物化。
- 所有新增表、字段、索引、保留期和清理任务随 migration 与数据字典发布。
## 9. 容量和保留
- 约 1,000 到 10,000 车辆、530 秒上报意味着时序写入持续增长TDengine 查询必须带 VIN/时间,禁止开放无界扫描。
- RAW 保留期、位置保留期、导出文件保留期和告警/审计保留期分别配置,不能共用一个默认值。
- MySQL 告警事件和审计按时间索引并规划归档;导出任务定时过期,文件删除和 DB 状态更新需幂等。
- 单 ECS 上告警和导出 worker 设置独立并发、内存、CPU 和 systemd 限制,避免影响接入写链路。
## 10. 仍需确认的数据问题
1. `vehicle_identity_binding` 的实际 DDL、权威维护方和未来是否继续兼容。
2. 车型、车辆类型、运营公司、车辆编号、运营状态、接入厂家的权威来源与更新频率。
3. 各协议 SOC、方向、启动、告警等级、电机/燃料电池等规范字段映射和单位。
4. “累计运行时长”是已有车端字段、由状态积分计算,还是一期可不提供。
5. 在线时间基准使用 eventTime 还是 receivedAt以及不同协议的默认阈值。
6. 轨迹停车阈值、漂移判定、迟到点容忍和跨协议主轨迹选择规则。
7. 导出存储与保留期、单用户并发/最大行数。
8. 告警规则适用范围组合、迟到窗口、重复/恢复语义和事件归属人规则。

View File

@@ -0,0 +1,162 @@
# 车辆数据中台一期实施路线图
> 本路线图从分析完成后的下一阶段开始。任何阶段都以“契约、实现、测试、运行检查、文档”全部完成为退出条件,不并行铺开所有页面。
## 1. 阶段总览
| 阶段 | 目标 | 主要交付 | 退出条件 |
| --- | --- | --- | --- |
| 0 分析与决策 | 冻结一期边界和关键决策 | 本次四份文档、P0 决策记录 | 档案来源、认证、指标元数据、在线阈值、文件存储有负责人和结论 |
| 1 数据与 API 基座 | 建立统一模型和可演进契约 | MySQL 向前迁移、指标目录、V2 OpenAPI、鉴权/审计、统一错误/分页 | 契约测试通过;旧链路不受影响;迁移可回滚 |
| 2 V2 前端基础 | 建立独立、无 mock 的前端壳 | 路由/菜单、权限、token、API client、AsyncState、地图/图表/表格基础组件 | 构建和组件测试通过1440×900 基础布局可用;无 mock fallback |
| 3 全局监控 | 先打通实时主路径 | 地图聚合、组合筛选、统计、车辆信息卡、详情跳转 | 1k/10k 数据基准达标;视口移动不卡顿;状态口径一致 |
| 4 单车数字档案 | 形成统一车辆入口 | 档案、实时状态、地图、动态遥测分组、来源证据 | 车牌/VIN/车辆编号均可解析;缺失字段不渲染空大表 |
| 5 轨迹回放 | 完成历史位置产品能力 | 分段/抽稀查询、过滤、停车点、播放控制、列表联动 | 长时段不会全量阻塞;抽稀/过滤可解释;播放状态正确 |
| 6 历史分析与导出 | 建立通用指标分析 | 指标目录、多车多指标列表/曲线、异步 CSV/XLSX | 百万点查询经服务端聚合;导出有任务进度和受控下载 |
| 7 接入管理 | 产品化接入质量 | 动态在线阈值、延迟、最近消息/错误、协议/厂家统计 | 从未上报与离线可区分;事件/接收时间延迟口径明确 |
| 8 告警中心 | 建立可持久业务告警 | 规则 CRUD、评估 worker、事件、处理记录、站内通知 | 数值/Boolean 规则可验证;去重、恢复、确认、关闭链路完整 |
| 9 联调与生产发布 | 达到一期验收并部署当前 ECS | 性能/异常/安全测试、运维文档、release、监控和回滚 | 验收清单通过ECS 健康、无 lag、旧 release 可回滚 |
进度注记2026-07-14阶段 7 的 V2 接入汇总、车辆状态、动态阈值、协议覆盖、版本审计及桌面/移动端页面已实现。网关 realtime snapshot 同一条原子 upsert 现维护独立于设备事件时间的首次/前次/最新接收时间、连续间隔、样本数和 event ID 去重;平台读取该投影并合并车型/企业,明确区分 `live_writer` 真首次观测和 `snapshot_backfill` 上线基线。`platform-v2-20260714080619` 又将缺少权威 VIN 的真实 JT808 终端纳入“身份待绑定”队列,只暴露哈希 ID、脱敏终端号和登记/上报证据;当前生产 1 条,页面与复制内容均未泄漏原始手机号,查询 p95 2.44ms。实际 phone→VIN 绑定仍必须由运维核验权威来源,不能猜测。
阶段 8 进度注记2026-07-14durable 规则/审计、持续候选、事件/动作、站内通知、独立 evaluator、V2 API 与告警中心三个工作区已上线;确认、规则创建、通知已读和权限通过真实接口联调。`platform-v2-20260714070507` 将持续时间改为由不同 source event 的单调事件时间推进,重复快照和迟到观测不再制造持续异常,`freshness_sec` 作为明确的墙钟规则单独处理;后续 Kafka fields active evaluator 又补齐分区 checkpoint、迟到窗口和数据库权威重放抑制。真实规则灰度、规则编辑锁竞争及 31 分钟连续运行门禁均通过,阶段 8 一期退出项已关闭。
Kafka 告警流进度注记2026-07-14`platform-v2-20260714080619` 当前生产仍运行 active event-time evaluator。动态规则锁、批次范围 candidate/Boolean/活跃事件/重复窗口、事件/动作/通知和 `(group,topic,partition)` checkpoint 在同一个 MySQL 事务内完成,随后才 commit Kafka失败事务不会越过 offset。快照 evaluator 在 active 模式只保留 `freshness_sec`,空动态规则扫描由约 13ms 降至 0.51ms。迁移 `010` 已用真实 RAW 样本修正国标告警位和宇通速度/SOC 映射。受限 JT808 `speed>=0`、持续 20 秒、重复 1 小时灰度只打开 1 条事件;关闭后又有 2 批 candidate 推进但未重开随后规则停用并清理状态。914 个带活跃规则的生产批次 p50 7.80ms、p95 12.07ms、最大 18.68ms,候选推进 11、重复观测 9、迟到观测 1、批次失败 0。迁移 `011` 进一步持久化最近非法原因/时间;当前已定位为单个未绑定 VIN 的 JT808 终端,保留 `missing_vin_jt808` 运维告警并在接入页生成脱敏处置队列。当前 9 分区 lag 0跨窗口历史重算和代表性峰值长稳仍待阶段 9 门禁。
规则编辑锁竞争门禁2026-07-14生产创建了不可能命中业务车辆的启用规则在 active stream evaluator 持续消费时完成 60 次顺序版本化编辑HTTP 200 为 60/60冲突和 5xx 均为 0编辑延迟 p50 3.991ms、p95 7.220ms、p99 8.584ms、最大 10.875ms。并发窗口 41 个活跃规则批次处理 303 条消息,失败、重放、误开和恢复均为 0规则随后停用且事件总数为 0。锁竞争退出项已关闭仍保留代表性峰值长稳与 RDS 峰值观测。
阶段 5 进度注记2026-07-14生产真实轨迹基线显示活跃车辆总点数约 9,50071,361旧接口仅加工最新 5,000 点但摘要未充分表达覆盖边界。`platform-v2-20260714060112` 已在当前 ECS 上线默认今天、最长 7 天校验、完整/最新切片证据、同协议无效坐标/重复/疑似漂移过滤、多源主来源选择、GPS 推断停车、行驶/停车/数据间隔分段、质量摘要以及保留事件和分段边界的抽稀映射。8 辆生产 canary 今天窗口为 5.734.22ms;正式切流抽样 2,254 个源点完整读取、1,200 个地图点、33 段和 5 次停车为 32.83ms。并行协议不会被拼成一条路线,停车也不冒充 ignition 状态;服务端持久 queryId/游标分段和长周期观测仍待后续数据合同。
轨迹与接入补齐注记2026-07-14`platform-v2-20260714083640` 已在当前 ECS 上线。`vehicle_locations` 向前增加 SOC轨迹返回 SOC 可用性、方向和告警并使用绝对时间戳;页面以 `Asia/Shanghai` 展示,同步卡片只为暂停点做一小时坐标缓存的地址解析。生产宇通 5 分钟 300 点全部有 SOCJT808 5 分钟 10 点全部有方向/告警;修复了 TDengine 时间窗二次偏移及亚秒采样被截断成零秒后误判 gap 的问题5,000 点页面分段从 3,203 降到 14仅保留 1 个真实 28 分钟间隔,无横向溢出或控制台错误。接入页新增车型、接入厂家、首次接入和最新上报范围;生产 `G7s + latestSeenFrom` 深链筛选当前页 50/50 匹配。小窗位置导出 120/120 行完成Kafka lag 0MySQL/TDengine 可写。
全局监控性能闭环2026-07-14`platform-v2-20260714090040` 已在当前 ECS 上线。此前页面虽有后端聚合接口,实际仍把 1,000 行普通车辆列表直接送入地图;现改为 200 行服务端筛选列表与独立地图查询,`moveend/zoomend` 300ms 防抖并由 React Query 去重缓存。低缩放返回聚合,高缩放按 bounds 返回轻量 MassMarks超过 2,000 个点时自适应扩大网格10,000 合成车辆不会退化成 10,000 个点或聚合标记。10k 纯聚合基准 30 次为 0.976ms/op当前 ECS 727 辆真实数据 30 次低缩放聚合为 29 组API p50 57.31ms、p95 63.44ms、最大 70.65ms,城市视口 168 点 p50 94.89ms、p95 108.56ms、最大 114.54ms。生产浏览器首屏为 15 个聚合/200 行,无横向溢出;行驶筛选所有加载行均匹配,唯一车牌查询自动从聚合切到 1 个点并聚焦到 2km 比例尺。
全局监控产品闭环2026-07-14`platform-v2-20260714093533` 将活跃告警 VIN 与当前关键词/协议/状态筛选后的车辆集合在服务端求交,`alertDataAvailable=true` 后页面展示权威告警车辆数,不再使用占位。选中车辆卡片按需并发读取档案/日里程、当前业务告警和地址,并展示真实来源、接入供应商、今日里程、解析位置及当前告警;无权威供应商时明确“待补充”。生产 727 车显示告警车辆 0抽样车辆今日里程 59.4 km、三协议来源、嘉兴解析地址和 0 条当前告警1280×720 无页面溢出。
单车最新遥测闭环2026-07-14`platform-v2-20260714092531` 已在当前 ECS 上线 `GET /api/v2/vehicles/:vin/telemetry/latest`。服务端复用权威指标目录统一 key/中文名/单位/类型/分类,同时保留真实源字段、厂家扩展、协议、端点、帧、设备/接收时间、freshness/delay 和 `good|stale|warning` 原因;前端不再从 RAW key 猜测标签或质量。身份只解析一次,三个协议表固定并行各取最多 5 帧且跳过分页计数,真实三源车辆扫描 10 帧返回 155 值/7 分组;生产 30 次 p50 174ms、p95 179ms、最大 179ms本地 100 帧×50 字段纯归并基准 0.389ms/op。页面使用独立 10 秒缓存、20 秒刷新一次遍历建立分组与来源索引1280 桌面和 390×844 移动端均无横向溢出,行内时间为紧凑时分秒,控制台无错误或警告。
连续运行门禁2026-07-14alert stream evaluator 自 08:31:50 起连续 active 运行超过 31 分钟且未重启;该窗口 16,226 批、216,118 条消息全部完成处理valid 216,072、历史未绑定身份导致 invalid 46、replay skipped 0、失败日志 0批次延迟 p50 6.511ms、p95 10.481ms、p99 12.806ms、最大 29.697ms。结合此前 929 个带活跃规则批次和 60 次并发版本编辑均无失败/误开,已覆盖正常流量长稳及规则锁竞争。窗口结束时 checkpoint 累计 processed 563,598、9 分区 lag 0、MySQL/TDengine 可写;最近非法身份时间已超过 20 分钟且脱敏处置队列仍保留。阶段 9 的代表性连续运行门禁关闭,云数据库 Performance Insights 仍作为容量扩容观察项,不再阻塞单 ECS 一期上线。
阶段 6 进度注记2026-07-14`platform-v2-20260714063725` 已在当前 ECS 上线历史时间序列聚合与百万行受控导出。位置历史按车辆与协议在 TDengine 使用 `PARTITION BY protocol INTERVAL(...)` 服务端聚合,速度使用桶均值、总里程使用桶末值并保留 min/max/样本数;最长 31 天、每序列目标 60600 点、空窗不 `FILL`,前端按单位拆成速度/总里程小多图并明确粒度、覆盖率、缺失桶、原始点数和查询耗时。真实 canary 单车双协议 1,392 个原始点聚合为 127 桶、254 浏览器点20.67ms5 台车 8,197 个原始点聚合为 376 桶、752 浏览器点86.53ms。导出采用单并发队列、5,000 行 TDengine 复合游标、30 分钟超时、100 万行双重配额、流式 CSV、`.part` + `fsync` + 原子发布,并持久化总行数、已处理行数、文件大小和完成时间。预发布门禁发现并修复 8 小时时区偏移和移动端面板压缩;最终桌面与 390×844 验收无横向溢出。RAW 保持离散证据而不伪造连续趋势;日里程趋势、多实例数据库队列/对象存储与更多可聚合指标仍待后续数据合同。
上线进度注记2026-07-14`platform-v2-20260714033929` 已发布到当前 ECS `115.29.187.205:20300`。API 与告警 evaluator 均为 active数据模式强制 `production`MySQL/TDengine/Redis/容量检查已接入Kafka lag 为 0viewer/operator/admin 鉴权、越权拒绝、核心真实查询、根页面与 hashed 静态资源均已冒烟。告警 evaluator 当前扫描 1159 条车辆来源快照约 89ms规则为 0未产生生产告警10k 车辆 × 20 规则的本地合成判定与批次规划约 20ms候选写入已改为每 500 条一次批处理、活跃事件不重复写、超过 5 分钟的动态遥测不继续累积。首轮发布发现 Web 归档遗漏并在 API 冒烟后修复runbook 已增加根页和主资源门禁。仍需补齐 ECS 10k 真实数据库写入基准、长时间稳定性、真实规则灰度及导出文件生产存储后,阶段 8/9 才可完全退出。
一期最终发布注记2026-07-14当前 release 为 `platform-v2-20260714093533`,前一 `platform-v2-20260714092531` 保留可回滚。全部 11 项迁移 journal 校验通过;平台 API、快照 evaluator、流 evaluator 均 activeMySQL/TDengine 可写9 个 Kafka 分区 lag 0、replay skipped 0三单元最近发布窗口 error 日志均为 0。未鉴权会话返回 401管理员会话、monitor、单车、active 告警、最新遥测、根页面与 hashed 主资源均通过真实生产冒烟Web 22 个测试文件 272 项、API `go test ./...``go vet ./...` 全部通过。阶段 9 的单 ECS 一期退出项关闭;云数据库峰值容量和多实例对象存储作为后续扩容观察项。
导出进度注记2026-07-14CSV 异步导出位于 `/opt/lingniu-vehicle-platform/data/exports` 持久目录,任务索引使用原子 `jobs.json`;完成文件和任务在 API 重启后仍可列出、下载,中断任务会显式失败而非永久 running。`platform-v2-20260714063725` 将旧的 1,000 行内存聚合/OFFSET/2 分钟实现升级为 100 万行流式游标任务。本地真实写入 1,000,000 行为 0.45 秒ECS 候选位置 1,451 行与 RAW 3,538 行在单并发队列中 0.41 秒完成。正式生产 1,457 行、162,533 字节文件在 API 重启前后 SHA-256 完全一致,桌面和 390×844 任务卡无溢出、控制台无错误。当前满足单 ECS 一期受控导出;多实例扩容前仍需迁移到数据库队列与 OSS/兼容对象存储。
高级告警与指标目录进度注记2026-07-14数值区间/区间外、布尔状态变化、OEM 范围、重复间隔、状态表和索引迁移已上线ECS 以不存在厂家范围的两条规则完成启用灰度evaluator 扫描 1159 条快照约 9ms、未误开事件随后停用并保留版本审计。规则编辑器已通过生产浏览器检查。统一 `/api/v2/metrics` 目录现提供协议字段映射及可检索/绘图/告警能力,编辑器和后端写入均按目录校验。事件 INSERT 占位符、状态批写和重复边界已有自动化契约测试;布尔真实翻转、重复抑制与活跃规则长稳仍需独立生产灰度。
事件时间告警灰度注记2026-07-14当前 ECS 对单车 `JT808` 建立 `speed_kmh >= 0`、持续 20 秒、重复间隔 1 小时的受限规则。真实新观测形成 candidate 后只打开 1 条事件,事件证据时间 `06:58:57.000`、接收时间 `06:58:57.934`、平台触发时间 `06:59:06.871`;事件随后关闭、通知设为已读、规则停用并保留审计。再次启用时 evaluator 记录 `candidates_advanced``duplicate_observations`,重复窗口内 `opened=0`。该门禁同时发现 MySQL `UNIX_TIMESTAMP(DATETIME(3))` 返回小数字符串导致候选及最近触发时间整数扫描失败,最终改为直接扫描 `DATETIME(3)` 并增加毫秒精度 SQL mock 契约。发布期间两次失败校验均由软链回滚陷阱恢复,最终 release 为 `platform-v2-20260714070507`
指标配置持久化进度注记2026-07-14`005_metric_catalog.sql` 已建立指标定义与协议源字段映射权威表;生产目录读取和告警规则校验共享该数据源,关闭或修改能力标记会立即约束后续规则写入。种子数据使用 `INSERT IGNORE`,发布不会回写覆盖运维配置;尚缺管理端 CRUD、版本审计以及把动态 RAW 指标纳入统一目录的受控发现流程。
告警写入性能注记2026-07-14`alert-benchmark` 默认使用连接级 MySQL 临时表,同时提供必须显式确认且带 advisory lock 的独立物理表持久模式,两者都复制候选表主键与持续时间索引且绝不写业务候选/事件。当前 ECS 到生产 RDS 的持久单事务基准10,000 行/500 批次为 195ms约 51,234 行/秒100,000 行为 1,971ms约 50,713 行/秒两档行数和物理表清理均独立校验。10 万行窗口观察到实例全局 redo 增量约 31MB但全局计数包含并发业务流量不能当作本事务精确归因。当前结果证明真实提交、网络和索引路径具备明显余量最终容量退出仍需 evaluator 活跃规则长稳、锁竞争和 RDS Performance Insights 峰值窗口证据。
车辆主数据进度注记2026-07-14采用“网关权威身份 + 平台补充主档”的混合边界。`vehicle_profile` 不覆盖 VIN/车牌/厂家,专门保存车型、车辆类型、所属企业、运营状态、接入服务商、首次接入和累计运行时长;`vehicle_profile_audit`、乐观版本和 admin-only 写入保证可追溯。单车详情已合并完整度与缺失字段,管理员可在档案卡原位维护,只读角色仅查看。外部主数据同步 API 与 CSV 管理入口已实现:单批 500 辆、dry-run、来源/版本幂等、同版本变更拒绝、人工及异源默认保护、显式接管、网关身份校验和逐 VIN 结果;写入持久化 `source_system/source_version/synced_at` 并生成不可变同步审计。具体车厂/GPS 厂商的定时连接器仍需按其认证和字段合同配置。
主数据同步上线注记2026-07-14`platform-v2-20260714051915` 已部署当前 ECSAPI 与 evaluator 均为 active。管理员 dry-run、operator 403、页面分包和外网根页均已验证生产可查询身份集合的 410 个唯一 VIN 在同一事务预演中耗时 175.44ms,重复预演 170.08ms,缺失身份为 0抽样档案前后完全一致。上线验证未向生产写入虚构车型/企业数据;正式写入需由权威车厂/GPS 数据文件或连接器触发。
主数据告警范围进度注记2026-07-14规则新增 `scopeModels/scopeCompanies`,评估器以 VIN 主键左联 `vehicle_profile`未维护对应维度的车辆不会误匹配协议、VIN、厂家、车型、企业范围均执行去重和大小上限避免配置膨胀拖慢 10 秒评估周期。规则编辑器、版本快照和数据库持久化共享同一字段合同。
## 2. 阶段 0必须先确认的决策
1. 车辆档案采用两者结合:网关身份表维护 VIN/车牌/厂家,平台 `vehicle_profile` 承载补充业务属性并预留外部同步来源元数据。
2. 认证是否复用 `ln-bi` 的 jumpToken/JWT 交换流程;角色和操作权限的权威来源是什么。
3. 动态指标目录的首批范围、中文名、单位、类型、协议字段映射及负责人。
4. 在线阈值是否采用“全局默认 + 协议覆盖”,建议默认 5 分钟并同时展示实际延迟。
5. 导出文件使用 OSS/兼容对象存储还是 ECS 本地盘,保留天数和单任务配额。
6. 告警时间基准使用事件时间,迟到数据容忍窗口和重复告警间隔的产品定义。
决策记录写 ADR不把未确认事项固化进页面常量。
## 3. 阶段 1数据与 API 基座
### 3.1 MySQL 迁移
- 扩充或关联车辆档案:车型、车辆类型、运营公司、运营状态、接入厂家、首次接入时间。
- 新建 `metric_definition``metric_protocol_mapping`
- 新建告警域:`alert_rule``alert_rule_scope``alert_event``alert_event_action``notification`
- 新建导出域:`export_job`,文件本体放对象/文件存储。
- 为所有业务表增加必要的状态、时间和组合索引;迁移只向前新增,不改写接入 writer 的现有主键语义。
### 3.2 BFF 契约
- 建立 `/api/v2`,保留旧只读接口作为过渡。
- 统一游标/offset 分页规则、排序白名单、时间格式、错误码、traceId 和最大查询范围。
- 生成 OpenAPI 与前端类型,避免 Go/TypeScript 手工重复漂移。
- 接入鉴权、操作权限、审计日志、限流、请求超时和幂等键。
### 3.3 验证
- 在生产结构副本/测试库运行 migration up/down 或等价回滚演练。
- 契约测试覆盖空值、未知协议、时区、分页边界和非法指标。
- 回归现有 Gateway、writers、realtime API 的写入与查询。
## 4. 阶段 2V2 前端基础
建议新建独立入口/目录,而不是在 `PrototypeApp.tsx` 内渐进堆叠:
```text
src-v2/
├── app/ # router、providers、权限和菜单
├── shared/ # ui、api、map、chart、table、utils
├── entities/vehicle/ # 车辆领域类型、状态、展示组件
├── features/ # 筛选、车辆选择、时间范围、指标选择
├── pages/ # 路由页面
└── widgets/ # 地图工作区、详情面板、事件时间线
```
基础能力包括正式路由、URL 查询恢复、认证门禁、统一 API client、查询缓存、错误边界、加载/空/错误状态、车辆状态 token、远程搜索选择器、服务端表格、ECharts 时序封装、AMap SDK。V2 默认禁止导入 `prototype/mockData.ts`
## 5. 阶段 3 至 8业务模块次序
每个模块遵循相同步骤:
1. 冻结页面信息架构和 DTO。
2. 先实现/验证后端接口和样本响应。
3. 实现页面及深链跳转。
4. 增加单元、契约和交互测试。
5. 用真实接口实际运行,检查加载、空、错误、权限和不同屏幕。
6. 更新已完成/未完成、接口和运维文档后再进入下一模块。
告警中心虽然菜单重要,但排在数据查询能力之后:规则依赖统一指标目录、车辆范围和时间语义。可在阶段 1 先建表/契约,阶段 8 再启用评估与通知。
## 6. 性能验收门槛
以下是建议基线,开发前应在目标 ECS/浏览器上校准:
| 场景 | 门槛 |
| --- | --- |
| 全局监控首屏 | 1,000 车 P95 API < 1s10,000 车聚合结果而非 10,000 DOM Marker主要交互保持流畅 |
| 地图拖动/缩放 | 只在 moveend/zoomend 防抖查询;旧请求可取消;无请求风暴 |
| 实时刷新 | 差量或低频批量刷新,不按车轮询;页面隐藏时降频/暂停 |
| 轨迹 | 单次返回受控;长时间段自动抽稀/分段;保留起终点、停车和告警关键点 |
| 历史曲线 | 浏览器展示点数设上限TDengine 服务端聚合/抽稀;缩放后按窗口补查 |
| 表格 | 全部服务端分页/排序/筛选;长列表虚拟化 |
| 导出 | 异步任务;并发、行数、时间范围和文件大小有限额;不占用同步 API worker |
| 告警 | 消费失败不丢 offset事件按规则+车辆+窗口幂等;可观测积压和评估延迟 |
## 7. 阶段 9当前 ECS 发布计划
现有运维说明表明接入链路和平台都使用 systemd 裸机 release + `current` 软链。最终发布应继续该模式,但平台、告警 worker、导出 worker使用独立 unit 和健康检查。
发布顺序:
1. 备份并执行 MySQL 向前兼容迁移,确认现有服务继续健康。
2. 上传新 release先启动告警/导出 worker 的 disabled 或 idle 模式。
3. 启动/切换平台 BFF验证 `/readyz``/api/ops/health`、核心查询和运行时配置。
4. 切换 V2 静态 Web进行全局监控、单车、轨迹、历史、接入、告警 smoke。
5. 开启 worker小范围规则验证后再开放全部规则。
6. 检查 Kafka lag、Redis 在线 key、TDengine/MySQL 可写性、CPU/内存/磁盘和错误日志。
7. 保留前一 release 与数据库兼容窗口;失败时先回切软链并停新 worker不做破坏性数据库回滚。
密钥放 `/opt/lingniu-vehicle-platform/env` 或对应 worker 环境文件,禁止写入 git、静态 JS 或部署日志。具体命令以实施时更新后的 runbook 为准,不能直接复制过时 release 名。
## 8. 一期完成清单
- 六大业务模块和导出任务满足目标文件的功能验收。
- 页面在 1440×900 正常,常用操作三次点击内,真实数据无 mock 残留。
- 1k/10k 地图、百万点历史、异步导出、告警积压均有测试记录。
- OpenAPI、数据字典、环境变量、启动、部署、回滚、监控与未完成事项完整。
- ECS 上全部新旧服务健康Kafka lag 回零,运行版本可从健康接口确认。

View File

@@ -1,146 +0,0 @@
# Vehicle Telemetry Internal Fields
This document defines the internal telemetry field model for Lingniu hydrogen
vehicle operations. Protocol fields such as GB/T 32960, JT808, MQTT, and vendor
extensions must be mapped into these fields before storage, statistics, or
business reporting.
## Design Rules
- Business statistics use internal fields, not protocol field names.
- Every internal field has one stable unit and one business meaning.
- Protocol-specific raw values remain traceable through `source_protocol`,
`protocol_version`, `metadata`, and raw archive references.
- Core operational fields are stored as typed columns or typed payload fields.
- Unstable vendor-specific fields can stay in extension metadata until promoted.
- Hydrogen leak safety is a first-class critical safety event, not a generic
alarm count.
## Identity And Time
| Internal field | Type | Unit | Meaning | GB/T 32960 source |
|---|---:|---|---|---|
| `vin` | string | - | Vehicle VIN and partition key | header VIN |
| `event_time` | instant | - | Device collection time | command body timestamp |
| `ingest_time` | instant | - | Platform receive time | ingest server clock |
| `source_protocol` | enum | - | Source adapter | `GB32960` |
| `protocol_version` | string | - | Source protocol version | `V2016` / `V2025` |
## Real-Time Vehicle State
| Internal field | Type | Unit | Meaning | GB/T 32960 source |
|---|---:|---|---|---|
| `vehicle_state` | enum | - | Started, shutdown, other, invalid | vehicle block `vehicleState` |
| `charging_state` | enum | - | Charging state | vehicle block `chargingState` |
| `running_mode` | enum | - | Electric, hybrid, fuel, other | vehicle block `runningMode` |
| `speed_kmh` | double | km/h | Current speed | vehicle block `speedKmh` |
| `total_mileage_km` | double | km | Odometer mileage | vehicle block `totalMileageKm` |
| `gear_level` | integer | - | Gear value | vehicle block `gearRaw` low 4 bits |
| `accelerator_pedal` | double | % | Accelerator pedal opening | V2016 vehicle block |
| `brake_pedal` | double | % | Brake pedal opening | V2016 vehicle block |
## Location
| Internal field | Type | Unit | Meaning | GB/T 32960 source |
|---|---:|---|---|---|
| `longitude` | double | deg | Longitude | position block |
| `latitude` | double | deg | Latitude | position block |
| `altitude_m` | double | m | Altitude when available | extension or other protocol |
| `direction_deg` | double | deg | Direction when available | extension or other protocol |
| `location_status_raw` | long | - | Raw location status bits | position block `statusFlag` |
## Battery And Electricity
| Internal field | Type | Unit | Meaning | GB/T 32960 source |
|---|---:|---|---|---|
| `battery_soc` | double | % | Battery SOC | vehicle block `socPercent` |
| `battery_voltage_v` | double | V | Battery total voltage | vehicle block `totalVoltageV` |
| `battery_current_a` | double | A | Battery total current | vehicle block `totalCurrentA` |
| `battery_power_kw` | double | kW | Derived instantaneous battery power | `voltage * current / 1000` in statistics |
| `battery_temperature_max_c` | integer | C | Max battery temperature | temperature or extreme blocks |
| `battery_temperature_min_c` | integer | C | Min battery temperature | temperature or extreme blocks |
| `daily_electricity_kwh` | double | kWh | Daily electricity usage | statistics time integral |
Daily electricity must be computed in the statistics layer from typed telemetry
points. Do not accumulate protocol raw values directly.
## Hydrogen And Fuel Cell
| Internal field | Type | Unit | Meaning | GB/T 32960 source |
|---|---:|---|---|---|
| `fc_voltage_v` | double | V | Fuel cell voltage | fuel cell block |
| `fc_current_a` | double | A | Fuel cell current | fuel cell block |
| `fc_temp_c` | double | C | Fuel cell or hydrogen system temperature | fuel cell block |
| `hydrogen_remaining_kg` | double | kg | Remaining hydrogen mass | vendor extension or capacity conversion |
| `hydrogen_remaining_percent` | double | % | Remaining hydrogen percent | V2025 fuel cell block |
| `hydrogen_high_pressure_mpa` | double | MPa | Hydrogen high pressure | fuel cell block |
| `hydrogen_low_pressure_mpa` | double | MPa | Hydrogen low pressure when available | extension or other protocol |
| `daily_hydrogen_kg` | double | kg | Daily hydrogen consumed | statistics layer |
| `hydrogen_mileage_efficiency` | double | km/kg | Mileage per kg hydrogen | statistics layer |
Daily hydrogen usage should be calculated from adjacent hydrogen remaining
values. A rising value indicates refueling and must not be counted as negative
consumption.
## Hydrogen Tank Safety
| Internal field | Type | Unit | Meaning | GB/T 32960 source |
|---|---:|---|---|---|
| `safety_category` | enum | - | `GENERAL`, `TANK_PRESSURE`, `TANK_TEMPERATURE`, `HYDROGEN_LEAK` | alarm bit classification |
| `hydrogen_leak_detected` | boolean | - | Whether hydrogen leak is detected | general alarm bit `HYDROGEN_LEAK` |
| `hydrogen_leak_level` | enum | - | `NONE`, `WARNING`, `CRITICAL`, `UNKNOWN` | internal rule |
| `hydrogen_leak_action_required` | boolean | - | Whether immediate handling is required | internal rule |
| `tank_pressure_status` | enum | - | Normal, warning, critical | hydrogen pressure bit or threshold rule |
| `tank_temperature_status` | enum | - | Normal, warning, critical | hydrogen temp bit or threshold rule |
Safety rule:
```text
If HYDROGEN_LEAK is present:
alarm.level = CRITICAL
safety_category = HYDROGEN_LEAK
hydrogen_leak_detected = true
hydrogen_leak_level = CRITICAL
hydrogen_leak_action_required = true
```
Hydrogen leak overrides otherwise normal vehicle, speed, battery, or hydrogen
state. It must be counted independently from generic alarms.
## Daily Operation Statistics
Daily statistics are generated by `stat_date + vin`.
| Internal field | Type | Unit | Meaning |
|---|---:|---|---|
| `stat_date` | date | - | Statistics date |
| `vin` | string | - | Vehicle VIN |
| `first_event_time` | instant | - | First valid telemetry time |
| `last_event_time` | instant | - | Last valid telemetry time |
| `online_minutes` | double | min | Online duration |
| `running_minutes` | double | min | Running duration |
| `daily_mileage_km` | double | km | Daily mileage |
| `daily_electricity_kwh` | double | kWh | Daily electricity usage |
| `daily_hydrogen_kg` | double | kg | Daily hydrogen usage |
| `hydrogen_added_kg` | double | kg | Refueled hydrogen |
| `km_per_kg_hydrogen` | double | km/kg | Hydrogen efficiency |
| `kwh_per_100km` | double | kWh/100km | Electricity efficiency |
| `alarm_count` | integer | - | Generic alarm count |
| `tank_safety_alarm_count` | integer | - | Tank safety alarm count |
| `hydrogen_leak_alarm_count` | integer | - | Hydrogen leak alarm count |
| `hydrogen_leak_duration_seconds` | long | s | Hydrogen leak duration |
| `data_quality_level` | enum | - | Good, partial, bad |
## Implementation Status
- `RealtimePayload` already uses internal operational field names for state,
location, battery, fuel cell, and hydrogen fields.
- `AlarmPayload` carries internal hydrogen safety fields.
- `Gb32960EventMapper` maps GB/T 32960 alarm bits into internal safety fields.
- `TelemetrySnapshot` publishes full-field internal telemetry values through the
Kafka protobuf envelope.
- `event-history-service`, `vehicle-state-service`, and `vehicle-stat-service`
consume the same full-field telemetry snapshot contract.
- `event-history-service` CSV export can flatten selected internal fields with
`fields=field_a,field_b`; media archive references are exposed through
`rawArchiveUri` by falling back to `MediaMeta.archiveRef`.

View File

@@ -6,6 +6,7 @@ RUN go mod download
COPY . .
RUN CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags="-s -w" -o /out/gateway ./cmd/gateway \
&& CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags="-s -w" -o /out/feichi-bridge ./cmd/feichi-bridge \
&& CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags="-s -w" -o /out/history-writer ./cmd/history-writer \
&& CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags="-s -w" -o /out/stat-writer ./cmd/stat-writer \
&& CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -trimpath -ldflags="-s -w" -o /out/realtime-api ./cmd/realtime-api
@@ -18,6 +19,7 @@ RUN apt-get update \
WORKDIR /app
COPY --from=build /out/gateway /app/gateway
COPY --from=build /out/feichi-bridge /app/feichi-bridge
COPY --from=build /out/history-writer /app/history-writer
COPY --from=build /out/stat-writer /app/stat-writer
COPY --from=build /out/realtime-api /app/realtime-api

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,211 @@
package main
import (
"context"
"encoding/json"
"errors"
"fmt"
"os"
"os/signal"
"strconv"
"strings"
"syscall"
"time"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/feichibridge"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/health"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/observability"
)
type authSecret struct {
Username string `json:"username"`
Password string `json:"password"`
}
type targetSecret struct {
PlatformID string `json:"platformId"`
Username string `json:"username"`
Password string `json:"password"`
}
type config struct {
BaseURL string
AuthSecretFile string
TargetSecret string
TargetAddress string
StateFile string
HealthAddress string
HTTPTimeout time.Duration
TargetTimeout time.Duration
OCRImage string
OCRTimeout time.Duration
LoginAttempts int
LoginRetryDelay time.Duration
Service feichibridge.ServiceConfig
}
func main() {
logger := observability.NewLogger("feichi-bridge")
cfg, err := loadConfig()
if err != nil {
logger.Error("load configuration failed", "error", err)
os.Exit(1)
}
auth, err := readJSONSecret[authSecret](cfg.AuthSecretFile)
if err != nil {
logger.Error("read Feichi API secret failed", "error", err)
os.Exit(1)
}
credentials, err := readJSONSecret[targetSecret](cfg.TargetSecret)
if err != nil {
logger.Error("read GB/T 32960 target secret failed", "error", err)
os.Exit(1)
}
source, err := feichibridge.NewAuthenticatedAPIClient(
cfg.BaseURL,
feichibridge.LoginCredentials{
Username: auth.Username, Password: auth.Password,
MaxAttempts: cfg.LoginAttempts, RetryDelay: cfg.LoginRetryDelay,
},
feichibridge.DockerCaptchaSolver{Image: cfg.OCRImage, Timeout: cfg.OCRTimeout},
cfg.HTTPTimeout,
)
if err != nil {
logger.Error("build Feichi API client failed", "error", err)
os.Exit(1)
}
state, err := feichibridge.OpenStateStore(cfg.StateFile)
if err != nil {
logger.Error("open bridge state failed", "error", err)
os.Exit(1)
}
target, err := feichibridge.NewTarget(feichibridge.TargetConfig{
Address: cfg.TargetAddress, PlatformID: credentials.PlatformID,
Username: credentials.Username, Password: credentials.Password,
Timeout: cfg.TargetTimeout,
}, state)
if err != nil {
logger.Error("build GB/T 32960 target failed", "error", err)
os.Exit(1)
}
registry := metrics.NewRegistry()
service, err := feichibridge.NewService(cfg.Service, source, target, state, logger, registry)
if err != nil {
logger.Error("build bridge service failed", "error", err)
os.Exit(1)
}
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
health.Start(ctx, logger, health.NewServer(cfg.HealthAddress, "feichi-bridge", []health.Check{
{Name: "bridge", Check: service.Ready},
}, registry))
logger.Info("Feichi bridge starting",
"base_url", cfg.BaseURL,
"target_address", cfg.TargetAddress,
"poll_interval", cfg.Service.PollInterval,
"backfill_enabled", cfg.Service.BackfillEnabled,
)
if strings.HasPrefix(strings.ToLower(cfg.BaseURL), "http://") {
logger.Warn("Feichi source uses clear-text HTTP; deploy only through a controlled egress path")
}
if err := service.Run(ctx); err != nil && !errors.Is(err, context.Canceled) {
logger.Error("Feichi bridge stopped unexpectedly", "error", err)
os.Exit(1)
}
logger.Info("Feichi bridge stopped")
}
func loadConfig() (config, error) {
cfg := config{
BaseURL: strings.TrimSpace(os.Getenv("FEICHI_BASE_URL")),
AuthSecretFile: env("FEICHI_AUTH_SECRET_FILE", strings.TrimSpace(os.Getenv("FEICHI_AUTH_HEADERS_FILE"))),
TargetSecret: strings.TrimSpace(os.Getenv("FEICHI_TARGET_SECRET_FILE")),
TargetAddress: env("FEICHI_TARGET_ADDR", "127.0.0.1:32960"),
StateFile: env("FEICHI_STATE_FILE", "/var/lib/lingniu-feichi-bridge/state.json"),
HealthAddress: env("HEALTH_ADDR", "127.0.0.1:20219"),
HTTPTimeout: seconds("FEICHI_HTTP_TIMEOUT_SECONDS", 15),
TargetTimeout: seconds("FEICHI_TARGET_TIMEOUT_SECONDS", 10),
OCRImage: env("FEICHI_OCR_IMAGE", "lingniu/feichi-captcha-ocr:1.0.0"),
OCRTimeout: seconds("FEICHI_OCR_TIMEOUT_SECONDS", 20),
LoginAttempts: envInt("FEICHI_LOGIN_MAX_ATTEMPTS", 20),
LoginRetryDelay: seconds("FEICHI_LOGIN_RETRY_SECONDS", 1),
Service: feichibridge.ServiceConfig{
PollInterval: seconds("FEICHI_POLL_INTERVAL_SECONDS", 10),
DiscoveryInterval: seconds("FEICHI_DISCOVERY_INTERVAL_SECONDS", 300),
BackfillInterval: seconds("FEICHI_BACKFILL_INTERVAL_SECONDS", 3600),
BackfillLookback: seconds("FEICHI_BACKFILL_LOOKBACK_SECONDS", 3600),
BackfillWindow: seconds("FEICHI_BACKFILL_WINDOW_SECONDS", 1200),
BackfillSafetyLag: seconds("FEICHI_BACKFILL_SAFETY_SECONDS", 30),
SourceStaleAfter: seconds("FEICHI_SOURCE_STALE_SECONDS", 120),
FetchConcurrency: envInt("FEICHI_FETCH_CONCURRENCY", 4),
BackfillEnabled: envBool("FEICHI_BACKFILL_ENABLED", true),
StaleReissueEnabled: envBool("FEICHI_STALE_REISSUE_ENABLED", true),
},
}
var missing []string
if cfg.BaseURL == "" {
missing = append(missing, "FEICHI_BASE_URL")
}
if cfg.AuthSecretFile == "" {
missing = append(missing, "FEICHI_AUTH_SECRET_FILE")
}
if cfg.TargetSecret == "" {
missing = append(missing, "FEICHI_TARGET_SECRET_FILE")
}
if len(missing) > 0 {
return config{}, fmt.Errorf("required configuration missing: %s", strings.Join(missing, ", "))
}
if cfg.Service.BackfillWindow > 24*time.Hour {
return config{}, errors.New("FEICHI_BACKFILL_WINDOW_SECONDS must not exceed 86400")
}
return cfg, nil
}
func readJSONSecret[T any](path string) (T, error) {
var value T
encoded, err := os.ReadFile(path)
if err != nil {
return value, err
}
if err := json.Unmarshal(encoded, &value); err != nil {
return value, fmt.Errorf("decode %s: %w", path, err)
}
return value, nil
}
func env(name, fallback string) string {
if value := strings.TrimSpace(os.Getenv(name)); value != "" {
return value
}
return fallback
}
func envInt(name string, fallback int) int {
value := strings.TrimSpace(os.Getenv(name))
if value == "" {
return fallback
}
parsed, err := strconv.Atoi(value)
if err != nil || parsed <= 0 {
return fallback
}
return parsed
}
func envBool(name string, fallback bool) bool {
value := strings.TrimSpace(os.Getenv(name))
if value == "" {
return fallback
}
parsed, err := strconv.ParseBool(value)
if err != nil {
return fallback
}
return parsed
}
func seconds(name string, fallback int) time.Duration {
return time.Duration(envInt(name, fallback)) * time.Second
}

View File

@@ -0,0 +1,490 @@
package main
import (
"context"
"encoding/json"
"errors"
"fmt"
"log/slog"
"os"
"os/signal"
"strconv"
"strings"
"sync"
"syscall"
"time"
"github.com/segmentio/kafka-go"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/health"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/observability"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/realtime"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/topics"
)
func main() {
logger := observability.NewLogger("vehicle-fields-projector")
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
cfg := loadConfig()
if err := cfg.Validate(); err != nil {
logger.Error("invalid fields projector config", "error", err)
os.Exit(1)
}
if err := pingKafka(ctx, cfg.KafkaBrokers); err != nil {
logger.Error("kafka connectivity check failed", "error", err)
os.Exit(1)
}
registry := metrics.NewRegistry()
recordConfigMetrics(registry, cfg)
for _, route := range cfg.Routes {
metrics.RegisterKafkaConsumerInfo(registry, "vehicle-fields-projector", route.GroupID, []string{route.RawTopic})
}
health.Start(ctx, logger, health.NewServer(env("HEALTH_ADDR", ""), "vehicle-fields-projector", nil, registry))
logger.Info("fields projector started",
"kafka_brokers", strings.Join(cfg.KafkaBrokers, ","),
"group_prefix", cfg.GroupPrefix,
"workers_per_protocol", cfg.WorkersPerProtocol,
"batch_size", cfg.BatchSize,
"batch_wait_ms", cfg.BatchWait.Milliseconds(),
"operation_timeout_ms", cfg.OperationTimeout.Milliseconds(),
"retry_delay_ms", cfg.RetryDelay.Milliseconds(),
"start_offset", cfg.StartOffsetName)
var workers sync.WaitGroup
for _, route := range cfg.Routes {
for workerID := 1; workerID <= cfg.WorkersPerProtocol; workerID++ {
workers.Add(1)
go func(route projectionRoute, workerID int) {
defer workers.Done()
runProjector(ctx, logger.With("protocol", route.Protocol, "worker", workerID), registry, cfg, route, workerID)
}(route, workerID)
}
}
workers.Wait()
}
type config struct {
KafkaBrokers []string
GroupPrefix string
Routes []projectionRoute
WorkersPerProtocol int
BatchSize int
BatchWait time.Duration
OperationTimeout time.Duration
RetryDelay time.Duration
StartOffset int64
StartOffsetName string
}
type projectionRoute struct {
Protocol envelope.Protocol
RawTopic string
FieldsTopic string
GroupID string
}
func loadConfig() config {
groupPrefix := env("FIELDS_PROJECTOR_GROUP_PREFIX", "vehicle-fields-projector-v1")
startOffsetName := strings.ToLower(env("KAFKA_START_OFFSET", "last"))
startOffset := int64(kafka.LastOffset)
if startOffsetName == "first" {
startOffset = kafka.FirstOffset
} else {
startOffsetName = "last"
}
routes := []projectionRoute{
{Protocol: envelope.ProtocolGB32960, RawTopic: env("KAFKA_TOPIC_GB32960_RAW", topics.RawGB32960), FieldsTopic: env("KAFKA_TOPIC_GB32960_FIELDS", topics.FieldsGB32960)},
{Protocol: envelope.ProtocolJT808, RawTopic: env("KAFKA_TOPIC_JT808_RAW", topics.RawJT808), FieldsTopic: env("KAFKA_TOPIC_JT808_FIELDS", topics.FieldsJT808)},
{Protocol: envelope.ProtocolYutongMQTT, RawTopic: env("KAFKA_TOPIC_YUTONG_MQTT_RAW", topics.RawYutongMQTT), FieldsTopic: env("KAFKA_TOPIC_YUTONG_MQTT_FIELDS", topics.FieldsYutongMQTT)},
}
for index := range routes {
routes[index].GroupID = groupPrefix + "-" + strings.ToLower(strings.ReplaceAll(string(routes[index].Protocol), "_", "-"))
}
return config{
KafkaBrokers: splitCSV(env("KAFKA_BROKERS", "127.0.0.1:9092")),
GroupPrefix: groupPrefix,
Routes: routes,
WorkersPerProtocol: envInt("FIELDS_PROJECTOR_WORKERS_PER_PROTOCOL", 1),
BatchSize: envInt("FIELDS_PROJECTOR_BATCH_SIZE", 500),
BatchWait: time.Duration(envInt("FIELDS_PROJECTOR_BATCH_WAIT_MS", 20)) * time.Millisecond,
OperationTimeout: time.Duration(envInt("FIELDS_PROJECTOR_OPERATION_TIMEOUT_MS", 30000)) * time.Millisecond,
RetryDelay: time.Duration(envInt("FIELDS_PROJECTOR_RETRY_DELAY_MS", 500)) * time.Millisecond,
StartOffset: startOffset,
StartOffsetName: startOffsetName,
}
}
func (c config) Validate() error {
if len(c.KafkaBrokers) == 0 {
return errors.New("kafka brokers are required")
}
if strings.TrimSpace(c.GroupPrefix) == "" {
return errors.New("fields projector group prefix is required")
}
if c.WorkersPerProtocol < 1 || c.BatchSize < 1 || c.BatchWait <= 0 || c.OperationTimeout <= 0 || c.RetryDelay <= 0 {
return errors.New("fields projector worker, batch and timeout settings must be positive")
}
raw := make(map[string]string, len(c.Routes))
fields := make(map[string]string, len(c.Routes))
groups := map[string]struct{}{}
for _, route := range c.Routes {
protocol := string(route.Protocol)
raw[protocol] = route.RawTopic
fields[protocol] = route.FieldsTopic
if strings.TrimSpace(route.GroupID) == "" {
return fmt.Errorf("consumer group is required for protocol %s", route.Protocol)
}
if _, exists := groups[route.GroupID]; exists {
return fmt.Errorf("duplicate consumer group %q", route.GroupID)
}
groups[route.GroupID] = struct{}{}
}
return topics.ValidateKafkaRawFields(raw, fields)
}
type kafkaBatchWriter interface {
WriteMessages(context.Context, ...kafka.Message) error
}
type kafkaMessageFetcher interface {
FetchMessage(context.Context) (kafka.Message, error)
}
type kafkaMessageCommitter interface {
CommitMessages(context.Context, ...kafka.Message) error
}
func runProjector(ctx context.Context, logger *slog.Logger, registry *metrics.Registry, cfg config, route projectionRoute, workerID int) {
reader := kafka.NewReader(kafka.ReaderConfig{
Brokers: cfg.KafkaBrokers,
GroupID: route.GroupID,
GroupTopics: []string{route.RawTopic},
StartOffset: cfg.StartOffset,
MinBytes: 1,
MaxBytes: 10e6,
})
defer reader.Close()
writer := &kafka.Writer{
Addr: kafka.TCP(cfg.KafkaBrokers...),
Balancer: &kafka.Hash{},
RequiredAcks: kafka.RequireAll,
AllowAutoTopicCreation: false,
BatchTimeout: cfg.BatchWait,
Async: false,
}
defer writer.Close()
labels := metrics.Labels{"protocol": string(route.Protocol), "worker": strconv.Itoa(workerID)}
registry.SetGauge("vehicle_fields_projector_worker_active", labels, 1)
defer registry.SetGauge("vehicle_fields_projector_worker_active", labels, 0)
for {
first, err := reader.FetchMessage(ctx)
if err != nil {
if ctx.Err() != nil {
return
}
logger.Error("kafka fetch failed", "error", err)
continue
}
batch := collectBatch(ctx, reader, first, cfg.BatchSize, cfg.BatchWait)
processBatchReliably(ctx, logger, registry, writer, reader, route, batch, cfg.OperationTimeout, cfg.RetryDelay)
}
}
func collectBatch(ctx context.Context, fetcher kafkaMessageFetcher, first kafka.Message, maxSize int, maxWait time.Duration) []kafka.Message {
if maxSize <= 1 {
return []kafka.Message{first}
}
batch := []kafka.Message{first}
deadline := time.Now().Add(maxWait)
for len(batch) < maxSize {
remaining := time.Until(deadline)
if remaining <= 0 {
break
}
fetchCtx, cancel := context.WithTimeout(ctx, remaining)
message, err := fetcher.FetchMessage(fetchCtx)
cancel()
if err != nil {
break
}
batch = append(batch, message)
}
return batch
}
func processBatchReliably(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, writer kafkaBatchWriter, committer kafkaMessageCommitter, route projectionRoute, messages []kafka.Message, operationTimeout time.Duration, retryDelay time.Duration) {
labels := metrics.Labels{"protocol": string(route.Protocol)}
defer registry.SetGauge("vehicle_fields_projector_retry_pending_messages", labels, 0)
for len(messages) > 0 {
operationCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), operationTimeout)
outputs, err := projectBatch(operationCtx, logger, registry, writer, route, messages)
cancel()
if err != nil {
registry.SetGauge("vehicle_fields_projector_retry_pending_messages", labels, float64(len(messages)))
registry.IncCounter("vehicle_fields_projector_batch_retries_total", metrics.Labels{"protocol": string(route.Protocol), "reason": "write_error"})
if !waitForRetry(ctx, retryDelay) {
return
}
continue
}
commitCtx, commitCancel := context.WithTimeout(context.WithoutCancel(ctx), operationTimeout)
err = committer.CommitMessages(commitCtx, messages...)
commitCancel()
if err == nil {
for _, message := range messages {
recordMessageMetric(registry, "vehicle_fields_projector_kafka_commits_total", message.Topic, "ok")
}
_ = outputs
return
}
for _, message := range messages {
recordMessageMetric(registry, "vehicle_fields_projector_kafka_commits_total", message.Topic, "error")
}
registry.SetGauge("vehicle_fields_projector_retry_pending_messages", labels, float64(len(messages)))
registry.IncCounter("vehicle_fields_projector_batch_retries_total", metrics.Labels{"protocol": string(route.Protocol), "reason": "commit_error"})
logger.Error("kafka source offset commit failed", "topic", route.RawTopic, "messages", len(messages), "error", err)
if !retryCommit(ctx, logger, registry, committer, messages, route.Protocol, operationTimeout, retryDelay) {
return
}
return
}
}
func projectBatch(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, writer kafkaBatchWriter, route projectionRoute, messages []kafka.Message) (int, error) {
if len(messages) == 0 {
return 0, nil
}
outputs := make([]kafka.Message, 0, len(messages))
setBatchPending(registry, route.Protocol, len(messages), 0)
defer setBatchPending(registry, route.Protocol, -len(messages), -len(outputs))
for _, message := range messages {
recordMessageMetric(registry, "vehicle_fields_projector_kafka_messages_total", message.Topic, "received")
recordKafkaLag(registry, message)
var raw envelope.FrameEnvelope
if err := json.Unmarshal(message.Value, &raw); err != nil {
recordMessageMetric(registry, "vehicle_fields_projector_kafka_messages_total", message.Topic, "invalid_json")
logger.Warn("skip invalid raw envelope json", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "error", err)
continue
}
if status, err := topics.ValidateRawEnvelope(route.RawTopic, raw); err != nil {
recordMessageMetric(registry, "vehicle_fields_projector_kafka_messages_total", message.Topic, status)
logger.Warn("skip mismatched raw envelope", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "protocol", raw.Protocol, "event_id", raw.StableEventID(), "error", err)
continue
}
fields, ok := realtime.BuildFieldsEnvelope(raw)
if !ok {
status := "skipped_missing_fields"
if !envelope.IsRealtimeTelemetryFrame(raw) {
status = "skipped_non_realtime"
}
recordProjectionMetric(registry, route.Protocol, status, 0)
continue
}
if status, err := topics.ValidateFieldsEnvelope(route.FieldsTopic, fields); err != nil {
recordProjectionMetric(registry, route.Protocol, status, len(fields.Fields))
logger.Warn("skip invalid projected fields envelope", "topic", route.FieldsTopic, "protocol", fields.Protocol, "event_id", fields.StableEventID(), "error", err)
continue
}
payload, err := fields.MarshalJSONBytes()
if err != nil {
recordProjectionMetric(registry, route.Protocol, "marshal_error", len(fields.Fields))
logger.Warn("skip fields envelope marshal error", "topic", route.FieldsTopic, "event_id", fields.StableEventID(), "error", err)
continue
}
outputs = append(outputs, kafka.Message{
Topic: route.FieldsTopic,
Key: fields.KafkaKey(),
Value: payload,
Time: message.Time,
})
recordProjectionMetric(registry, route.Protocol, "projected", len(fields.Fields))
}
setBatchPending(registry, route.Protocol, 0, len(outputs))
defer setBatchPending(registry, route.Protocol, 0, -len(outputs))
if len(outputs) == 0 {
return 0, nil
}
started := time.Now()
err := writer.WriteMessages(ctx, outputs...)
status := "ok"
if err != nil {
status = "error"
}
recordWriteDuration(registry, route.FieldsTopic, status, time.Since(started))
for range outputs {
recordMessageMetric(registry, "vehicle_fields_projector_kafka_writes_total", route.FieldsTopic, status)
}
if err != nil {
logger.Error("fields kafka write failed", "topic", route.FieldsTopic, "messages", len(outputs), "error", err)
return len(outputs), err
}
return len(outputs), nil
}
func retryCommit(ctx context.Context, logger interface {
Error(string, ...any)
}, registry *metrics.Registry, committer kafkaMessageCommitter, messages []kafka.Message, protocol envelope.Protocol, operationTimeout, retryDelay time.Duration) bool {
for {
if !waitForRetry(ctx, retryDelay) {
return false
}
operationCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), operationTimeout)
err := committer.CommitMessages(operationCtx, messages...)
cancel()
if err == nil {
for _, message := range messages {
recordMessageMetric(registry, "vehicle_fields_projector_kafka_commits_total", message.Topic, "ok")
}
return true
}
for _, message := range messages {
recordMessageMetric(registry, "vehicle_fields_projector_kafka_commits_total", message.Topic, "error")
}
registry.IncCounter("vehicle_fields_projector_batch_retries_total", metrics.Labels{"protocol": string(protocol), "reason": "commit_error"})
logger.Error("kafka source offset commit retry failed", "messages", len(messages), "error", err)
}
}
func waitForRetry(ctx context.Context, delay time.Duration) bool {
timer := time.NewTimer(delay)
defer timer.Stop()
select {
case <-ctx.Done():
return false
case <-timer.C:
return true
}
}
func pingKafka(ctx context.Context, brokers []string) error {
if len(brokers) == 0 {
return errors.New("kafka broker is required")
}
checkCtx, cancel := context.WithTimeout(ctx, 5*time.Second)
defer cancel()
conn, err := kafka.DialContext(checkCtx, "tcp", brokers[0])
if err != nil {
return err
}
return conn.Close()
}
func recordConfigMetrics(registry *metrics.Registry, cfg config) {
registry.SetGauge("vehicle_fields_projector_config", metrics.Labels{"setting": "workers_per_protocol"}, float64(cfg.WorkersPerProtocol))
registry.SetGauge("vehicle_fields_projector_config", metrics.Labels{"setting": "batch_size"}, float64(cfg.BatchSize))
registry.SetGauge("vehicle_fields_projector_config", metrics.Labels{"setting": "batch_wait_ms"}, float64(cfg.BatchWait.Milliseconds()))
registry.SetGauge("vehicle_fields_projector_config", metrics.Labels{"setting": "operation_timeout_ms"}, float64(cfg.OperationTimeout.Milliseconds()))
}
func recordMessageMetric(registry *metrics.Registry, name, topic, status string) {
if registry == nil {
return
}
labels := metrics.Labels{"topic": topic, "status": status}
registry.IncCounter(name, labels)
switch name {
case "vehicle_fields_projector_kafka_messages_total":
metrics.RecordLastActivity(registry, "vehicle_fields_projector_last_message_unix_seconds", labels)
case "vehicle_fields_projector_kafka_writes_total":
metrics.RecordLastActivity(registry, "vehicle_fields_projector_last_write_unix_seconds", labels)
case "vehicle_fields_projector_kafka_commits_total":
metrics.RecordLastActivity(registry, "vehicle_fields_projector_last_commit_unix_seconds", labels)
}
}
func recordProjectionMetric(registry *metrics.Registry, protocol envelope.Protocol, status string, fieldCount int) {
if registry == nil {
return
}
labels := metrics.Labels{"protocol": string(protocol), "status": status}
registry.IncCounter("vehicle_fields_projector_projections_total", labels)
if fieldCount > 0 {
registry.SetGauge("vehicle_fields_projector_field_count", labels, float64(fieldCount))
}
}
func recordKafkaLag(registry *metrics.Registry, message kafka.Message) {
if registry == nil {
return
}
lag := message.HighWaterMark - message.Offset - 1
if lag < 0 {
lag = 0
}
registry.SetGauge("vehicle_fields_projector_kafka_lag", metrics.Labels{
"topic": message.Topic, "partition": strconv.Itoa(message.Partition),
}, float64(lag))
}
var projectorPendingMu sync.Mutex
var projectorPendingMessages = map[envelope.Protocol]int{}
var projectorPendingFields = map[envelope.Protocol]int{}
func setBatchPending(registry *metrics.Registry, protocol envelope.Protocol, messagesDelta, fieldsDelta int) {
if registry == nil {
return
}
projectorPendingMu.Lock()
projectorPendingMessages[protocol] += messagesDelta
projectorPendingFields[protocol] += fieldsDelta
messages := projectorPendingMessages[protocol]
fields := projectorPendingFields[protocol]
projectorPendingMu.Unlock()
labels := metrics.Labels{"protocol": string(protocol)}
registry.SetGauge("vehicle_fields_projector_batch_pending_messages", labels, float64(messages))
registry.SetGauge("vehicle_fields_projector_batch_pending_fields", labels, float64(fields))
}
var projectorWriteBucketsMS = []float64{1, 5, 10, 25, 50, 100, 250, 500, 1000, 5000}
func recordWriteDuration(registry *metrics.Registry, topic, status string, elapsed time.Duration) {
if registry == nil {
return
}
registry.ObserveHistogram("vehicle_fields_projector_write_duration_ms_histogram", metrics.Labels{
"topic": topic, "status": status,
}, projectorWriteBucketsMS, float64(elapsed.Milliseconds()))
}
func env(key, fallback string) string {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
return value
}
func envInt(key string, fallback int) int {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
parsed, err := strconv.Atoi(value)
if err != nil {
return fallback
}
return parsed
}
func splitCSV(value string) []string {
var out []string
for _, item := range strings.Split(value, ",") {
item = strings.TrimSpace(item)
if item != "" {
out = append(out, item)
}
}
return out
}

View File

@@ -0,0 +1,242 @@
package main
import (
"context"
"encoding/json"
"errors"
"strings"
"sync"
"testing"
"time"
"github.com/segmentio/kafka-go"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/topics"
)
func TestLoadConfigCreatesProtocolIsolatedConsumerGroups(t *testing.T) {
t.Setenv("FIELDS_PROJECTOR_GROUP_PREFIX", "projector-test")
t.Setenv("KAFKA_START_OFFSET", "first")
cfg := loadConfig()
if cfg.StartOffset != kafka.FirstOffset || cfg.StartOffsetName != "first" {
t.Fatalf("start offset = %d/%q", cfg.StartOffset, cfg.StartOffsetName)
}
wantGroups := map[envelope.Protocol]string{
envelope.ProtocolGB32960: "projector-test-gb32960",
envelope.ProtocolJT808: "projector-test-jt808",
envelope.ProtocolYutongMQTT: "projector-test-yutong-mqtt",
}
for _, route := range cfg.Routes {
if route.GroupID != wantGroups[route.Protocol] {
t.Fatalf("group for %s = %q", route.Protocol, route.GroupID)
}
}
if err := cfg.Validate(); err != nil {
t.Fatalf("Validate() error = %v", err)
}
}
func TestProjectBatchUsesPrecomputedFieldsAndPreservesSourceMetadata(t *testing.T) {
raw := projectorRawEnvelope()
payload, err := raw.MarshalJSONBytes()
if err != nil {
t.Fatalf("marshal raw: %v", err)
}
writer := &recordingProjectorWriter{}
registry := metrics.NewRegistry()
route := jt808ProjectionRoute()
count, err := projectBatch(context.Background(), discardProjectorLogger{}, registry, writer, route, []kafka.Message{{
Topic: route.RawTopic, Key: raw.KafkaKey(), Value: payload, Partition: 2, Offset: 10, HighWaterMark: 11,
}})
if err != nil {
t.Fatalf("projectBatch() error = %v", err)
}
if count != 1 || writer.callCount() != 1 || len(writer.messages) != 1 {
t.Fatalf("projected=%d calls=%d messages=%d", count, writer.callCount(), len(writer.messages))
}
var fields envelope.FrameEnvelope
if err := json.Unmarshal(writer.messages[0].Value, &fields); err != nil {
t.Fatalf("decode fields: %v", err)
}
if fields.EventKind != envelope.EventKindFields || fields.SourceEventID != raw.EventID || fields.EventID != raw.EventID+":fields" {
t.Fatalf("fields identity = %#v", fields)
}
if fields.SourceCode != raw.SourceCode || fields.SourceKind != raw.SourceKind || fields.SourceEndpoint != raw.SourceEndpoint {
t.Fatalf("source metadata not preserved: %#v", fields)
}
if got := fields.Fields["jt808.location.total_mileage_km"]; got != 1234.5 {
t.Fatalf("total mileage = %#v", got)
}
if len(fields.Parsed) != 0 || len(fields.ParsedFields) != 0 {
t.Fatalf("fields projection must not duplicate raw payload: parsed=%v parsed_fields=%v", fields.Parsed, fields.ParsedFields)
}
text := registry.Render()
for _, want := range []string{
`vehicle_fields_projector_projections_total{protocol="JT808",status="projected"} 1`,
`vehicle_fields_projector_kafka_writes_total{status="ok",topic="vehicle.fields.go.jt808.v1"} 1`,
`vehicle_fields_projector_kafka_lag{partition="2",topic="vehicle.raw.go.jt808.v1"} 0`,
} {
if !strings.Contains(text, want) {
t.Fatalf("metric missing %q:\n%s", want, text)
}
}
}
func TestProcessBatchReliablySkipsNonRealtimeAndInvalidWithoutWriting(t *testing.T) {
route := jt808ProjectionRoute()
nonRealtime := projectorRawEnvelope()
nonRealtime.MessageID = "0x0002"
nonRealtime.ParsedFields = map[string]any{"jt808.header.message_id": "0x0002"}
payload, _ := nonRealtime.MarshalJSONBytes()
messages := []kafka.Message{
{Topic: route.RawTopic, Value: []byte("{bad"), Partition: 0, Offset: 1, HighWaterMark: 3},
{Topic: route.RawTopic, Value: payload, Partition: 0, Offset: 2, HighWaterMark: 3},
}
writer := &recordingProjectorWriter{}
committer := &recordingProjectorCommitter{}
registry := metrics.NewRegistry()
processBatchReliably(context.Background(), discardProjectorLogger{}, registry, writer, committer, route, messages, time.Second, time.Millisecond)
if writer.callCount() != 0 {
t.Fatalf("writer calls = %d, want 0", writer.callCount())
}
if committer.callCount() != 1 || committer.messageCount != 2 {
t.Fatalf("commit calls/messages = %d/%d", committer.callCount(), committer.messageCount)
}
text := registry.Render()
if !strings.Contains(text, `vehicle_fields_projector_kafka_messages_total{status="invalid_json",topic="vehicle.raw.go.jt808.v1"} 1`) ||
!strings.Contains(text, `vehicle_fields_projector_projections_total{protocol="JT808",status="skipped_non_realtime"} 1`) {
t.Fatalf("skip metrics missing:\n%s", text)
}
}
func TestProcessBatchReliablyRetriesWriteBeforeCommitting(t *testing.T) {
route := jt808ProjectionRoute()
payload, _ := projectorRawEnvelope().MarshalJSONBytes()
messages := []kafka.Message{{Topic: route.RawTopic, Value: payload, Partition: 1, Offset: 7, HighWaterMark: 8}}
writer := &recordingProjectorWriter{errors: []error{errors.New("kafka unavailable"), nil}}
committer := &recordingProjectorCommitter{}
registry := metrics.NewRegistry()
processBatchReliably(context.Background(), discardProjectorLogger{}, registry, writer, committer, route, messages, time.Second, time.Millisecond)
if writer.callCount() != 2 {
t.Fatalf("writer calls = %d, want 2", writer.callCount())
}
if committer.callCount() != 1 {
t.Fatalf("commit calls = %d, want 1 after successful write", committer.callCount())
}
if !strings.Contains(registry.Render(), `vehicle_fields_projector_batch_retries_total{protocol="JT808",reason="write_error"} 1`) {
t.Fatalf("write retry metric missing:\n%s", registry.Render())
}
}
func TestProcessBatchReliablyRetriesOnlyCommitAfterSuccessfulWrite(t *testing.T) {
route := jt808ProjectionRoute()
payload, _ := projectorRawEnvelope().MarshalJSONBytes()
messages := []kafka.Message{{Topic: route.RawTopic, Value: payload, Partition: 1, Offset: 7, HighWaterMark: 8}}
writer := &recordingProjectorWriter{}
committer := &recordingProjectorCommitter{errors: []error{errors.New("commit timeout"), nil}}
processBatchReliably(context.Background(), discardProjectorLogger{}, metrics.NewRegistry(), writer, committer, route, messages, time.Second, time.Millisecond)
if writer.callCount() != 1 {
t.Fatalf("writer calls = %d, want 1", writer.callCount())
}
if committer.callCount() != 2 {
t.Fatalf("commit calls = %d, want 2", committer.callCount())
}
}
func TestProjectBatchWriteFailureDoesNotCommitByItself(t *testing.T) {
route := jt808ProjectionRoute()
payload, _ := projectorRawEnvelope().MarshalJSONBytes()
writer := &recordingProjectorWriter{errors: []error{errors.New("write failed")}}
count, err := projectBatch(context.Background(), discardProjectorLogger{}, nil, writer, route, []kafka.Message{{Topic: route.RawTopic, Value: payload}})
if err == nil || count != 1 {
t.Fatalf("projectBatch() count/error = %d/%v", count, err)
}
}
func projectorRawEnvelope() envelope.FrameEnvelope {
return envelope.FrameEnvelope{
EventID: "raw-event-1",
EventKind: envelope.EventKindRaw,
Protocol: envelope.ProtocolJT808,
MessageID: "0x0200",
VIN: "VIN001",
Phone: "13307795425",
SourceEndpoint: "115.231.168.135:43625",
SourceCode: "g7s",
SourceKind: "PLATFORM",
PlatformName: "G7s",
EventTimeMS: 1783960000000,
ReceivedAtMS: 1783960000010,
ParsedFields: map[string]any{
"jt808.location.latitude": 30.1,
"jt808.location.longitude": 121.2,
"jt808.location.total_mileage_km": 1234.5,
},
ParseStatus: envelope.ParseOK,
}
}
func jt808ProjectionRoute() projectionRoute {
return projectionRoute{
Protocol: envelope.ProtocolJT808, RawTopic: topics.RawJT808, FieldsTopic: topics.FieldsJT808, GroupID: "projector-jt808",
}
}
type recordingProjectorWriter struct {
mu sync.Mutex
errors []error
calls int
messages []kafka.Message
}
func (w *recordingProjectorWriter) WriteMessages(_ context.Context, messages ...kafka.Message) error {
w.mu.Lock()
defer w.mu.Unlock()
w.calls++
w.messages = append(w.messages, messages...)
if len(w.errors) == 0 {
return nil
}
err := w.errors[0]
w.errors = w.errors[1:]
return err
}
func (w *recordingProjectorWriter) callCount() int {
w.mu.Lock()
defer w.mu.Unlock()
return w.calls
}
type recordingProjectorCommitter struct {
mu sync.Mutex
errors []error
calls int
messageCount int
}
func (c *recordingProjectorCommitter) CommitMessages(_ context.Context, messages ...kafka.Message) error {
c.mu.Lock()
defer c.mu.Unlock()
c.calls++
c.messageCount += len(messages)
if len(c.errors) == 0 {
return nil
}
err := c.errors[0]
c.errors = c.errors[1:]
return err
}
func (c *recordingProjectorCommitter) callCount() int {
c.mu.Lock()
defer c.mu.Unlock()
return c.calls
}
type discardProjectorLogger struct{}
func (discardProjectorLogger) Error(string, ...any) {}
func (discardProjectorLogger) Warn(string, ...any) {}

View File

@@ -3,6 +3,8 @@ package main
import (
"context"
"database/sql"
"encoding/json"
"fmt"
"log/slog"
"os"
"os/signal"
@@ -13,13 +15,17 @@ import (
_ "github.com/go-sql-driver/mysql"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/authentication"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/eventbus"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/gateway"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/health"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/identity"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/observability"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/protocol/gb32960"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/protocol/jt808"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/topics"
)
func main() {
@@ -27,33 +33,53 @@ func main() {
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
sink, err := buildSink(ctx, logger)
registry := metrics.NewRegistry()
sink, err := buildSink(ctx, logger, registry)
if err != nil {
logger.Error("build sink failed", "error", err)
os.Exit(1)
}
defer sink.Close()
resolver, closeResolver, err := buildIdentityResolver(ctx, logger)
resolver, jt808DeviceTokens, closeResolver, err := buildIdentityResolver(ctx, logger, registry)
if err != nil {
logger.Error("build identity resolver failed", "error", err)
os.Exit(1)
}
defer closeResolver()
health.Start(ctx, logger, health.NewServer(env("HEALTH_ADDR", ""), "vehicle-gateway", nil, registry))
publishUnified := envBool("PUBLISH_UNIFIED_ENABLED", false)
delegateFields := envBool("FIELDS_DERIVE_FROM_RAW_ENABLED", strings.TrimSpace(os.Getenv("NATS_URL")) != "")
gb32960Authenticator, gb32960AuthMode, gb32960CredentialCount, err := buildGB32960Authenticator()
if err != nil {
logger.Error("build gb32960 authenticator failed", "error", err)
os.Exit(1)
}
jt808AuthCode := env("JT808_REGISTER_AUTH_CODE", "g7gps")
jt808Authenticator, jt808AuthMode, err := buildJT808Authenticator(jt808AuthCode, jt808DeviceTokens)
if err != nil {
logger.Error("build jt808 authenticator failed", "error", err)
os.Exit(1)
}
recordAuthenticationConfig(registry, envelope.ProtocolGB32960, gb32960AuthMode, gb32960CredentialCount)
recordAuthenticationConfig(registry, envelope.ProtocolJT808, jt808AuthMode, int(boolMetric(strings.TrimSpace(jt808AuthCode) != "")))
logger.Info("protocol authentication configured", "gb32960_mode", gb32960AuthMode, "gb32960_accounts", gb32960CredentialCount, "jt808_mode", jt808AuthMode)
protocols := []gateway.TCPProtocol{
{
Protocol: envelope.ProtocolGB32960,
Addr: env("GB32960_TCP_ADDR", ":32960"),
Extract: gb32960.ExtractFrames,
Parse: gb32960.ParseFrame,
Respond: gb32960.AutoResponse,
Protocol: envelope.ProtocolGB32960,
Addr: env("GB32960_TCP_ADDR", ":32960"),
Extract: gb32960.ExtractFrames,
Parse: gb32960.ParseFrame,
Authenticate: gb32960Authenticator,
Respond: gb32960.AutoResponse,
},
{
Protocol: envelope.ProtocolJT808,
Addr: env("JT808_TCP_ADDR", ":808"),
Extract: jt808.ExtractFrames,
Parse: jt808.ParseFrame,
Respond: jt808.NewAutoResponder(env("JT808_REGISTER_AUTH_CODE", "g7gps")).Respond,
Protocol: envelope.ProtocolJT808,
Addr: env("JT808_TCP_ADDR", ":808"),
Extract: jt808.ExtractFrames,
Parse: jt808.ParseFrame,
Authenticate: jt808Authenticator,
Respond: jt808.NewAutoResponder(jt808AuthCode).Respond,
},
}
@@ -65,9 +91,12 @@ func main() {
Sink: sink,
Resolver: resolver,
Logger: logger,
Metrics: registry,
ReadBufferSize: envInt("TCP_READ_BUFFER_BYTES", 64*1024),
IdleTimeout: time.Duration(envInt("TCP_IDLE_TIMEOUT_SECONDS", 180)) * time.Second,
MaxConnections: envInt("TCP_MAX_CONNECTIONS", 20_000),
MaxConnections: envInt("TCP_MAX_CONNECTIONS", 120_000),
PublishUnified: publishUnified,
DelegateFields: delegateFields,
})
if err != nil {
logger.Error("build tcp server failed", "protocol", protocol.Protocol, "error", err)
@@ -100,6 +129,9 @@ func main() {
Sink: sink,
Resolver: resolver,
Logger: logger,
Metrics: registry,
PublishUnified: publishUnified,
DelegateFields: delegateFields,
})
if err != nil {
logger.Error("build yutong mqtt client failed", "error", err)
@@ -112,7 +144,7 @@ func main() {
logger.Info("yutong mqtt client started")
}
logger.Info("vehicle gateway started")
logger.Info("vehicle gateway started", "fields_derive_from_raw_enabled", delegateFields)
select {
case <-ctx.Done():
case err := <-errs:
@@ -122,26 +154,435 @@ func main() {
wg.Wait()
}
func buildIdentityResolver(ctx context.Context, logger *slog.Logger) (identity.Resolver, func(), error) {
func buildGB32960Authenticator() (authentication.Authenticator, authentication.Mode, int, error) {
mode, err := authentication.ParseMode(os.Getenv("GB32960_AUTH_MODE"), authentication.ModeObserve)
if err != nil {
return nil, "", 0, err
}
credentials, err := loadGB32960Credentials()
if err != nil {
return nil, "", 0, err
}
if mode == authentication.ModeEnforce && len(credentials) == 0 {
return nil, "", 0, fmt.Errorf("GB32960_AUTH_MODE=enforce requires configured platform credentials")
}
return authentication.NewGB32960PlatformAuthenticator(mode, credentials), mode, len(credentials), nil
}
func buildJT808Authenticator(authCode string, deviceTokens authentication.JT808DeviceTokenProvider) (authentication.Authenticator, authentication.Mode, error) {
mode, err := authentication.ParseMode(os.Getenv("JT808_AUTH_MODE"), authentication.ModeObserve)
if err != nil {
return nil, "", err
}
if mode == authentication.ModeEnforce && strings.TrimSpace(authCode) == "" && deviceTokens == nil {
return nil, "", fmt.Errorf("JT808_AUTH_MODE=enforce requires a configured or device token provider")
}
return authentication.NewJT808Authenticator(mode, authCode, deviceTokens), mode, nil
}
func loadGB32960Credentials() (map[string][]string, error) {
payload := strings.TrimSpace(os.Getenv("GB32960_PLATFORM_CREDENTIALS_JSON"))
if path := strings.TrimSpace(os.Getenv("GB32960_PLATFORM_CREDENTIALS_FILE")); path != "" {
contents, err := os.ReadFile(path)
if err != nil {
return nil, fmt.Errorf("read gb32960 credentials file: %w", err)
}
payload = strings.TrimSpace(string(contents))
}
if payload == "" {
return map[string][]string{}, nil
}
rawCredentials := map[string]json.RawMessage{}
if err := json.Unmarshal([]byte(payload), &rawCredentials); err != nil {
return nil, fmt.Errorf("parse gb32960 platform credentials: %w", err)
}
credentials := make(map[string][]string, len(rawCredentials))
for username, rawPassword := range rawCredentials {
trimmed := strings.TrimSpace(username)
if trimmed == "" {
continue
}
var passwords []string
var single string
if err := json.Unmarshal(rawPassword, &single); err == nil {
passwords = []string{single}
} else if err := json.Unmarshal(rawPassword, &passwords); err != nil {
return nil, fmt.Errorf("parse gb32960 credentials for account %q: expected string or string array", trimmed)
}
for _, password := range passwords {
if password != "" {
credentials[trimmed] = append(credentials[trimmed], password)
}
}
if len(credentials[trimmed]) == 0 {
delete(credentials, trimmed)
}
}
return credentials, nil
}
func recordAuthenticationConfig(registry *metrics.Registry, protocol envelope.Protocol, mode authentication.Mode, credentialCount int) {
if registry == nil {
return
}
registry.SetGauge("vehicle_gateway_authentication_mode", metrics.Labels{
"protocol": string(protocol),
"mode": string(mode),
}, 1)
registry.SetGauge("vehicle_gateway_authentication_credentials", metrics.Labels{
"protocol": string(protocol),
}, float64(credentialCount))
}
func buildIdentityResolver(ctx context.Context, logger *slog.Logger, registry *metrics.Registry) (identity.Resolver, authentication.JT808DeviceTokenProvider, func(), error) {
dsn := env("IDENTITY_MYSQL_DSN", strings.TrimSpace(os.Getenv("MYSQL_DSN")))
if strings.TrimSpace(dsn) == "" {
logger.Warn("identity mysql dsn is empty; using noop identity resolver")
return identity.NoopResolver{}, func() {}, nil
return identity.NoopResolver{}, nil, func() {}, nil
}
db, err := sql.Open("mysql", dsn)
if err != nil {
return nil, nil, err
}
if err := db.PingContext(ctx); err != nil {
_ = db.Close()
return nil, nil, err
logger.Error("identity mysql configuration failed; continuing with unresolved identities", "error", err)
return identity.NoopResolver{}, nil, func() {}, nil
}
db.SetMaxOpenConns(envInt("IDENTITY_MYSQL_MAX_OPEN_CONNS", 16))
db.SetMaxIdleConns(envInt("IDENTITY_MYSQL_MAX_IDLE_CONNS", 8))
db.SetConnMaxLifetime(time.Duration(envInt("IDENTITY_MYSQL_CONN_MAX_LIFETIME_SECONDS", 300)) * time.Second)
db.SetConnMaxIdleTime(time.Duration(envInt("IDENTITY_MYSQL_CONN_MAX_IDLE_SECONDS", 60)) * time.Second)
table := env("VEHICLE_IDENTITY_TABLE", "vehicle_identity_binding")
logger.Info("identity mysql resolver enabled", "table", table)
return identity.NewMySQLResolver(db, table), func() { _ = db.Close() }, nil
cacheMaxEntries := envInt("IDENTITY_LOOKUP_CACHE_MAX_ENTRIES", 300000)
cacheCleanupInterval := time.Duration(envInt("IDENTITY_LOOKUP_CACHE_CLEANUP_INTERVAL_SECONDS", 60)) * time.Second
staleLookupTTLSeconds := envInt("IDENTITY_STALE_LOOKUP_TTL_SECONDS", 3600)
registrationGatewayWritesEnabled := envBool("JT808_REGISTRATION_GATEWAY_WRITES_ENABLED", false)
resolver := identity.NewMySQLResolverWithOptions(db, table, identity.MySQLResolverOptions{
SnapshotOnlyLookups: envBool("IDENTITY_SNAPSHOT_ONLY_ENABLED", true),
LocationTouchInterval: time.Duration(envInt("JT808_REGISTRATION_LOCATION_TOUCH_INTERVAL_SECONDS", 600)) * time.Second,
LocationTouchRetryInterval: time.Duration(envInt("JT808_REGISTRATION_LOCATION_TOUCH_RETRY_INTERVAL_SECONDS", 5)) * time.Second,
RegistrationWriteAttempts: envInt("JT808_REGISTRATION_WRITE_RETRY_ATTEMPTS", 2),
RegistrationWriteRetryDelay: time.Duration(envInt("JT808_REGISTRATION_WRITE_RETRY_DELAY_MS", 20)) * time.Millisecond,
LookupCacheTTL: time.Duration(envInt("IDENTITY_LOOKUP_CACHE_TTL_SECONDS", 600)) * time.Second,
StaleLookupTTL: time.Duration(staleLookupTTLSeconds) * time.Second,
CacheCleanupInterval: cacheCleanupInterval,
MaxCacheEntries: cacheMaxEntries,
SourceCodeLookup: envBool("IDENTITY_SOURCE_CODE_LOOKUP_ENABLED", true),
AsyncRegistrationWrites: envBool("JT808_REGISTRATION_ASYNC_WRITE_ENABLED", true),
RegistrationWriteQueueSize: envInt("JT808_REGISTRATION_WRITE_QUEUE_SIZE", 100000),
RegistrationWriteWorkers: envInt("JT808_REGISTRATION_WRITE_WORKERS", 4),
RegistrationWriteTimeout: time.Duration(envInt("JT808_REGISTRATION_WRITE_TIMEOUT_MS", 5000)) * time.Millisecond,
RegistrationEnqueueTimeout: time.Duration(envInt("JT808_REGISTRATION_WRITE_ENQUEUE_TIMEOUT_MS", 50)) * time.Millisecond,
DisableRegistrationWrites: !registrationGatewayWritesEnabled,
OnRegistrationWriteResult: func(result identity.RegistrationWriteResult) {
recordJT808RegistrationWriteResult(registry, result)
},
OnRegistrationWriteError: func(err error) {
logger.Warn("jt808 registration async write failed", "error", err)
},
})
startIdentityDatabaseMaintenance(
ctx,
logger,
db,
resolver,
envBool("IDENTITY_MYSQL_ENSURE_SCHEMA", false),
time.Duration(envInt("IDENTITY_MYSQL_PING_TIMEOUT_MS", 3000))*time.Millisecond,
time.Duration(envInt("IDENTITY_MYSQL_SCHEMA_TIMEOUT_SECONDS", 10))*time.Second,
)
startIdentitySnapshotRefresh(
ctx,
logger,
registry,
resolver,
time.Duration(envInt("IDENTITY_SNAPSHOT_REFRESH_INTERVAL_SECONDS", 60))*time.Second,
time.Duration(envInt("IDENTITY_SNAPSHOT_REFRESH_TIMEOUT_SECONDS", 10))*time.Second,
)
startIdentityCacheMetrics(ctx, registry, resolver, time.Duration(envInt("IDENTITY_CACHE_METRICS_INTERVAL_SECONDS", 30))*time.Second)
resolveTimeout := time.Duration(envInt("IDENTITY_RESOLVE_TIMEOUT_MS", 50)) * time.Millisecond
registry.SetGauge("vehicle_gateway_jt808_registration_gateway_writes_enabled", nil, boolMetric(registrationGatewayWritesEnabled))
logger.Info("identity mysql resolver enabled", "table", table, "snapshot_only_enabled", envBool("IDENTITY_SNAPSHOT_ONLY_ENABLED", true), "snapshot_refresh_interval_seconds", envInt("IDENTITY_SNAPSHOT_REFRESH_INTERVAL_SECONDS", 60), "lookup_cache_ttl_seconds", envInt("IDENTITY_LOOKUP_CACHE_TTL_SECONDS", 600), "stale_lookup_ttl_seconds", staleLookupTTLSeconds, "lookup_cache_max_entries", cacheMaxEntries, "lookup_cache_cleanup_interval_seconds", cacheCleanupInterval.Seconds(), "source_code_lookup_enabled", envBool("IDENTITY_SOURCE_CODE_LOOKUP_ENABLED", true), "resolve_timeout_ms", resolveTimeout.Milliseconds(), "jt808_registration_gateway_writes_enabled", registrationGatewayWritesEnabled, "jt808_registration_location_touch_retry_interval_seconds", envInt("JT808_REGISTRATION_LOCATION_TOUCH_RETRY_INTERVAL_SECONDS", 5), "jt808_registration_write_retry_attempts", envInt("JT808_REGISTRATION_WRITE_RETRY_ATTEMPTS", 2), "jt808_registration_write_retry_delay_ms", envInt("JT808_REGISTRATION_WRITE_RETRY_DELAY_MS", 20), "jt808_registration_async_write_enabled", envBool("JT808_REGISTRATION_ASYNC_WRITE_ENABLED", true), "jt808_registration_write_queue_size", envInt("JT808_REGISTRATION_WRITE_QUEUE_SIZE", 100000), "jt808_registration_write_workers", envInt("JT808_REGISTRATION_WRITE_WORKERS", 4), "jt808_registration_write_timeout_ms", envInt("JT808_REGISTRATION_WRITE_TIMEOUT_MS", 5000), "jt808_registration_write_enqueue_timeout_ms", envInt("JT808_REGISTRATION_WRITE_ENQUEUE_TIMEOUT_MS", 50))
return identity.TimeoutResolver{Delegate: resolver, Timeout: resolveTimeout}, resolver, func() {
_ = resolver.Close()
_ = db.Close()
}, nil
}
func buildSink(ctx context.Context, logger *slog.Logger) (eventbus.Sink, error) {
type identitySnapshotRefresher interface {
RefreshSnapshot(context.Context) (identity.SnapshotRefreshResult, error)
}
type identityDatabasePinger interface {
PingContext(context.Context) error
}
type identitySchemaEnsurer interface {
EnsureSchema(context.Context) error
}
func startIdentityDatabaseMaintenance(ctx context.Context, logger *slog.Logger, db identityDatabasePinger, schema identitySchemaEnsurer, ensureSchema bool, pingTimeout time.Duration, schemaTimeout time.Duration) {
if db == nil {
return
}
if pingTimeout <= 0 {
pingTimeout = 3 * time.Second
}
if schemaTimeout <= 0 {
schemaTimeout = 10 * time.Second
}
go func() {
pingCtx, cancelPing := context.WithTimeout(ctx, pingTimeout)
err := db.PingContext(pingCtx)
cancelPing()
if err != nil {
logger.Warn("identity mysql unavailable; ingress remains enabled and snapshot refresh will retry", "error", err)
return
}
if !ensureSchema || schema == nil {
return
}
schemaCtx, cancelSchema := context.WithTimeout(ctx, schemaTimeout)
err = schema.EnsureSchema(schemaCtx)
cancelSchema()
if err != nil {
logger.Warn("identity mysql schema check failed; ingress remains enabled", "error", err)
}
}()
}
func startIdentitySnapshotRefresh(ctx context.Context, logger *slog.Logger, registry *metrics.Registry, refresher identitySnapshotRefresher, interval time.Duration, timeout time.Duration) {
if refresher == nil {
return
}
if interval <= 0 {
interval = time.Minute
}
if timeout <= 0 {
timeout = 10 * time.Second
}
refresh := func() {
refreshCtx, cancel := context.WithTimeout(ctx, timeout)
result, err := refresher.RefreshSnapshot(refreshCtx)
cancel()
if err != nil {
recordIdentitySnapshotRefresh(registry, identity.SnapshotRefreshResult{}, "error")
logger.Warn("identity snapshot refresh failed; keeping last known good snapshot", "error", err)
return
}
recordIdentitySnapshotRefresh(registry, result, "ok")
logger.Info("identity snapshot refreshed", "binding_entries", result.BindingEntries, "identifier_entries", result.IdentifierEntries, "registration_entries", result.RegistrationEntries, "source_entries", result.SourceEntries)
}
go func() {
refresh()
ticker := time.NewTicker(interval)
defer ticker.Stop()
for {
select {
case <-ctx.Done():
return
case <-ticker.C:
refresh()
}
}
}()
}
func recordIdentitySnapshotRefresh(registry *metrics.Registry, result identity.SnapshotRefreshResult, status string) {
if registry == nil {
return
}
registry.IncCounter("vehicle_gateway_identity_snapshot_refresh_total", metrics.Labels{"status": status})
if status != "ok" {
return
}
for _, item := range []struct {
kind string
value int
}{
{kind: "binding", value: result.BindingEntries},
{kind: "identifier", value: result.IdentifierEntries},
{kind: "registration", value: result.RegistrationEntries},
{kind: "source", value: result.SourceEntries},
} {
registry.SetGauge("vehicle_gateway_identity_snapshot_entries", metrics.Labels{"kind": item.kind}, float64(item.value))
}
registry.SetGauge("vehicle_gateway_identity_snapshot_last_success_unix_seconds", nil, float64(result.RefreshedAt.Unix()))
}
type identityCacheStatsReporter interface {
CacheStats() identity.CacheStats
}
func startIdentityCacheMetrics(ctx context.Context, registry *metrics.Registry, reporter identityCacheStatsReporter, interval time.Duration) {
if registry == nil || reporter == nil {
return
}
if interval <= 0 {
interval = 30 * time.Second
}
recordIdentityCacheStats(registry, reporter.CacheStats())
go func() {
ticker := time.NewTicker(interval)
defer ticker.Stop()
for {
select {
case <-ctx.Done():
return
case <-ticker.C:
recordIdentityCacheStats(registry, reporter.CacheStats())
}
}
}()
}
func recordIdentityCacheStats(registry *metrics.Registry, stats identity.CacheStats) {
if registry == nil {
return
}
for _, item := range []struct {
name string
value int
max int
}{
{name: "lookup", value: stats.LookupEntries, max: stats.MaxEntries},
{name: "registration", value: stats.RegistrationEntries, max: stats.MaxEntries},
{name: "source_code", value: stats.SourceCodeEntries, max: stats.MaxEntries},
{name: "location_touch", value: stats.LocationTouchEntries, max: stats.MaxEntries},
{name: "location_touch_failure", value: stats.LocationTouchFailureEntries, max: stats.MaxEntries},
{name: "registration_write_queue", value: stats.RegistrationWriteQueueDepth, max: stats.RegistrationWriteQueueCap},
{name: "snapshot_binding", value: stats.SnapshotBindingEntries, max: stats.MaxEntries},
{name: "snapshot_identifier", value: stats.SnapshotIdentifierEntries, max: stats.MaxEntries},
{name: "snapshot_registration", value: stats.SnapshotRegistrationEntries, max: stats.MaxEntries},
{name: "snapshot_source", value: stats.SnapshotSourceEntries, max: stats.MaxEntries},
} {
registry.SetGauge("vehicle_gateway_identity_cache_entries", metrics.Labels{"cache": item.name}, float64(item.value))
registry.SetGauge("vehicle_gateway_identity_cache_max_entries", metrics.Labels{"cache": item.name}, float64(item.max))
}
ready := 0.0
if stats.SnapshotReady {
ready = 1
}
registry.SetGauge("vehicle_gateway_identity_snapshot_ready", nil, ready)
if !stats.SnapshotRefreshedAt.IsZero() {
registry.SetGauge("vehicle_gateway_identity_snapshot_last_success_unix_seconds", nil, float64(stats.SnapshotRefreshedAt.Unix()))
}
}
func recordJT808RegistrationWriteResult(registry *metrics.Registry, result identity.RegistrationWriteResult) {
if registry == nil {
return
}
mode := strings.TrimSpace(result.Mode)
if mode == "" {
mode = "unknown"
}
status := strings.TrimSpace(result.Status)
if status == "" {
status = "unknown"
}
registry.IncCounter("vehicle_gateway_jt808_registration_write_total", metrics.Labels{
"mode": mode,
"status": status,
})
metrics.RecordLastActivity(registry, "vehicle_gateway_last_jt808_registration_write_unix_seconds", metrics.Labels{
"mode": mode,
"status": status,
})
}
func buildSink(ctx context.Context, logger *slog.Logger, registry *metrics.Registry) (eventbus.Sink, error) {
if strings.TrimSpace(os.Getenv("NATS_URL")) != "" {
natsConfig := natsSinkConfigFromEnv()
sink, err := eventbus.NewNATSSink(natsConfig)
if err != nil {
return nil, err
}
outboxRuntime := natsOutboxConfigFromEnv(registry, func(err error) {
logger.Warn("nats durable outbox publish failed", "error", err)
})
if outboxRuntime.Enabled {
outbox, err := eventbus.NewDurableOutboxSink(sink, outboxRuntime.Config)
if err != nil {
_ = sink.Close()
return nil, err
}
go outbox.ReplayLoop(ctx, outboxRuntime.ReplayInterval)
logger.Info("nats durable outbox enabled",
"dir", outboxRuntime.Config.Directory,
"fsync", outboxRuntime.Config.SyncWrites,
"replay_interval_ms", outboxRuntime.ReplayInterval.Milliseconds(),
"replay_batch_size", outboxRuntime.Config.ReplayBatchSize,
"close_timeout_ms", outboxRuntime.Config.CloseTimeout.Milliseconds(),
"wal_segment_bytes", outboxRuntime.Config.WALSegmentBytes,
"wal_segment_age_ms", outboxRuntime.Config.WALSegmentAge.Milliseconds(),
"wal_append_queue_size", outboxRuntime.Config.WALAppendQueue,
"wal_commit_batch_size", outboxRuntime.Config.WALCommitBatch,
"wal_commit_interval_ms", outboxRuntime.Config.WALCommitWait.Milliseconds(),
"max_inflight", natsConfig.AsyncMaxPending,
"ack_timeout_ms", natsConfig.AsyncAckTimeout.Milliseconds(),
)
return outbox, nil
}
var out eventbus.Sink = eventbus.NewRetryingSink(sink, eventbus.RetryConfig{
Attempts: envInt("NATS_PUBLISH_ATTEMPTS", 3),
Backoff: time.Duration(envInt("NATS_PUBLISH_BACKOFF_MS", 100)) * time.Millisecond,
AttemptTimeout: time.Duration(envInt("NATS_PUBLISH_TIMEOUT_MS", 3000)) * time.Millisecond,
})
spoolDir := strings.TrimSpace(os.Getenv("NATS_SPOOL_DIR"))
if spoolDir != "" {
replayBatchSize := envInt("NATS_SPOOL_REPLAY_BATCH_SIZE", 200)
durable := eventbus.NewDurableSink(out, eventbus.DurableConfig{
Directory: spoolDir,
ReplayBatchSize: replayBatchSize,
Metrics: registry,
Name: "nats",
})
interval := time.Duration(envInt("NATS_SPOOL_REPLAY_INTERVAL_MS", 1000)) * time.Millisecond
go durable.ReplayLoop(ctx, interval, func(err error) {
logger.Warn("nats spool replay failed", "error", err)
})
logger.Info("nats durable spool enabled", "dir", spoolDir, "replay_interval_ms", interval.Milliseconds(), "replay_batch_size", replayBatchSize)
out = durable
}
if envBool("NATS_ASYNC_ENABLED", true) {
queueSize := envInt("NATS_ASYNC_QUEUE_SIZE", envInt("KAFKA_ASYNC_QUEUE_SIZE", 100000))
workers := envInt("NATS_ASYNC_WORKERS", envInt("KAFKA_ASYNC_WORKERS", 8))
enqueueTimeout := time.Duration(envInt("NATS_ASYNC_ENQUEUE_TIMEOUT_MS", envInt("KAFKA_ASYNC_ENQUEUE_TIMEOUT_MS", 1000))) * time.Millisecond
timeout := time.Duration(envInt("NATS_ASYNC_PUBLISH_TIMEOUT_MS", envInt("KAFKA_ASYNC_PUBLISH_TIMEOUT_MS", 30000))) * time.Millisecond
if envBool("NATS_PARTITIONED_ASYNC_ENABLED", true) {
rawQueueSize, derivedQueueSize, rawWorkers, derivedWorkers := partitionedAsyncConfigFromEnv("NATS", queueSize, workers)
rawEnqueueTimeout, derivedEnqueueTimeout := partitionedAsyncEnqueueTimeoutsFromEnv("NATS", enqueueTimeout)
out = eventbus.NewPartitionedAsyncSink(out, eventbus.PartitionedAsyncConfig{
RawQueueSize: rawQueueSize,
DerivedQueueSize: derivedQueueSize,
RawWorkers: rawWorkers,
DerivedWorkers: derivedWorkers,
RawEnqueueTimeout: rawEnqueueTimeout,
DerivedEnqueueTimeout: derivedEnqueueTimeout,
EnqueueTimeout: enqueueTimeout,
OperationTimeout: timeout,
Metrics: registry,
Name: "nats",
OnError: func(err error) {
logger.Warn("nats partitioned async publish failed", "error", err)
},
})
logger.Info("nats partitioned async publish enabled", "raw_queue_size", rawQueueSize, "derived_queue_size", derivedQueueSize, "raw_workers", rawWorkers, "derived_workers", derivedWorkers, "raw_enqueue_timeout_ms", rawEnqueueTimeout.Milliseconds(), "derived_enqueue_timeout_ms", derivedEnqueueTimeout.Milliseconds(), "publish_timeout_ms", timeout.Milliseconds())
} else {
out = eventbus.NewAsyncSink(out, eventbus.AsyncConfig{
QueueSize: queueSize,
Workers: workers,
EnqueueTimeout: enqueueTimeout,
OperationTimeout: timeout,
Metrics: registry,
Name: "nats",
OnError: func(err error) {
logger.Warn("nats async publish failed", "error", err)
},
})
logger.Info("nats async publish enabled", "queue_size", queueSize, "workers", workers, "enqueue_timeout_ms", enqueueTimeout.Milliseconds(), "publish_timeout_ms", timeout.Milliseconds())
}
}
logger.Info("nats jetstream sink enabled", "url", env("NATS_URL", ""), "unified_subject", natsSinkConfigFromEnv().UnifiedSubject, "publish_unified_enabled", envBool("PUBLISH_UNIFIED_ENABLED", false), "publish_fields_enabled", !envBool("FIELDS_DERIVE_FROM_RAW_ENABLED", true))
return out, nil
}
brokers := splitCSV(os.Getenv("KAFKA_BROKERS"))
if len(brokers) == 0 {
logger.Warn("KAFKA_BROKERS is empty; using log sink")
@@ -150,11 +591,16 @@ func buildSink(ctx context.Context, logger *slog.Logger) (eventbus.Sink, error)
sink, err := eventbus.NewKafkaSink(eventbus.KafkaConfig{
Brokers: brokers,
RawTopics: map[envelope.Protocol]string{
envelope.ProtocolGB32960: env("KAFKA_TOPIC_GB32960_RAW", "vehicle.raw.gb32960.v1"),
envelope.ProtocolJT808: env("KAFKA_TOPIC_JT808_RAW", "vehicle.raw.jt808.v1"),
envelope.ProtocolYutongMQTT: env("KAFKA_TOPIC_YUTONG_MQTT_RAW", "vehicle.raw.yutong-mqtt.v1"),
envelope.ProtocolGB32960: env("KAFKA_TOPIC_GB32960_RAW", topics.RawGB32960),
envelope.ProtocolJT808: env("KAFKA_TOPIC_JT808_RAW", topics.RawJT808),
envelope.ProtocolYutongMQTT: env("KAFKA_TOPIC_YUTONG_MQTT_RAW", topics.RawYutongMQTT),
},
UnifiedTopic: env("KAFKA_TOPIC_UNIFIED", "vehicle.event.unified.v1"),
FieldsTopics: map[envelope.Protocol]string{
envelope.ProtocolGB32960: env("KAFKA_TOPIC_GB32960_FIELDS", topics.FieldsGB32960),
envelope.ProtocolJT808: env("KAFKA_TOPIC_JT808_FIELDS", topics.FieldsJT808),
envelope.ProtocolYutongMQTT: env("KAFKA_TOPIC_YUTONG_MQTT_FIELDS", topics.FieldsYutongMQTT),
},
UnifiedTopic: env("KAFKA_TOPIC_UNIFIED", topics.Unified),
})
if err != nil {
return nil, err
@@ -166,17 +612,132 @@ func buildSink(ctx context.Context, logger *slog.Logger) (eventbus.Sink, error)
})
spoolDir := strings.TrimSpace(os.Getenv("KAFKA_SPOOL_DIR"))
if spoolDir != "" {
durable := eventbus.NewDurableSink(out, eventbus.DurableConfig{Directory: spoolDir})
replayBatchSize := envInt("KAFKA_SPOOL_REPLAY_BATCH_SIZE", 200)
durable := eventbus.NewDurableSink(out, eventbus.DurableConfig{
Directory: spoolDir,
ReplayBatchSize: replayBatchSize,
Metrics: registry,
Name: "kafka",
})
interval := time.Duration(envInt("KAFKA_SPOOL_REPLAY_INTERVAL_MS", 1000)) * time.Millisecond
go durable.ReplayLoop(ctx, interval, func(err error) {
logger.Warn("kafka spool replay failed", "error", err)
})
logger.Info("kafka durable spool enabled", "dir", spoolDir, "replay_interval_ms", interval.Milliseconds())
logger.Info("kafka durable spool enabled", "dir", spoolDir, "replay_interval_ms", interval.Milliseconds(), "replay_batch_size", replayBatchSize)
out = durable
}
if envBool("KAFKA_ASYNC_ENABLED", true) {
queueSize := envInt("KAFKA_ASYNC_QUEUE_SIZE", 100000)
workers := envInt("KAFKA_ASYNC_WORKERS", 8)
enqueueTimeout := time.Duration(envInt("KAFKA_ASYNC_ENQUEUE_TIMEOUT_MS", 1000)) * time.Millisecond
timeout := time.Duration(envInt("KAFKA_ASYNC_PUBLISH_TIMEOUT_MS", 30000)) * time.Millisecond
if envBool("KAFKA_PARTITIONED_ASYNC_ENABLED", true) {
rawQueueSize, derivedQueueSize, rawWorkers, derivedWorkers := partitionedAsyncConfigFromEnv("KAFKA", queueSize, workers)
rawEnqueueTimeout, derivedEnqueueTimeout := partitionedAsyncEnqueueTimeoutsFromEnv("KAFKA", enqueueTimeout)
out = eventbus.NewPartitionedAsyncSink(out, eventbus.PartitionedAsyncConfig{
RawQueueSize: rawQueueSize,
DerivedQueueSize: derivedQueueSize,
RawWorkers: rawWorkers,
DerivedWorkers: derivedWorkers,
RawEnqueueTimeout: rawEnqueueTimeout,
DerivedEnqueueTimeout: derivedEnqueueTimeout,
EnqueueTimeout: enqueueTimeout,
OperationTimeout: timeout,
Metrics: registry,
Name: "kafka",
OnError: func(err error) {
logger.Warn("kafka partitioned async publish failed", "error", err)
},
})
logger.Info("kafka partitioned async publish enabled", "raw_queue_size", rawQueueSize, "derived_queue_size", derivedQueueSize, "raw_workers", rawWorkers, "derived_workers", derivedWorkers, "raw_enqueue_timeout_ms", rawEnqueueTimeout.Milliseconds(), "derived_enqueue_timeout_ms", derivedEnqueueTimeout.Milliseconds(), "publish_timeout_ms", timeout.Milliseconds())
} else {
out = eventbus.NewAsyncSink(out, eventbus.AsyncConfig{
QueueSize: queueSize,
Workers: workers,
EnqueueTimeout: enqueueTimeout,
OperationTimeout: timeout,
Metrics: registry,
Name: "kafka",
OnError: func(err error) {
logger.Warn("kafka async publish failed", "error", err)
},
})
logger.Info("kafka async publish enabled", "queue_size", queueSize, "workers", workers, "enqueue_timeout_ms", enqueueTimeout.Milliseconds(), "publish_timeout_ms", timeout.Milliseconds())
}
}
return out, nil
}
func partitionedAsyncConfigFromEnv(prefix string, queueSize int, workers int) (rawQueueSize int, derivedQueueSize int, rawWorkers int, derivedWorkers int) {
derivedQueueDefault := queueSize / 2
if derivedQueueDefault <= 0 {
derivedQueueDefault = 1
}
derivedWorkersDefault := workers / 2
if derivedWorkersDefault <= 0 {
derivedWorkersDefault = 1
}
rawQueueSize = envInt(prefix+"_ASYNC_RAW_QUEUE_SIZE", queueSize)
derivedQueueSize = envInt(prefix+"_ASYNC_DERIVED_QUEUE_SIZE", derivedQueueDefault)
rawWorkers = envInt(prefix+"_ASYNC_RAW_WORKERS", workers)
derivedWorkers = envInt(prefix+"_ASYNC_DERIVED_WORKERS", derivedWorkersDefault)
return rawQueueSize, derivedQueueSize, rawWorkers, derivedWorkers
}
func partitionedAsyncEnqueueTimeoutsFromEnv(prefix string, enqueueTimeout time.Duration) (rawEnqueueTimeout time.Duration, derivedEnqueueTimeout time.Duration) {
rawEnqueueTimeout = time.Duration(envInt(prefix+"_ASYNC_RAW_ENQUEUE_TIMEOUT_MS", int(enqueueTimeout/time.Millisecond))) * time.Millisecond
derivedEnqueueTimeout = time.Duration(envInt(prefix+"_ASYNC_DERIVED_ENQUEUE_TIMEOUT_MS", 50)) * time.Millisecond
return rawEnqueueTimeout, derivedEnqueueTimeout
}
func natsSinkConfigFromEnv() eventbus.NATSConfig {
return eventbus.NATSConfig{
URL: env("NATS_URL", ""),
Name: env("NATS_CLIENT_NAME", "lingniu-vehicle-gateway"),
AsyncMaxPending: envInt("NATS_OUTBOX_MAX_INFLIGHT", 10_000),
AsyncAckTimeout: time.Duration(envInt("NATS_OUTBOX_ACK_TIMEOUT_MS", 3000)) * time.Millisecond,
RawSubjects: map[envelope.Protocol]string{
envelope.ProtocolGB32960: env("NATS_SUBJECT_GB32960_RAW", env("KAFKA_TOPIC_GB32960_RAW", topics.RawGB32960)),
envelope.ProtocolJT808: env("NATS_SUBJECT_JT808_RAW", env("KAFKA_TOPIC_JT808_RAW", topics.RawJT808)),
envelope.ProtocolYutongMQTT: env("NATS_SUBJECT_YUTONG_MQTT_RAW", env("KAFKA_TOPIC_YUTONG_MQTT_RAW", topics.RawYutongMQTT)),
},
FieldsSubjects: map[envelope.Protocol]string{
envelope.ProtocolGB32960: env("NATS_SUBJECT_GB32960_FIELDS", env("KAFKA_TOPIC_GB32960_FIELDS", topics.FieldsGB32960)),
envelope.ProtocolJT808: env("NATS_SUBJECT_JT808_FIELDS", env("KAFKA_TOPIC_JT808_FIELDS", topics.FieldsJT808)),
envelope.ProtocolYutongMQTT: env("NATS_SUBJECT_YUTONG_MQTT_FIELDS", env("KAFKA_TOPIC_YUTONG_MQTT_FIELDS", topics.FieldsYutongMQTT)),
},
UnifiedSubject: env("NATS_SUBJECT_UNIFIED", env("KAFKA_TOPIC_UNIFIED", topics.Unified)),
}
}
type natsOutboxRuntimeConfig struct {
Enabled bool
Config eventbus.DurableOutboxConfig
ReplayInterval time.Duration
}
func natsOutboxConfigFromEnv(registry *metrics.Registry, onError func(error)) natsOutboxRuntimeConfig {
directory := strings.TrimSpace(os.Getenv("NATS_OUTBOX_DIR"))
return natsOutboxRuntimeConfig{
Enabled: envBool("NATS_DURABLE_OUTBOX_ENABLED", false),
Config: eventbus.DurableOutboxConfig{
Directory: directory,
ReplayBatchSize: envInt("NATS_OUTBOX_REPLAY_BATCH_SIZE", 1000),
SyncWrites: envBool("NATS_OUTBOX_FSYNC", true),
CloseTimeout: time.Duration(envInt("NATS_OUTBOX_CLOSE_TIMEOUT_MS", 5000)) * time.Millisecond,
WALSegmentBytes: int64(envInt("NATS_OUTBOX_WAL_SEGMENT_BYTES", 16<<20)),
WALSegmentAge: time.Duration(envInt("NATS_OUTBOX_WAL_SEGMENT_AGE_MS", 5000)) * time.Millisecond,
WALAppendQueue: envInt("NATS_OUTBOX_WAL_APPEND_QUEUE_SIZE", 100_000),
WALCommitBatch: envInt("NATS_OUTBOX_WAL_COMMIT_BATCH_SIZE", 256),
WALCommitWait: time.Duration(envInt("NATS_OUTBOX_WAL_COMMIT_INTERVAL_MS", 1)) * time.Millisecond,
Metrics: registry,
Name: "nats-outbox",
OnError: onError,
},
ReplayInterval: time.Duration(envInt("NATS_OUTBOX_REPLAY_INTERVAL_MS", 1000)) * time.Millisecond,
}
}
func env(key string, fallback string) string {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
@@ -211,6 +772,13 @@ func envBool(key string, fallback bool) bool {
return value == "1" || value == "true" || value == "yes" || value == "on"
}
func boolMetric(value bool) float64 {
if value {
return 1
}
return 0
}
func splitCSV(value string) []string {
var out []string
for _, item := range strings.Split(value, ",") {

View File

@@ -0,0 +1,431 @@
package main
import (
"context"
"errors"
"io"
"log/slog"
"os"
"strings"
"testing"
"time"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/authentication"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/identity"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics"
)
func TestNATSSinkConfigFromEnvUsesExplicitSubjects(t *testing.T) {
t.Setenv("NATS_URL", "nats://172.17.111.56:4222")
t.Setenv("NATS_SUBJECT_GB32960_RAW", "custom.raw.gb32960")
t.Setenv("NATS_SUBJECT_JT808_RAW", "custom.raw.jt808")
t.Setenv("NATS_SUBJECT_YUTONG_MQTT_RAW", "custom.raw.yutong")
t.Setenv("NATS_SUBJECT_GB32960_FIELDS", "custom.fields.gb32960")
t.Setenv("NATS_SUBJECT_JT808_FIELDS", "custom.fields.jt808")
t.Setenv("NATS_SUBJECT_YUTONG_MQTT_FIELDS", "custom.fields.yutong")
t.Setenv("NATS_SUBJECT_UNIFIED", "custom.unified")
t.Setenv("NATS_OUTBOX_MAX_INFLIGHT", "12000")
t.Setenv("NATS_OUTBOX_ACK_TIMEOUT_MS", "4500")
cfg := natsSinkConfigFromEnv()
if got, want := cfg.URL, "nats://172.17.111.56:4222"; got != want {
t.Fatalf("URL = %q, want %q", got, want)
}
if got, want := cfg.RawSubjects[envelope.ProtocolGB32960], "custom.raw.gb32960"; got != want {
t.Fatalf("gb32960 subject = %q, want %q", got, want)
}
if got, want := cfg.RawSubjects[envelope.ProtocolJT808], "custom.raw.jt808"; got != want {
t.Fatalf("jt808 subject = %q, want %q", got, want)
}
if got, want := cfg.RawSubjects[envelope.ProtocolYutongMQTT], "custom.raw.yutong"; got != want {
t.Fatalf("yutong subject = %q, want %q", got, want)
}
if got, want := cfg.FieldsSubjects[envelope.ProtocolGB32960], "custom.fields.gb32960"; got != want {
t.Fatalf("gb32960 fields subject = %q, want %q", got, want)
}
if got, want := cfg.FieldsSubjects[envelope.ProtocolJT808], "custom.fields.jt808"; got != want {
t.Fatalf("jt808 fields subject = %q, want %q", got, want)
}
if got, want := cfg.FieldsSubjects[envelope.ProtocolYutongMQTT], "custom.fields.yutong"; got != want {
t.Fatalf("yutong fields subject = %q, want %q", got, want)
}
if got, want := cfg.UnifiedSubject, "custom.unified"; got != want {
t.Fatalf("unified subject = %q, want %q", got, want)
}
if got, want := cfg.AsyncMaxPending, 12000; got != want {
t.Fatalf("async max pending = %d, want %d", got, want)
}
if got, want := cfg.AsyncAckTimeout, 4500*time.Millisecond; got != want {
t.Fatalf("async ack timeout = %s, want %s", got, want)
}
}
func TestNATSOutboxConfigFromEnvIsExplicitAndDefaultsToSafeFsync(t *testing.T) {
t.Setenv("NATS_DURABLE_OUTBOX_ENABLED", "true")
t.Setenv("NATS_OUTBOX_DIR", "/var/lib/lingniu-go/nats-outbox")
t.Setenv("NATS_OUTBOX_REPLAY_BATCH_SIZE", "750")
t.Setenv("NATS_OUTBOX_REPLAY_INTERVAL_MS", "250")
t.Setenv("NATS_OUTBOX_CLOSE_TIMEOUT_MS", "7000")
t.Setenv("NATS_OUTBOX_WAL_SEGMENT_BYTES", "8388608")
t.Setenv("NATS_OUTBOX_WAL_SEGMENT_AGE_MS", "4000")
t.Setenv("NATS_OUTBOX_WAL_APPEND_QUEUE_SIZE", "50000")
t.Setenv("NATS_OUTBOX_WAL_COMMIT_BATCH_SIZE", "128")
t.Setenv("NATS_OUTBOX_WAL_COMMIT_INTERVAL_MS", "2")
t.Setenv("NATS_OUTBOX_FSYNC", "")
registry := metrics.NewRegistry()
onError := func(error) {}
runtime := natsOutboxConfigFromEnv(registry, onError)
if !runtime.Enabled {
t.Fatal("outbox should be enabled")
}
if got, want := runtime.Config.Directory, "/var/lib/lingniu-go/nats-outbox"; got != want {
t.Fatalf("directory = %q, want %q", got, want)
}
if got, want := runtime.Config.ReplayBatchSize, 750; got != want {
t.Fatalf("replay batch = %d, want %d", got, want)
}
if !runtime.Config.SyncWrites {
t.Fatal("outbox fsync should default to true")
}
if got, want := runtime.Config.CloseTimeout, 7*time.Second; got != want {
t.Fatalf("close timeout = %s, want %s", got, want)
}
if got, want := runtime.Config.WALSegmentBytes, int64(8<<20); got != want {
t.Fatalf("WAL segment bytes = %d, want %d", got, want)
}
if got, want := runtime.Config.WALSegmentAge, 4*time.Second; got != want {
t.Fatalf("WAL segment age = %s, want %s", got, want)
}
if got, want := runtime.Config.WALAppendQueue, 50_000; got != want {
t.Fatalf("WAL append queue = %d, want %d", got, want)
}
if got, want := runtime.Config.WALCommitBatch, 128; got != want {
t.Fatalf("WAL commit batch = %d, want %d", got, want)
}
if got, want := runtime.Config.WALCommitWait, 2*time.Millisecond; got != want {
t.Fatalf("WAL commit wait = %s, want %s", got, want)
}
if got, want := runtime.ReplayInterval, 250*time.Millisecond; got != want {
t.Fatalf("replay interval = %s, want %s", got, want)
}
if runtime.Config.Metrics != registry || runtime.Config.OnError == nil {
t.Fatal("metrics and error callback should be wired")
}
}
func TestNATSOutboxConfigDoesNotReuseLegacySpoolDirectory(t *testing.T) {
t.Setenv("NATS_DURABLE_OUTBOX_ENABLED", "")
t.Setenv("NATS_OUTBOX_DIR", "")
t.Setenv("NATS_SPOOL_DIR", "/var/lib/lingniu-go/nats-spool")
t.Setenv("NATS_OUTBOX_FSYNC", "false")
runtime := natsOutboxConfigFromEnv(nil, nil)
if runtime.Enabled {
t.Fatal("outbox must remain opt-in")
}
if runtime.Config.Directory != "" {
t.Fatalf("WAL directory must be explicit, got %q", runtime.Config.Directory)
}
if runtime.Config.SyncWrites {
t.Fatal("explicit false should disable fsync")
}
}
func TestBuildGB32960AuthenticatorUsesConfiguredCredentials(t *testing.T) {
t.Setenv("GB32960_AUTH_MODE", "enforce")
t.Setenv("GB32960_PLATFORM_CREDENTIALS_FILE", "")
t.Setenv("GB32960_PLATFORM_CREDENTIALS_JSON", `{"platform-a":"secret-a","platform-b":["secret-b-old","secret-b"]}`)
authenticator, mode, count, err := buildGB32960Authenticator()
if err != nil {
t.Fatalf("buildGB32960Authenticator() error = %v", err)
}
if mode != authentication.ModeEnforce || count != 2 {
t.Fatalf("mode=%q count=%d", mode, count)
}
result := authenticator.Authenticate(envelope.FrameEnvelope{
Protocol: envelope.ProtocolGB32960,
MessageID: "0x05",
Parsed: map[string]any{
"platform_login": map[string]any{"username": "platform-b", "password": "secret-b"},
},
})
if !result.Allowed || result.Status != authentication.StatusAccepted {
t.Fatalf("authentication result = %#v", result)
}
}
func TestBuildGB32960AuthenticatorRejectsEnforceWithoutCredentials(t *testing.T) {
t.Setenv("GB32960_AUTH_MODE", "enforce")
t.Setenv("GB32960_PLATFORM_CREDENTIALS_FILE", "")
t.Setenv("GB32960_PLATFORM_CREDENTIALS_JSON", "")
if _, _, _, err := buildGB32960Authenticator(); err == nil {
t.Fatal("expected enforce mode configuration error")
}
}
func TestGatewayConfiguresIdentityLookupCacheTTL(t *testing.T) {
source, err := os.ReadFile("main.go")
if err != nil {
t.Fatalf("read main.go: %v", err)
}
for _, want := range []string{"LookupCacheTTL", "StaleLookupTTL", "IDENTITY_LOOKUP_CACHE_TTL_SECONDS", "IDENTITY_STALE_LOOKUP_TTL_SECONDS", "IDENTITY_LOOKUP_CACHE_MAX_ENTRIES", "IDENTITY_LOOKUP_CACHE_CLEANUP_INTERVAL_SECONDS", "vehicle_gateway_identity_cache_entries", "vehicle_gateway_jt808_registration_write_total", "OnRegistrationWriteResult", "IDENTITY_SOURCE_CODE_LOOKUP_ENABLED", "IDENTITY_RESOLVE_TIMEOUT_MS", "IDENTITY_SNAPSHOT_ONLY_ENABLED", "IDENTITY_SNAPSHOT_REFRESH_INTERVAL_SECONDS", "startIdentitySnapshotRefresh", "JT808_REGISTRATION_LOCATION_TOUCH_RETRY_INTERVAL_SECONDS", "JT808_REGISTRATION_WRITE_RETRY_ATTEMPTS", "JT808_REGISTRATION_WRITE_RETRY_DELAY_MS", "JT808_REGISTRATION_ASYNC_WRITE_ENABLED", "JT808_REGISTRATION_WRITE_QUEUE_SIZE", "JT808_REGISTRATION_WRITE_WORKERS", "JT808_REGISTRATION_WRITE_ENQUEUE_TIMEOUT_MS", "TimeoutResolver"} {
if !strings.Contains(string(source), want) {
t.Fatalf("gateway should expose identity lookup cache ttl, missing %s", want)
}
}
}
func TestRecordIdentityCacheStatsIncludesLocationTouchFailures(t *testing.T) {
registry := metrics.NewRegistry()
recordIdentityCacheStats(registry, identity.CacheStats{
LookupEntries: 1,
RegistrationEntries: 2,
SourceCodeEntries: 3,
LocationTouchEntries: 4,
LocationTouchFailureEntries: 5,
SnapshotBindingEntries: 6,
SnapshotIdentifierEntries: 7,
SnapshotRegistrationEntries: 8,
SnapshotSourceEntries: 9,
SnapshotReady: true,
SnapshotRefreshedAt: time.Unix(1234, 0),
MaxEntries: 9,
})
rendered := registry.Render()
for _, want := range []string{
`vehicle_gateway_identity_cache_entries{cache="location_touch_failure"} 5`,
`vehicle_gateway_identity_cache_entries{cache="registration_write_queue"} 0`,
`vehicle_gateway_identity_cache_entries{cache="snapshot_binding"} 6`,
`vehicle_gateway_identity_cache_entries{cache="snapshot_identifier"} 7`,
`vehicle_gateway_identity_cache_entries{cache="snapshot_registration"} 8`,
`vehicle_gateway_identity_cache_entries{cache="snapshot_source"} 9`,
`vehicle_gateway_identity_cache_max_entries{cache="location_touch_failure"} 9`,
`vehicle_gateway_identity_snapshot_ready 1`,
`vehicle_gateway_identity_snapshot_last_success_unix_seconds 1234`,
} {
if !strings.Contains(rendered, want) {
t.Fatalf("identity cache metrics missing %s in:\n%s", want, rendered)
}
}
}
func TestRecordIdentitySnapshotRefresh(t *testing.T) {
registry := metrics.NewRegistry()
recordIdentitySnapshotRefresh(registry, identity.SnapshotRefreshResult{
BindingEntries: 10,
IdentifierEntries: 20,
RegistrationEntries: 30,
SourceEntries: 40,
RefreshedAt: time.Unix(5678, 0),
}, "ok")
rendered := registry.Render()
for _, want := range []string{
`vehicle_gateway_identity_snapshot_refresh_total{status="ok"} 1`,
`vehicle_gateway_identity_snapshot_entries{kind="binding"} 10`,
`vehicle_gateway_identity_snapshot_entries{kind="identifier"} 20`,
`vehicle_gateway_identity_snapshot_entries{kind="registration"} 30`,
`vehicle_gateway_identity_snapshot_entries{kind="source"} 40`,
`vehicle_gateway_identity_snapshot_last_success_unix_seconds 5678`,
} {
if !strings.Contains(rendered, want) {
t.Fatalf("identity snapshot metrics missing %s in:\n%s", want, rendered)
}
}
}
func TestIdentityDatabaseMaintenanceDoesNotBlockIngressStartup(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
db := &blockingIdentityDatabase{
started: make(chan struct{}),
release: make(chan struct{}),
}
logger := slog.New(slog.NewTextHandler(io.Discard, nil))
startedAt := time.Now()
startIdentityDatabaseMaintenance(ctx, logger, db, nil, false, time.Second, time.Second)
if elapsed := time.Since(startedAt); elapsed > 100*time.Millisecond {
t.Fatalf("database maintenance blocked gateway startup for %s", elapsed)
}
select {
case <-db.started:
case <-time.After(time.Second):
t.Fatal("database maintenance did not start in background")
}
close(db.release)
}
func TestIdentitySnapshotRefreshDoesNotBlockIngressStartup(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
refresher := &blockingIdentitySnapshotRefresher{started: make(chan struct{})}
logger := slog.New(slog.NewTextHandler(io.Discard, nil))
startedAt := time.Now()
startIdentitySnapshotRefresh(ctx, logger, metrics.NewRegistry(), refresher, time.Hour, time.Second)
if elapsed := time.Since(startedAt); elapsed > 100*time.Millisecond {
t.Fatalf("snapshot refresh blocked gateway startup for %s", elapsed)
}
select {
case <-refresher.started:
case <-time.After(time.Second):
t.Fatal("snapshot refresh did not start in background")
}
}
func TestRecordJT808RegistrationWriteResult(t *testing.T) {
registry := metrics.NewRegistry()
recordJT808RegistrationWriteResult(registry, identity.RegistrationWriteResult{
Mode: "async_background",
Status: "error",
})
rendered := registry.Render()
for _, want := range []string{
`vehicle_gateway_jt808_registration_write_total{mode="async_background",status="error"} 1`,
`vehicle_gateway_last_jt808_registration_write_unix_seconds{mode="async_background",status="error"}`,
} {
if !strings.Contains(rendered, want) {
t.Fatalf("registration write metric missing %s in:\n%s", want, rendered)
}
}
}
type blockingIdentityDatabase struct {
started chan struct{}
release chan struct{}
}
func (d *blockingIdentityDatabase) PingContext(ctx context.Context) error {
close(d.started)
select {
case <-ctx.Done():
return ctx.Err()
case <-d.release:
return nil
}
}
type blockingIdentitySnapshotRefresher struct {
started chan struct{}
}
func (r *blockingIdentitySnapshotRefresher) RefreshSnapshot(ctx context.Context) (identity.SnapshotRefreshResult, error) {
close(r.started)
<-ctx.Done()
return identity.SnapshotRefreshResult{}, errors.New("test refresh stopped")
}
func TestGatewayDefaultsTo100KConnectionCeiling(t *testing.T) {
source, err := os.ReadFile("main.go")
if err != nil {
t.Fatalf("read main.go: %v", err)
}
if !strings.Contains(string(source), `envInt("TCP_MAX_CONNECTIONS", 120_000)`) {
t.Fatal("gateway should default TCP_MAX_CONNECTIONS to a 100K-ready ceiling")
}
}
func TestGatewayPassesMetricsRegistryToAsyncSink(t *testing.T) {
source, err := os.ReadFile("main.go")
if err != nil {
t.Fatalf("read main.go: %v", err)
}
for _, want := range []string{
"buildSink(ctx, logger, registry)",
"Metrics: registry",
"EnqueueTimeout: enqueueTimeout",
"NewPartitionedAsyncSink",
"NATS_PARTITIONED_ASYNC_ENABLED",
"KAFKA_PARTITIONED_ASYNC_ENABLED",
"_ASYNC_RAW_QUEUE_SIZE",
"_ASYNC_DERIVED_QUEUE_SIZE",
"_ASYNC_RAW_ENQUEUE_TIMEOUT_MS",
"_ASYNC_DERIVED_ENQUEUE_TIMEOUT_MS",
"RawEnqueueTimeout",
"DerivedEnqueueTimeout",
"NATS_ASYNC_ENQUEUE_TIMEOUT_MS",
"KAFKA_ASYNC_ENQUEUE_TIMEOUT_MS",
`Name: "nats"`,
`Name: "kafka"`,
} {
if !strings.Contains(string(source), want) {
t.Fatalf("gateway async sink metrics wiring missing %s", want)
}
}
}
func TestGatewayExposesNATSDurableSpoolConfig(t *testing.T) {
source, err := os.ReadFile("main.go")
if err != nil {
t.Fatalf("read main.go: %v", err)
}
for _, want := range []string{
"NATS_DURABLE_OUTBOX_ENABLED",
"NATS_OUTBOX_DIR",
"NATS_OUTBOX_MAX_INFLIGHT",
"NATS_OUTBOX_ACK_TIMEOUT_MS",
"NATS_OUTBOX_FSYNC",
"NATS_OUTBOX_CLOSE_TIMEOUT_MS",
"NATS_OUTBOX_REPLAY_BATCH_SIZE",
"NATS_OUTBOX_REPLAY_INTERVAL_MS",
"NATS_OUTBOX_WAL_SEGMENT_BYTES",
"NATS_OUTBOX_WAL_SEGMENT_AGE_MS",
"NATS_OUTBOX_WAL_APPEND_QUEUE_SIZE",
"NATS_OUTBOX_WAL_COMMIT_BATCH_SIZE",
"NATS_OUTBOX_WAL_COMMIT_INTERVAL_MS",
"eventbus.NewDurableOutboxSink",
"nats durable outbox enabled",
"NATS_SPOOL_DIR",
"NATS_SPOOL_REPLAY_BATCH_SIZE",
"NATS_SPOOL_REPLAY_INTERVAL_MS",
"nats durable spool enabled",
"nats spool replay failed",
"eventbus.NewDurableSink",
"Metrics: registry",
`Name: "nats"`,
`Name: "kafka"`,
} {
if !strings.Contains(string(source), want) {
t.Fatalf("gateway nats spool wiring missing %s", want)
}
}
}
func TestNATSSinkConfigFromEnvDefaultsToGoSubjects(t *testing.T) {
t.Setenv("NATS_URL", "nats://172.17.111.56:4222")
cfg := natsSinkConfigFromEnv()
if got, want := cfg.RawSubjects[envelope.ProtocolGB32960], "vehicle.raw.go.gb32960.v1"; got != want {
t.Fatalf("gb32960 subject = %q, want %q", got, want)
}
if got, want := cfg.RawSubjects[envelope.ProtocolJT808], "vehicle.raw.go.jt808.v1"; got != want {
t.Fatalf("jt808 subject = %q, want %q", got, want)
}
if got, want := cfg.RawSubjects[envelope.ProtocolYutongMQTT], "vehicle.raw.go.yutong-mqtt.v1"; got != want {
t.Fatalf("yutong subject = %q, want %q", got, want)
}
if got, want := cfg.FieldsSubjects[envelope.ProtocolGB32960], "vehicle.fields.go.gb32960.v1"; got != want {
t.Fatalf("gb32960 fields subject = %q, want %q", got, want)
}
if got, want := cfg.FieldsSubjects[envelope.ProtocolJT808], "vehicle.fields.go.jt808.v1"; got != want {
t.Fatalf("jt808 fields subject = %q, want %q", got, want)
}
if got, want := cfg.FieldsSubjects[envelope.ProtocolYutongMQTT], "vehicle.fields.go.yutong-mqtt.v1"; got != want {
t.Fatalf("yutong fields subject = %q, want %q", got, want)
}
if got, want := cfg.UnifiedSubject, "vehicle.event.go.unified.v1"; got != want {
t.Fatalf("unified subject = %q, want %q", got, want)
}
}

View File

@@ -4,17 +4,27 @@ import (
"context"
"database/sql"
"encoding/json"
"errors"
"fmt"
"os"
"os/signal"
"sort"
"strconv"
"strings"
"sync"
"syscall"
"time"
"github.com/segmentio/kafka-go"
_ "github.com/taosdata/driver-go/v3/taosWS"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/eventbus"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/health"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/history"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/observability"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/topics"
)
func main() {
@@ -23,6 +33,10 @@ func main() {
defer stop()
cfg := loadConfig()
if err := cfg.Validate(); err != nil {
logger.Error("invalid history writer config", "error", err)
os.Exit(1)
}
db, err := sql.Open(cfg.TDengineDriver, cfg.TDengineDSN)
if err != nil {
logger.Error("tdengine open failed", "error", err)
@@ -33,15 +47,52 @@ func main() {
logger.Error("tdengine ping failed", "error", err)
os.Exit(1)
}
registry := metrics.NewRegistry()
metrics.RegisterKafkaConsumerInfo(registry, "vehicle-history-writer", cfg.KafkaGroup, cfg.KafkaTopics)
registry.SetGauge("vehicle_history_config", metrics.Labels{"setting": "workers"}, float64(cfg.Workers))
health.Start(ctx, logger, health.NewServer(env("HEALTH_ADDR", ""), "vehicle-history-writer", []health.Check{
{Name: "tdengine", Check: db.PingContext},
}, registry))
writer := history.NewWriter(db)
writer := history.NewWriterWithDatabase(db, cfg.TDengineDatabase)
if cfg.EnsureSchema {
if err := writer.EnsureSchema(ctx, cfg.TDengineDatabase); err != nil {
logger.Error("tdengine schema bootstrap failed", "error", err)
os.Exit(1)
}
}
var appender historyAppender = retryHistoryAppender{
delegate: writer,
attempts: cfg.RetryAttempts,
delay: cfg.RetryDelay,
registry: registry,
}
logger.Info("history writer started",
"driver", cfg.TDengineDriver,
"group", cfg.KafkaGroup,
"topics", strings.Join(cfg.KafkaTopics, ","),
"workers", cfg.Workers,
"batch_size", cfg.BatchSize,
"batch_wait_ms", cfg.BatchWait,
"retry_attempts", cfg.RetryAttempts,
"retry_delay_ms", cfg.RetryDelay.Milliseconds())
var workers sync.WaitGroup
for workerID := 1; workerID <= cfg.Workers; workerID++ {
workers.Add(1)
go func(id int) {
defer workers.Done()
runHistoryConsumer(ctx, logger, registry, appender, cfg, id)
}(workerID)
}
workers.Wait()
}
func runHistoryConsumer(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender historyAppender, cfg config, workerID int) {
reader := kafka.NewReader(kafka.ReaderConfig{
Brokers: cfg.KafkaBrokers,
GroupID: cfg.KafkaGroup,
@@ -51,10 +102,9 @@ func main() {
})
defer reader.Close()
logger.Info("history writer started",
"driver", cfg.TDengineDriver,
"group", cfg.KafkaGroup,
"topics", strings.Join(cfg.KafkaTopics, ","))
workerLabels := metrics.Labels{"worker": strconv.Itoa(workerID)}
registry.SetGauge("vehicle_history_worker_active", workerLabels, 1)
defer registry.SetGauge("vehicle_history_worker_active", workerLabels, 0)
for {
message, err := reader.FetchMessage(ctx)
@@ -65,20 +115,667 @@ func main() {
logger.Error("kafka fetch failed", "error", err)
continue
}
batch := collectHistoryBatch(ctx, reader, message, cfg.BatchSize, time.Duration(cfg.BatchWait)*time.Millisecond)
processHistoryBatchReliablyForWorker(ctx, logger, registry, appender, reader, batch, cfg.RetryDelay, workerLabels)
}
}
const kafkaMessageOperationTimeout = 30 * time.Second
type historyAppender interface {
AppendAll(context.Context, envelope.FrameEnvelope) error
AppendAllBatch(context.Context, []envelope.FrameEnvelope) error
}
type historyResultAppender interface {
AppendAllWithResult(context.Context, envelope.FrameEnvelope) (history.AppendResult, error)
AppendAllBatchWithResult(context.Context, []envelope.FrameEnvelope) (history.AppendResult, error)
}
type kafkaMessageCommitter interface {
CommitMessages(context.Context, ...kafka.Message) error
}
type kafkaMessageFetcher interface {
FetchMessage(context.Context) (kafka.Message, error)
}
type historyBatchItem struct {
message kafka.Message
valid bool
processed bool
}
type historyBatchOutcome struct {
commitMessages []kafka.Message
retryMessages []kafka.Message
}
func collectHistoryBatch(ctx context.Context, fetcher kafkaMessageFetcher, first kafka.Message, maxSize int, maxWait time.Duration) []kafka.Message {
if maxSize <= 1 {
return []kafka.Message{first}
}
if maxWait <= 0 {
maxWait = 100 * time.Millisecond
}
batch := []kafka.Message{first}
deadline := time.Now().Add(maxWait)
for len(batch) < maxSize {
remaining := time.Until(deadline)
if remaining <= 0 {
break
}
fetchCtx, cancel := context.WithTimeout(ctx, remaining)
message, err := fetcher.FetchMessage(fetchCtx)
cancel()
if err != nil {
break
}
batch = append(batch, message)
}
return batch
}
func processHistoryMessage(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender historyAppender, committer kafkaMessageCommitter, message kafka.Message) {
messageCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), kafkaMessageOperationTimeout)
defer cancel()
addWriterMetric(registry, "vehicle_history_kafka_messages_total", message, "received")
addWriterLagMetric(registry, message)
var env envelope.FrameEnvelope
if err := json.Unmarshal(message.Value, &env); err != nil {
addWriterMetric(registry, "vehicle_history_kafka_messages_total", message, "invalid_json")
logger.Warn("skip invalid envelope json", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "error", err)
if err := committer.CommitMessages(messageCtx, message); err != nil {
addWriterMetric(registry, "vehicle_history_kafka_commits_total", message, "error")
logger.Error("kafka commit invalid envelope failed", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "error", err)
return
}
addWriterMetric(registry, "vehicle_history_kafka_commits_total", message, "ok")
return
}
if status, err := topics.ValidateRawEnvelope(message.Topic, env); err != nil {
addWriterMetric(registry, "vehicle_history_kafka_messages_total", message, status)
logger.Warn("skip mismatched history raw envelope", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "protocol", env.Protocol, "event_id", env.StableEventID(), "event_kind", env.EventKind, "error", err)
if err := committer.CommitMessages(messageCtx, message); err != nil {
addWriterMetric(registry, "vehicle_history_kafka_commits_total", message, "error")
logger.Error("kafka commit mismatched raw envelope failed", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "event_id", env.StableEventID(), "status", status, "error", err)
return
}
addWriterMetric(registry, "vehicle_history_kafka_commits_total", message, "ok")
return
}
recordHistoryParsedFieldMetrics(registry, message, env)
result, err := appendHistoryEnvelope(messageCtx, appender, env)
if err != nil {
addWriterMetric(registry, "vehicle_history_writes_total", message, "error")
logger.Error("tdengine raw append failed", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "event_id", env.StableEventID(), "error", err)
return
}
recordHistoryDerivedMetrics(registry, []kafka.Message{message}, []envelope.FrameEnvelope{env}, result)
if result.LocationError != nil {
logger.Warn("tdengine location append failed after raw append", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "event_id", env.StableEventID(), "error", result.LocationError)
}
addWriterMetric(registry, "vehicle_history_writes_total", message, "ok")
recordHistoryWriteE2EDuration(registry, message, env)
if err := committer.CommitMessages(messageCtx, message); err != nil {
addWriterMetric(registry, "vehicle_history_kafka_commits_total", message, "error")
logger.Error("kafka commit failed", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "error", err)
return
}
addWriterMetric(registry, "vehicle_history_kafka_commits_total", message, "ok")
}
func processHistoryBatch(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender historyAppender, committer kafkaMessageCommitter, messages []kafka.Message) historyBatchOutcome {
return processHistoryBatchForWorker(ctx, logger, registry, appender, committer, messages, nil)
}
func processHistoryBatchForWorker(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender historyAppender, committer kafkaMessageCommitter, messages []kafka.Message, workerLabels metrics.Labels) historyBatchOutcome {
if len(messages) == 0 {
return historyBatchOutcome{}
}
messageCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), kafkaMessageOperationTimeout)
defer cancel()
envelopes := make([]envelope.FrameEnvelope, 0, len(messages))
validMessages := make([]kafka.Message, 0, len(messages))
validItemIndexes := make([]int, 0, len(messages))
items := make([]historyBatchItem, 0, len(messages))
for _, message := range messages {
itemIndex := len(items)
addWriterMetric(registry, "vehicle_history_kafka_messages_total", message, "received")
addWriterLagMetric(registry, message)
var env envelope.FrameEnvelope
if err := json.Unmarshal(message.Value, &env); err != nil {
addWriterMetric(registry, "vehicle_history_kafka_messages_total", message, "invalid_json")
logger.Warn("skip invalid envelope json", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "error", err)
_ = reader.CommitMessages(ctx, message)
items = append(items, historyBatchItem{message: message, processed: true})
continue
}
if err := writer.AppendAll(ctx, env); err != nil {
logger.Error("tdengine append failed", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "event_id", env.StableEventID(), "error", err)
if status, err := topics.ValidateRawEnvelope(message.Topic, env); err != nil {
addWriterMetric(registry, "vehicle_history_kafka_messages_total", message, status)
logger.Warn("skip mismatched history raw envelope", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "protocol", env.Protocol, "event_id", env.StableEventID(), "event_kind", env.EventKind, "error", err)
items = append(items, historyBatchItem{message: message, processed: true})
continue
}
if err := reader.CommitMessages(ctx, message); err != nil {
logger.Error("kafka commit failed", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "error", err)
recordHistoryParsedFieldMetrics(registry, message, env)
envelopes = append(envelopes, env)
validMessages = append(validMessages, message)
validItemIndexes = append(validItemIndexes, itemIndex)
items = append(items, historyBatchItem{message: message, valid: true})
}
if len(envelopes) > 0 {
setBatchPendingForWorker(registry, workerLabels, len(messages), len(envelopes))
defer setBatchPendingForWorker(registry, workerLabels, 0, 0)
started := time.Now()
result, err := appendHistoryBatch(messageCtx, appender, envelopes)
elapsed := time.Since(started)
status := "ok"
if err != nil {
status = "error"
}
addBatchMetric(registry, "vehicle_history_batch_flush_total", status, 1)
addBatchMetric(registry, "vehicle_history_batch_rows_total", status, float64(len(envelopes)))
setBatchDuration(registry, status, elapsed)
if err != nil {
first := messages[0]
logger.Error("tdengine raw batch append failed", "topic", first.Topic, "partition", first.Partition, "offset", first.Offset, "rows", len(envelopes), "error", err)
if isTransientTDengineHistoryError(err) {
for _, message := range validMessages {
addWriterMetric(registry, "vehicle_history_writes_total", message, "error")
}
addBatchMetric(registry, "vehicle_history_batch_fallback_total", "skipped_transient", 1)
committed, commitErr := commitHistoryProcessedPrefixAfterFailure(messageCtx, logger, registry, committer, items)
return historyFailureOutcome(messages, committed, commitErr)
}
if fallbackErr := fallbackHistoryBatchToSingles(messageCtx, logger, registry, appender, validMessages, envelopes, validItemIndexes, items); fallbackErr != nil {
addBatchMetric(registry, "vehicle_history_batch_fallback_total", "error", 1)
committed, commitErr := commitHistoryProcessedPrefixAfterFailure(messageCtx, logger, registry, committer, items)
return historyFailureOutcome(messages, committed, commitErr)
}
addBatchMetric(registry, "vehicle_history_batch_fallback_total", "ok", 1)
committed, commitErr := commitHistoryProcessedPrefixAfterFailure(messageCtx, logger, registry, committer, items)
return historyFailureOutcome(messages, committed, commitErr)
}
recordHistoryDerivedMetrics(registry, validMessages, envelopes, result)
if result.LocationError != nil {
first := messages[0]
logger.Warn("tdengine location batch append failed after raw append", "topic", first.Topic, "partition", first.Partition, "offset", first.Offset, "rows", len(envelopes), "location_rows", result.LocationRows, "error", result.LocationError)
}
for index, message := range validMessages {
addWriterMetric(registry, "vehicle_history_writes_total", message, "ok")
recordHistoryWriteE2EDuration(registry, message, envelopes[index])
}
}
if err := committer.CommitMessages(messageCtx, messages...); err != nil {
for _, message := range messages {
addWriterMetric(registry, "vehicle_history_kafka_commits_total", message, "error")
}
first := messages[0]
logger.Error("kafka batch commit failed", "topic", first.Topic, "partition", first.Partition, "offset", first.Offset, "messages", len(messages), "error", err)
return historyBatchOutcome{commitMessages: messages}
}
for _, message := range messages {
addWriterMetric(registry, "vehicle_history_kafka_commits_total", message, "ok")
}
return historyBatchOutcome{}
}
func historyFailureOutcome(messages []kafka.Message, committed []kafka.Message, commitErr error) historyBatchOutcome {
outcome := historyBatchOutcome{
retryMessages: eventbus.MessagesAfterCommittedPrefixes(messages, committed),
}
if commitErr != nil {
outcome.commitMessages = committed
}
return outcome
}
func processHistoryBatchReliably(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender historyAppender, committer kafkaMessageCommitter, messages []kafka.Message, retryDelay time.Duration) {
processHistoryBatchReliablyForWorker(ctx, logger, registry, appender, committer, messages, retryDelay, nil)
}
func processHistoryBatchReliablyForWorker(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender historyAppender, committer kafkaMessageCommitter, messages []kafka.Message, retryDelay time.Duration, workerLabels metrics.Labels) {
defer registry.SetGauge("vehicle_history_retry_pending_messages", workerLabels, 0)
pending := messages
for len(pending) > 0 {
outcome := processHistoryBatchForWorker(ctx, logger, registry, appender, committer, pending, workerLabels)
if len(outcome.commitMessages) > 0 {
registry.IncCounter("vehicle_history_batch_retries_total", metrics.Labels{"reason": "commit_error"})
if !retryHistoryCommit(ctx, logger, registry, committer, outcome.commitMessages, retryDelay) {
return
}
}
pending = outcome.retryMessages
registry.SetGauge("vehicle_history_retry_pending_messages", workerLabels, float64(len(pending)))
if len(pending) == 0 {
return
}
registry.IncCounter("vehicle_history_batch_retries_total", metrics.Labels{"reason": "write_error"})
if !waitForHistoryRetry(ctx, retryDelay) {
return
}
}
}
func retryHistoryCommit(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, committer kafkaMessageCommitter, messages []kafka.Message, retryDelay time.Duration) bool {
for len(messages) > 0 {
operationCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), kafkaMessageOperationTimeout)
err := committer.CommitMessages(operationCtx, messages...)
cancel()
if err == nil {
for _, message := range messages {
addWriterMetric(registry, "vehicle_history_kafka_commits_total", message, "ok")
}
return true
}
for _, message := range messages {
addWriterMetric(registry, "vehicle_history_kafka_commits_total", message, "error")
}
first := messages[0]
logger.Error("kafka commit retry failed", "topic", first.Topic, "partition", first.Partition, "offset", first.Offset, "messages", len(messages), "error", err)
registry.IncCounter("vehicle_history_batch_retries_total", metrics.Labels{"reason": "commit_error"})
if !waitForHistoryRetry(ctx, retryDelay) {
return false
}
}
return true
}
func waitForHistoryRetry(ctx context.Context, retryDelay time.Duration) bool {
if retryDelay <= 0 {
retryDelay = 100 * time.Millisecond
}
timer := time.NewTimer(retryDelay)
defer timer.Stop()
select {
case <-ctx.Done():
return false
case <-timer.C:
return true
}
}
func fallbackHistoryBatchToSingles(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender historyAppender, messages []kafka.Message, envelopes []envelope.FrameEnvelope, itemIndexes []int, items []historyBatchItem) error {
for index, env := range envelopes {
message := messages[index]
started := time.Now()
result, err := appendHistoryEnvelope(ctx, appender, env)
setFallbackDuration(registry, statusFromError(err), time.Since(started))
if err != nil {
addWriterMetric(registry, "vehicle_history_writes_total", message, "error")
logger.Error("tdengine raw single append failed after batch fallback", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "event_id", env.StableEventID(), "error", err)
return err
}
recordHistoryDerivedMetrics(registry, []kafka.Message{message}, []envelope.FrameEnvelope{env}, result)
if result.LocationError != nil {
logger.Warn("tdengine location append failed after raw single fallback", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "event_id", env.StableEventID(), "error", result.LocationError)
}
addWriterMetric(registry, "vehicle_history_writes_total", message, "ok")
recordHistoryWriteE2EDuration(registry, message, env)
items[itemIndexes[index]].processed = true
}
return nil
}
func commitHistoryProcessedPrefixAfterFailure(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, committer kafkaMessageCommitter, items []historyBatchItem) ([]kafka.Message, error) {
committable := processedHistoryPrefixMessagesByPartition(items)
if len(committable) == 0 {
return nil, nil
}
if err := committer.CommitMessages(ctx, committable...); err != nil {
for _, message := range committable {
addWriterMetric(registry, "vehicle_history_kafka_commits_total", message, "error")
}
first := committable[0]
logger.Error("kafka processed-prefix commit failed after raw batch append failure", "topic", first.Topic, "partition", first.Partition, "offset", first.Offset, "messages", len(committable), "error", err)
return committable, err
}
for _, message := range committable {
addWriterMetric(registry, "vehicle_history_kafka_commits_total", message, "ok")
}
return committable, nil
}
func processedHistoryPrefixMessagesByPartition(items []historyBatchItem) []kafka.Message {
type partitionKey struct {
topic string
partition int
}
groups := map[partitionKey][]historyBatchItem{}
for _, item := range items {
key := partitionKey{topic: item.message.Topic, partition: item.message.Partition}
groups[key] = append(groups[key], item)
}
var keys []partitionKey
for key := range groups {
keys = append(keys, key)
}
sort.Slice(keys, func(i, j int) bool {
if keys[i].topic != keys[j].topic {
return keys[i].topic < keys[j].topic
}
return keys[i].partition < keys[j].partition
})
var out []kafka.Message
for _, key := range keys {
group := groups[key]
sort.Slice(group, func(i, j int) bool {
return group[i].message.Offset < group[j].message.Offset
})
var previousOffset int64
for index, item := range group {
if index > 0 && item.message.Offset != previousOffset+1 {
break
}
if !item.processed {
break
}
out = append(out, item.message)
previousOffset = item.message.Offset
}
}
return out
}
func appendHistoryEnvelope(ctx context.Context, appender historyAppender, env envelope.FrameEnvelope) (history.AppendResult, error) {
if resultAppender, ok := appender.(historyResultAppender); ok {
return resultAppender.AppendAllWithResult(ctx, env)
}
if err := appender.AppendAll(ctx, env); err != nil {
return history.AppendResult{}, err
}
return history.AppendResult{RawRows: 1}, nil
}
func appendHistoryBatch(ctx context.Context, appender historyAppender, envelopes []envelope.FrameEnvelope) (history.AppendResult, error) {
if resultAppender, ok := appender.(historyResultAppender); ok {
return resultAppender.AppendAllBatchWithResult(ctx, envelopes)
}
if err := appender.AppendAllBatch(ctx, envelopes); err != nil {
return history.AppendResult{}, err
}
return history.AppendResult{RawRows: len(envelopes)}, nil
}
type retryHistoryAppender struct {
delegate historyAppender
attempts int
delay time.Duration
registry *metrics.Registry
}
func (a retryHistoryAppender) AppendAll(ctx context.Context, env envelope.FrameEnvelope) error {
_, err := a.AppendAllWithResult(ctx, env)
return err
}
func (a retryHistoryAppender) AppendAllWithResult(ctx context.Context, env envelope.FrameEnvelope) (history.AppendResult, error) {
if a.delegate == nil {
return history.AppendResult{}, nil
}
attempts := a.attempts
if attempts <= 0 {
attempts = 1
}
var result history.AppendResult
var err error
for attempt := 1; attempt <= attempts; attempt++ {
result, err = appendHistoryEnvelope(ctx, a.delegate, env)
if err == nil || !isTransientTDengineHistoryError(err) {
return result, err
}
if attempt == attempts {
a.recordRetry("single", "exhausted")
return result, err
}
a.recordRetry("single", "retry")
if waitErr := sleepBeforeRetry(ctx, a.delay); waitErr != nil {
return result, waitErr
}
}
return result, err
}
func (a retryHistoryAppender) AppendAllBatch(ctx context.Context, envelopes []envelope.FrameEnvelope) error {
_, err := a.AppendAllBatchWithResult(ctx, envelopes)
return err
}
func (a retryHistoryAppender) AppendAllBatchWithResult(ctx context.Context, envelopes []envelope.FrameEnvelope) (history.AppendResult, error) {
if a.delegate == nil {
return history.AppendResult{}, nil
}
attempts := a.attempts
if attempts <= 0 {
attempts = 1
}
var result history.AppendResult
var err error
for attempt := 1; attempt <= attempts; attempt++ {
result, err = appendHistoryBatch(ctx, a.delegate, envelopes)
if err == nil || !isTransientTDengineHistoryError(err) {
return result, err
}
if attempt == attempts {
a.recordRetry("batch", "exhausted")
return result, err
}
a.recordRetry("batch", "retry")
if waitErr := sleepBeforeRetry(ctx, a.delay); waitErr != nil {
return result, waitErr
}
}
return result, err
}
func (a retryHistoryAppender) recordRetry(operation string, status string) {
if a.registry == nil {
return
}
labels := metrics.Labels{"operation": operation, "status": status}
a.registry.IncCounter("vehicle_history_write_retries_total", labels)
metrics.RecordLastActivity(a.registry, "vehicle_history_last_write_retry_unix_seconds", labels)
}
func sleepBeforeRetry(ctx context.Context, delay time.Duration) error {
if delay <= 0 {
return nil
}
timer := time.NewTimer(delay)
defer timer.Stop()
select {
case <-ctx.Done():
return ctx.Err()
case <-timer.C:
return nil
}
}
func isTransientTDengineHistoryError(err error) bool {
if err == nil || errors.Is(err, context.Canceled) || errors.Is(err, context.DeadlineExceeded) {
return false
}
text := strings.ToLower(strings.TrimSpace(err.Error()))
return strings.Contains(text, "timeout") ||
strings.Contains(text, "temporary") ||
strings.Contains(text, "temporarily") ||
strings.Contains(text, "connection refused") ||
strings.Contains(text, "connection reset") ||
strings.Contains(text, "connection closed") ||
strings.Contains(text, "broken pipe") ||
strings.Contains(text, "bad connection") ||
strings.Contains(text, "i/o timeout") ||
text == "eof" ||
strings.Contains(text, "unexpected eof") ||
strings.Contains(text, "server is down") ||
strings.Contains(text, "network is unreachable") ||
strings.Contains(text, "no route to host")
}
func addWriterMetric(registry *metrics.Registry, name string, message kafka.Message, status string) {
if registry == nil {
return
}
labels := metrics.Labels{"topic": message.Topic, "status": status}
registry.IncCounter(name, labels)
switch name {
case "vehicle_history_kafka_messages_total":
metrics.RecordLastActivity(registry, "vehicle_history_last_message_unix_seconds", labels)
case "vehicle_history_writes_total":
metrics.RecordLastActivity(registry, "vehicle_history_last_write_unix_seconds", labels)
case "vehicle_history_kafka_commits_total":
metrics.RecordLastActivity(registry, "vehicle_history_last_commit_unix_seconds", labels)
case "vehicle_history_location_writes_total":
metrics.RecordLastActivity(registry, "vehicle_history_last_location_write_unix_seconds", labels)
}
}
func addBatchMetric(registry *metrics.Registry, name string, status string, value float64) {
if registry == nil {
return
}
registry.AddCounter(name, metrics.Labels{"status": status}, value)
}
func recordHistoryParsedFieldMetrics(registry *metrics.Registry, message kafka.Message, env envelope.FrameEnvelope) {
if registry == nil || !envelope.IsRealtimeTelemetryFrame(env) {
return
}
status := "present"
if len(env.ParsedFields) == 0 {
status = "missing"
}
registry.IncCounter("vehicle_history_parsed_fields_total", metrics.Labels{
"topic": message.Topic,
"protocol": historyProtocolLabel(env.Protocol),
"status": status,
})
}
func recordHistoryDerivedMetrics(registry *metrics.Registry, messages []kafka.Message, envelopes []envelope.FrameEnvelope, result history.AppendResult) {
if registry == nil || len(messages) == 0 {
return
}
batchStatus := "skipped"
if result.LocationError != nil {
batchStatus = "error"
} else if result.LocationRows > 0 {
batchStatus = "ok"
}
for index, message := range messages {
status := batchStatus
if index < len(envelopes) {
status = history.LocationStatus(envelopes[index])
if status == history.LocationStatusOK && result.LocationError != nil {
status = "error"
}
}
addWriterMetric(registry, "vehicle_history_location_writes_total", message, status)
}
addBatchMetric(registry, "vehicle_history_location_batch_flush_total", batchStatus, 1)
addBatchMetric(registry, "vehicle_history_location_rows_total", batchStatus, float64(result.LocationRows))
}
func historyProtocolLabel(protocol envelope.Protocol) string {
protocolLabel := strings.TrimSpace(string(protocol))
if protocolLabel == "" {
return "UNKNOWN"
}
return protocolLabel
}
func setBatchDuration(registry *metrics.Registry, status string, elapsed time.Duration) {
if registry == nil {
return
}
elapsedMS := float64(elapsed.Milliseconds())
labels := metrics.Labels{"status": status}
registry.SetGauge("vehicle_history_batch_flush_duration_ms", labels, elapsedMS)
registry.ObserveHistogram("vehicle_history_batch_flush_duration_ms_histogram", labels, historyBatchFlushDurationBucketsMS, elapsedMS)
}
var historyBatchFlushDurationBucketsMS = []float64{1, 5, 10, 25, 50, 100, 250, 500, 1000, 5000}
func setFallbackDuration(registry *metrics.Registry, status string, elapsed time.Duration) {
if registry == nil {
return
}
elapsedMS := float64(elapsed.Milliseconds())
labels := metrics.Labels{"status": status}
registry.SetGauge("vehicle_history_batch_fallback_duration_ms", labels, elapsedMS)
registry.ObserveHistogram("vehicle_history_batch_fallback_duration_ms_histogram", labels, historyBatchFallbackDurationBucketsMS, elapsedMS)
}
var historyBatchFallbackDurationBucketsMS = []float64{1, 5, 10, 25, 50, 100, 250, 500, 1000, 5000}
var historyWriteE2EDurationBucketsMS = []float64{10, 25, 50, 100, 250, 500, 1000, 2500, 5000, 10000}
var historyWriteE2ERecent = metrics.NewRecentLatencyByKey(512)
func recordHistoryWriteE2EDuration(registry *metrics.Registry, message kafka.Message, env envelope.FrameEnvelope) {
if registry == nil || env.ReceivedAtMS <= 0 {
return
}
elapsed := time.Since(time.UnixMilli(env.ReceivedAtMS)).Milliseconds()
if elapsed < 0 {
elapsed = 0
}
labels := metrics.Labels{"topic": message.Topic}
registry.ObserveHistogram("vehicle_history_write_e2e_duration_ms_histogram", labels, historyWriteE2EDurationBucketsMS, float64(elapsed))
p99, samples := historyWriteE2ERecent.Observe(message.Topic, float64(elapsed))
registry.SetGauge("vehicle_history_write_e2e_recent_p99_ms", labels, p99)
registry.SetGauge("vehicle_history_write_e2e_recent_samples", labels, float64(samples))
metrics.RecordLastActivity(registry, "vehicle_history_last_write_e2e_unix_seconds", labels)
}
func statusFromError(err error) string {
if err != nil {
return "error"
}
return "ok"
}
func setBatchPending(registry *metrics.Registry, messages int, rows int) {
setBatchPendingForWorker(registry, nil, messages, rows)
}
func setBatchPendingForWorker(registry *metrics.Registry, workerLabels metrics.Labels, messages int, rows int) {
if registry == nil {
return
}
registry.SetGauge("vehicle_history_batch_pending_messages", workerLabels, float64(messages))
registry.SetGauge("vehicle_history_batch_pending_rows", workerLabels, float64(rows))
}
func addWriterLagMetric(registry *metrics.Registry, message kafka.Message) {
if registry == nil {
return
}
registry.SetKafkaLag("vehicle_history_kafka_lag", message.Topic, message.Partition, message.Offset, message.HighWaterMark)
}
type config struct {
@@ -89,17 +786,42 @@ type config struct {
TDengineDSN string
TDengineDatabase string
EnsureSchema bool
Workers int
BatchSize int
BatchWait int
RetryAttempts int
RetryDelay time.Duration
}
func (c config) Validate() error {
if len(c.KafkaTopics) == 0 {
return errors.New("KAFKA_TOPICS must include raw topics")
}
for _, topic := range c.KafkaTopics {
if _, ok := topics.ProtocolForKnownRawTopic(topic); !ok {
return fmt.Errorf("history-writer consumes known raw topics only, got %q", topic)
}
}
if c.Workers <= 0 {
return errors.New("HISTORY_WORKERS must be greater than zero")
}
return nil
}
func loadConfig() config {
return config{
KafkaBrokers: splitCSV(env("KAFKA_BROKERS", "127.0.0.1:9092")),
KafkaTopics: splitCSV(env("KAFKA_TOPICS", "vehicle.raw.gb32960.v1,vehicle.raw.jt808.v1,vehicle.raw.yutong-mqtt.v1")),
KafkaTopics: splitCSV(env("KAFKA_TOPICS", strings.Join([]string{topics.RawGB32960, topics.RawJT808, topics.RawYutongMQTT}, ","))),
KafkaGroup: env("KAFKA_GROUP", "go-history-writer"),
TDengineDriver: env("TDENGINE_DRIVER", "taosWS"),
TDengineDSN: env("TDENGINE_DSN", ""),
TDengineDatabase: env("TDENGINE_DATABASE", history.DefaultDatabase),
EnsureSchema: env("TDENGINE_ENSURE_SCHEMA", "true") != "false",
Workers: envInt("HISTORY_WORKERS", 3),
BatchSize: envInt("HISTORY_BATCH_SIZE", 200),
BatchWait: envInt("HISTORY_BATCH_WAIT_MS", 20),
RetryAttempts: envInt("HISTORY_RETRY_ATTEMPTS", 3),
RetryDelay: time.Duration(envInt("HISTORY_RETRY_DELAY_MS", 20)) * time.Millisecond,
}
}
@@ -111,6 +833,18 @@ func env(key string, fallback string) string {
return value
}
func envInt(key string, fallback int) int {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
parsed, err := strconv.Atoi(value)
if err != nil {
return fallback
}
return parsed
}
func splitCSV(value string) []string {
var out []string
for _, item := range strings.Split(value, ",") {

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,695 @@
package main
import (
"context"
"database/sql"
"encoding/csv"
"encoding/json"
"errors"
"flag"
"fmt"
"os"
"strings"
"time"
_ "github.com/go-sql-driver/mysql"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/identity"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats"
)
func main() {
var input string
var dsn string
var legacyTable string
var apply bool
var ensureSchema bool
var reportLimit int
var unresolvedOut string
var conflictsOut string
var syncDataSources bool
var pruneUnmanagedDataSources bool
var pruneDataSourceMinAge time.Duration
var retireStaleUnmanagedDataSources bool
var retireDataSourceMinAge time.Duration
var timeout time.Duration
flag.StringVar(&input, "input", env("IDENTITY_MAPPING_INPUT", ""), "directory that contains 808 plate/phone mapping workbooks")
flag.StringVar(&dsn, "mysql-dsn", env("IDENTITY_MYSQL_DSN", env("MYSQL_DSN", "")), "MySQL DSN; empty means scan only")
flag.StringVar(&legacyTable, "legacy-table", env("VEHICLE_IDENTITY_TABLE", "vehicle_identity_binding"), "legacy VIN/plate binding table used to resolve VIN")
flag.BoolVar(&apply, "apply", false, "write resolved mappings to vehicle and vehicle_identifier")
flag.BoolVar(&ensureSchema, "ensure-schema", false, "create vehicle and vehicle_identifier tables before import")
flag.IntVar(&reportLimit, "report-limit", 50, "max unresolved/conflict items embedded in JSON; use -1 for all")
flag.StringVar(&unresolvedOut, "unresolved-out", "", "optional CSV path for all unresolved mappings")
flag.StringVar(&conflictsOut, "conflicts-out", "", "optional CSV path for all conflict mappings")
flag.BoolVar(&syncDataSources, "sync-data-sources", false, "infer JT808 vehicle_data_source platform names from jt808_registration and vehicle_identifier")
flag.BoolVar(&pruneUnmanagedDataSources, "prune-unmanaged-data-sources", false, "delete stale unclassified vehicle_data_source rows that have no registration evidence and are not referenced by final mileage")
flag.DurationVar(&pruneDataSourceMinAge, "prune-data-source-min-age", time.Hour, "minimum latest_seen/updated age before -prune-unmanaged-data-sources can delete rows")
flag.BoolVar(&retireStaleUnmanagedDataSources, "retire-stale-unmanaged-data-sources", false, "disable stale unclassified vehicle_data_source rows that have no registration evidence but may be referenced by historical mileage")
flag.DurationVar(&retireDataSourceMinAge, "retire-data-source-min-age", 24*time.Hour, "minimum latest_seen/updated age before -retire-stale-unmanaged-data-sources can disable rows")
flag.DurationVar(&timeout, "timeout", 2*time.Minute, "import timeout")
flag.Parse()
ctx, cancel := context.WithTimeout(context.Background(), timeout)
defer cancel()
input = strings.TrimSpace(input)
output := map[string]any{}
var records []identity.MappingRecord
var scan identity.MappingScanReport
if input != "" {
var err error
records, scan, err = identity.ReadMappingDirectory(input)
if err != nil {
fail(err)
}
output["input"] = input
output["scan"] = scan
} else if !syncDataSources && !pruneUnmanagedDataSources && !retireStaleUnmanagedDataSources {
fail(errors.New("-input is required unless -sync-data-sources, -prune-unmanaged-data-sources or -retire-stale-unmanaged-data-sources is set"))
}
if err := validateMappingApplyInput(apply, scan); err != nil {
fail(err)
}
if strings.TrimSpace(dsn) == "" {
if syncDataSources || pruneUnmanagedDataSources || retireStaleUnmanagedDataSources {
fail(errors.New("-mysql-dsn is required when -sync-data-sources, -prune-unmanaged-data-sources or -retire-stale-unmanaged-data-sources is set"))
}
output["mode"] = "scan_only"
writeJSON(output)
return
}
db, err := sql.Open("mysql", dsn)
if err != nil {
fail(err)
}
defer func() { _ = db.Close() }()
if err := db.PingContext(ctx); err != nil {
fail(err)
}
if input != "" && (ensureSchema || apply) {
if err := identity.EnsureVehicleIdentifierSchema(ctx, db); err != nil {
fail(err)
}
}
if input != "" {
report, err := identity.ImportMappingRecords(ctx, db, records, scan, identity.MappingImportOptions{
Apply: apply,
LegacyTable: legacyTable,
ReportItemLimit: reportItemLimit(reportLimit, unresolvedOut, conflictsOut),
})
if err != nil {
fail(err)
}
if strings.TrimSpace(unresolvedOut) != "" {
if err := writeUnresolvedCSV(unresolvedOut, report.UnresolvedItems); err != nil {
fail(err)
}
}
if strings.TrimSpace(conflictsOut) != "" {
if err := writeConflictsCSV(conflictsOut, report.ConflictItems); err != nil {
fail(err)
}
}
output["report"] = report
output["mode"] = map[bool]string{true: "apply", false: "dry_run"}[apply]
}
if syncDataSources {
syncReport, err := syncJT808DataSourcesFromIdentifiers(ctx, db, apply)
if err != nil {
fail(err)
}
output["data_source_sync"] = syncReport
}
if pruneUnmanagedDataSources {
pruneReport, err := pruneUnmanagedDataSourcesWithoutEvidence(ctx, db, apply, pruneDataSourceMinAge)
if err != nil {
fail(err)
}
output["data_source_prune"] = pruneReport
}
if retireStaleUnmanagedDataSources {
retireReport, err := retireStaleUnmanagedDataSourcesWithoutEvidence(ctx, db, apply, retireDataSourceMinAge)
if err != nil {
fail(err)
}
output["data_source_retire"] = retireReport
}
if input == "" {
output["mode"] = maintenanceMode(apply, syncDataSources, pruneUnmanagedDataSources, retireStaleUnmanagedDataSources)
}
writeJSON(output)
}
func maintenanceMode(apply bool, syncDataSources bool, pruneUnmanagedDataSources bool, retireStaleUnmanagedDataSources bool) string {
var parts []string
if syncDataSources {
parts = append(parts, "sync")
}
if pruneUnmanagedDataSources {
parts = append(parts, "prune")
}
if retireStaleUnmanagedDataSources {
parts = append(parts, "retire")
}
if len(parts) == 0 {
return map[bool]string{true: "apply", false: "dry_run"}[apply]
}
parts = append(parts, map[bool]string{true: "apply", false: "dry_run"}[apply])
return strings.Join(parts, "_")
}
func reportItemLimit(limit int, unresolvedOut string, conflictsOut string) int {
if strings.TrimSpace(unresolvedOut) != "" || strings.TrimSpace(conflictsOut) != "" {
return -1
}
return limit
}
func validateMappingApplyInput(apply bool, scan identity.MappingScanReport) error {
if !apply {
return nil
}
return unsupportedMappingFilesError(scan)
}
func unsupportedMappingFilesError(scan identity.MappingScanReport) error {
if scan.UnsupportedFiles <= 0 {
return nil
}
items := make([]string, 0, len(scan.UnsupportedItems))
for _, item := range scan.UnsupportedItems {
path := strings.TrimSpace(item.File)
if path == "" {
path = strings.TrimSpace(item.Ext)
}
if path != "" {
items = append(items, path)
}
if len(items) >= 3 {
break
}
}
suffix := ""
if len(items) > 0 {
suffix = ": " + strings.Join(items, ", ")
}
return fmt.Errorf("identity mapping import has %d unsupported workbook(s); convert .xls/.xlsb to .xlsx before -apply%s", scan.UnsupportedFiles, suffix)
}
func env(key string, fallback string) string {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
return value
}
func writeJSON(value any) {
encoder := json.NewEncoder(os.Stdout)
encoder.SetIndent("", " ")
if err := encoder.Encode(value); err != nil {
fail(err)
}
}
func writeUnresolvedCSV(path string, items []identity.MappingRecord) error {
file, err := os.Create(path)
if err != nil {
return err
}
defer file.Close()
writer := csv.NewWriter(file)
defer writer.Flush()
if err := writer.Write([]string{
"file", "sheet", "row", "source_code", "source_name", "protocol",
"identifier_type", "identifier_value", "raw_value", "plate", "oem", "reason",
}); err != nil {
return err
}
for _, item := range items {
if err := writer.Write([]string{
item.File,
item.Sheet,
fmt.Sprint(item.Row),
item.SourceCode,
item.SourceName,
item.Protocol,
item.IdentifierType,
item.IdentifierValue,
item.RawValue,
item.Plate,
item.OEM,
"vin_not_found_in_legacy_binding",
}); err != nil {
return err
}
}
return writer.Error()
}
func writeConflictsCSV(path string, items []identity.MappingConflict) error {
file, err := os.Create(path)
if err != nil {
return err
}
defer file.Close()
writer := csv.NewWriter(file)
defer writer.Flush()
if err := writer.Write([]string{
"file", "sheet", "row", "source_code", "source_name", "protocol",
"identifier_type", "identifier_value", "raw_value", "plate", "oem",
"existing_vin", "new_vin", "reason",
}); err != nil {
return err
}
for _, item := range items {
record := item.Record
if err := writer.Write([]string{
record.File,
record.Sheet,
fmt.Sprint(record.Row),
record.SourceCode,
record.SourceName,
record.Protocol,
record.IdentifierType,
record.IdentifierValue,
record.RawValue,
record.Plate,
record.OEM,
item.ExistingVIN,
item.NewVIN,
item.Reason,
}); err != nil {
return err
}
}
return writer.Error()
}
type dataSourceSyncReport struct {
Apply bool `json:"apply"`
CandidateSources int64 `json:"candidate_sources"`
SkippedSources int64 `json:"skipped_sources"`
ConflictingSources int64 `json:"conflicting_sources"`
PlatformKindCandidates int64 `json:"platform_kind_candidates"`
PlatformKindClassified int64 `json:"platform_kind_classified,omitempty"`
ReprojectDailyMileageTargets int64 `json:"reproject_daily_mileage_targets,omitempty"`
ReprojectedDailyMileageTargets int64 `json:"reprojected_daily_mileage_targets,omitempty"`
Synced int64 `json:"synced,omitempty"`
}
type dailyMileageProjectionTarget struct {
VIN string
StatDate string
Protocol envelope.Protocol
}
type dataSourcePruneReport struct {
Apply bool `json:"apply"`
MinAgeSeconds int64 `json:"min_age_seconds"`
CandidateSources int64 `json:"candidate_sources"`
Pruned int64 `json:"pruned,omitempty"`
}
type dataSourceRetireReport struct {
Apply bool `json:"apply"`
MinAgeSeconds int64 `json:"min_age_seconds"`
CandidateSources int64 `json:"candidate_sources"`
Retired int64 `json:"retired,omitempty"`
}
func syncJT808DataSourcesFromIdentifiers(ctx context.Context, db *sql.DB, apply bool) (dataSourceSyncReport, error) {
report := dataSourceSyncReport{Apply: apply}
var err error
report.CandidateSources, err = queryInt64(ctx, db, syncJT808DataSourcesCandidateCountSQL)
if err != nil {
return report, err
}
report.SkippedSources, err = queryInt64(ctx, db, syncJT808DataSourcesSkippedCountSQL)
if err != nil {
return report, err
}
report.ConflictingSources, err = queryInt64(ctx, db, syncJT808DataSourcesConflictCountSQL)
if err != nil {
return report, err
}
if apply {
result, err := db.ExecContext(ctx, syncJT808DataSourcesSQL)
if err != nil {
return report, err
}
rowsAffected, err := result.RowsAffected()
if err == nil {
report.Synced = rowsAffected
}
}
report.PlatformKindCandidates, err = queryInt64(ctx, db, classifyConfiguredDataSourcesCountSQL)
if err != nil {
return report, err
}
if apply {
result, err := db.ExecContext(ctx, classifyConfiguredDataSourcesSQL)
if err != nil {
return report, err
}
rowsAffected, err := result.RowsAffected()
if err == nil {
report.PlatformKindClassified = rowsAffected
}
}
targets, err := queryDailyMileageProjectionTargets(ctx, db)
if err != nil {
return report, err
}
report.ReprojectDailyMileageTargets = int64(len(targets))
if apply {
for _, target := range targets {
if err := stats.ProjectDailyMileage(ctx, db, target.VIN, target.StatDate, target.Protocol); err != nil {
return report, err
}
report.ReprojectedDailyMileageTargets++
}
}
return report, nil
}
func queryDailyMileageProjectionTargets(ctx context.Context, db *sql.DB) ([]dailyMileageProjectionTarget, error) {
rows, err := db.QueryContext(ctx, reprojectSelectableDailyMileageSourcesSQL)
if err != nil {
return nil, err
}
defer rows.Close()
var targets []dailyMileageProjectionTarget
for rows.Next() {
var target dailyMileageProjectionTarget
var protocol string
if err := rows.Scan(&target.VIN, &target.StatDate, &protocol); err != nil {
return nil, err
}
target.Protocol = envelope.Protocol(strings.TrimSpace(protocol))
if strings.TrimSpace(target.VIN) != "" && strings.TrimSpace(target.StatDate) != "" && strings.TrimSpace(string(target.Protocol)) != "" {
targets = append(targets, target)
}
}
return targets, rows.Err()
}
func pruneUnmanagedDataSourcesWithoutEvidence(ctx context.Context, db *sql.DB, apply bool, minAge time.Duration) (dataSourcePruneReport, error) {
if minAge < 0 {
minAge = 0
}
minAgeSeconds := int64(minAge.Seconds())
report := dataSourcePruneReport{Apply: apply, MinAgeSeconds: minAgeSeconds}
var err error
report.CandidateSources, err = queryInt64WithArgs(ctx, db, pruneUnmanagedDataSourcesCountSQL, minAgeSeconds)
if err != nil {
return report, err
}
if !apply {
return report, nil
}
result, err := db.ExecContext(ctx, pruneUnmanagedDataSourcesSQL, minAgeSeconds)
if err != nil {
return report, err
}
rowsAffected, err := result.RowsAffected()
if err == nil {
report.Pruned = rowsAffected
}
return report, nil
}
func retireStaleUnmanagedDataSourcesWithoutEvidence(ctx context.Context, db *sql.DB, apply bool, minAge time.Duration) (dataSourceRetireReport, error) {
if minAge < 0 {
minAge = 0
}
minAgeSeconds := int64(minAge.Seconds())
report := dataSourceRetireReport{Apply: apply, MinAgeSeconds: minAgeSeconds}
var err error
report.CandidateSources, err = queryInt64WithArgs(ctx, db, retireStaleUnmanagedDataSourcesCountSQL, minAgeSeconds)
if err != nil {
return report, err
}
if !apply {
return report, nil
}
result, err := db.ExecContext(ctx, retireStaleUnmanagedDataSourcesSQL, minAgeSeconds)
if err != nil {
return report, err
}
rowsAffected, err := result.RowsAffected()
if err == nil {
report.Retired = rowsAffected
}
return report, nil
}
func queryInt64(ctx context.Context, db *sql.DB, query string) (int64, error) {
return queryInt64WithArgs(ctx, db, query)
}
func queryInt64WithArgs(ctx context.Context, db *sql.DB, query string, args ...any) (int64, error) {
var count int64
err := db.QueryRowContext(ctx, query, args...).Scan(&count)
return count, err
}
const jt808DataSourceInferenceSQL = `
SELECT
'JT808' AS protocol,
TRIM(r.source_ip) AS source_ip,
MAX(TRIM(r.source_endpoint)) AS latest_source_endpoint,
MIN(COALESCE(NULLIF(TRIM(vi.oem), ''), NULLIF(TRIM(vi.source_code), ''))) AS platform_name,
MIN(NULLIF(TRIM(vi.source_code), '')) AS source_code,
MIN(COALESCE(r.first_registered_at, r.latest_registered_at, r.latest_authenticated_at, r.latest_seen_at, r.updated_at, CURRENT_TIMESTAMP)) AS first_seen_at,
MAX(COALESCE(r.latest_seen_at, r.latest_authenticated_at, r.latest_registered_at, r.updated_at, CURRENT_TIMESTAMP)) AS latest_seen_at,
COUNT(DISTINCT COALESCE(NULLIF(TRIM(vi.oem), ''), NULLIF(TRIM(vi.source_code), ''))) AS platform_count,
COUNT(DISTINCT NULLIF(TRIM(vi.source_code), '')) AS source_code_count
FROM jt808_registration r
JOIN vehicle_identifier vi
ON vi.protocol = 'JT808'
AND vi.identifier_type = 'JT808_PHONE'
AND vi.identifier_value = r.phone
AND vi.enabled = 1
WHERE r.source_endpoint IS NOT NULL
AND TRIM(r.source_endpoint) <> ''
AND r.source_ip IS NOT NULL
AND TRIM(r.source_ip) <> ''
GROUP BY TRIM(r.source_ip)
`
const jt808DataSourceCandidateWhereSQL = `inferred.source_ip IS NOT NULL
AND inferred.source_ip <> ''
AND inferred.platform_name IS NOT NULL
AND inferred.platform_name <> ''
AND inferred.platform_count = 1
AND inferred.source_code_count = 1`
const jt808DataSourceNeedsSyncWhereSQL = `(ds.id IS NULL
OR ds.platform_name IS NULL OR TRIM(ds.platform_name) = ''
OR ds.source_code IS NULL OR TRIM(ds.source_code) = ''
OR ds.source_kind IS NULL OR TRIM(ds.source_kind) = '' OR ds.source_kind = 'UNKNOWN'
OR (ds.enabled = 0 AND (ds.remark LIKE 'auto-retired:%' OR ds.remark = 'auto-reenabled: source evidence restored')))`
const syncJT808DataSourcesSQL = `
INSERT INTO vehicle_data_source
(protocol, source_ip, latest_source_endpoint, platform_name, source_code, source_kind, first_seen_at, latest_seen_at)
SELECT
inferred.protocol,
inferred.source_ip,
inferred.latest_source_endpoint,
inferred.platform_name,
inferred.source_code,
'PLATFORM',
inferred.first_seen_at,
inferred.latest_seen_at
FROM (` + jt808DataSourceInferenceSQL + `) inferred
LEFT JOIN vehicle_data_source ds
ON ds.protocol = inferred.protocol
AND ds.source_ip = inferred.source_ip
WHERE ` + jt808DataSourceCandidateWhereSQL + `
AND ` + jt808DataSourceNeedsSyncWhereSQL + `
ON DUPLICATE KEY UPDATE
latest_source_endpoint = VALUES(latest_source_endpoint),
platform_name = CASE
WHEN vehicle_data_source.platform_name IS NULL OR TRIM(vehicle_data_source.platform_name) = ''
THEN VALUES(platform_name)
ELSE vehicle_data_source.platform_name
END,
source_code = CASE
WHEN vehicle_data_source.source_code IS NULL OR TRIM(vehicle_data_source.source_code) = ''
THEN VALUES(source_code)
ELSE vehicle_data_source.source_code
END,
source_kind = CASE
WHEN vehicle_data_source.source_kind IS NULL OR TRIM(vehicle_data_source.source_kind) = '' OR vehicle_data_source.source_kind = 'UNKNOWN'
THEN VALUES(source_kind)
ELSE vehicle_data_source.source_kind
END,
enabled = CASE
WHEN vehicle_data_source.enabled = 0
AND (vehicle_data_source.remark LIKE 'auto-retired:%' OR vehicle_data_source.remark = 'auto-reenabled: source evidence restored')
AND (
(VALUES(platform_name) IS NOT NULL AND TRIM(VALUES(platform_name)) <> '')
OR (VALUES(source_code) IS NOT NULL AND TRIM(VALUES(source_code)) <> '')
OR VALUES(source_kind) <> 'UNKNOWN'
)
THEN 1
ELSE vehicle_data_source.enabled
END,
remark = CASE
WHEN vehicle_data_source.enabled = 0
AND (vehicle_data_source.remark LIKE 'auto-retired:%' OR vehicle_data_source.remark = 'auto-reenabled: source evidence restored')
AND (
(VALUES(platform_name) IS NOT NULL AND TRIM(VALUES(platform_name)) <> '')
OR (VALUES(source_code) IS NOT NULL AND TRIM(VALUES(source_code)) <> '')
OR VALUES(source_kind) <> 'UNKNOWN'
)
THEN 'auto-reenabled: source evidence restored'
ELSE vehicle_data_source.remark
END,
first_seen_at = COALESCE(vehicle_data_source.first_seen_at, VALUES(first_seen_at)),
latest_seen_at = GREATEST(COALESCE(vehicle_data_source.latest_seen_at, VALUES(latest_seen_at)), VALUES(latest_seen_at)),
updated_at = CURRENT_TIMESTAMP
`
const reprojectSelectableDailyMileageSourcesSQL = `
SELECT DISTINCT s.vin, s.stat_date, s.protocol
FROM vehicle_daily_mileage_source s
LEFT JOIN vehicle_data_source ds
ON ds.protocol = s.protocol AND ds.source_ip = s.source_ip
LEFT JOIN vehicle_daily_mileage m
ON m.vin = s.vin AND m.stat_date = s.stat_date AND m.protocol = s.protocol
WHERE s.protocol = 'JT808'
AND s.quality_status = 'OK'
AND (ds.id IS NULL OR ds.enabled = 1 OR (
COALESCE(NULLIF(TRIM(ds.source_kind), ''), 'UNKNOWN') = 'UNKNOWN'
AND (ds.source_code IS NULL OR TRIM(ds.source_code) = '')
AND (ds.platform_name IS NULL OR TRIM(ds.platform_name) = '')
))
AND (
m.vin IS NULL
OR NOT EXISTS (
SELECT 1
FROM vehicle_daily_mileage_source selected
WHERE selected.vin = s.vin
AND selected.stat_date = s.stat_date
AND selected.protocol = s.protocol
AND selected.is_selected = 1
)
)
ORDER BY s.stat_date DESC, s.vin ASC
`
const syncJT808DataSourcesCandidateCountSQL = `
SELECT COUNT(*)
FROM (` + jt808DataSourceInferenceSQL + `) inferred
LEFT JOIN vehicle_data_source ds
ON ds.protocol = inferred.protocol
AND ds.source_ip = inferred.source_ip
WHERE ` + jt808DataSourceCandidateWhereSQL + `
AND ` + jt808DataSourceNeedsSyncWhereSQL
const syncJT808DataSourcesSkippedCountSQL = `
SELECT COUNT(*)
FROM (` + jt808DataSourceInferenceSQL + `) inferred
WHERE inferred.source_ip IS NOT NULL
AND inferred.source_ip <> ''
AND inferred.platform_name IS NOT NULL
AND inferred.platform_name <> ''
AND NOT (` + jt808DataSourceCandidateWhereSQL + `)`
const syncJT808DataSourcesConflictCountSQL = `
SELECT COUNT(*)
FROM (` + jt808DataSourceInferenceSQL + `) inferred
JOIN vehicle_data_source ds
ON ds.protocol = inferred.protocol
AND ds.source_ip = inferred.source_ip
WHERE ` + jt808DataSourceCandidateWhereSQL + `
AND ds.source_code IS NOT NULL
AND TRIM(ds.source_code) <> ''
AND ds.source_code <> inferred.source_code`
const classifyConfiguredDataSourcesWhereSQL = `
(source_kind IS NULL OR TRIM(source_kind) = '' OR source_kind = 'UNKNOWN')
AND (
(source_code IS NOT NULL AND TRIM(source_code) <> '')
OR (platform_name IS NOT NULL AND TRIM(platform_name) <> '')
)`
const classifyConfiguredDataSourcesCountSQL = `
SELECT COUNT(*)
FROM vehicle_data_source
WHERE ` + classifyConfiguredDataSourcesWhereSQL
const classifyConfiguredDataSourcesSQL = `
UPDATE vehicle_data_source
SET source_kind = 'PLATFORM',
updated_at = CURRENT_TIMESTAMP
WHERE ` + classifyConfiguredDataSourcesWhereSQL
const pruneUnmanagedDataSourcesWhereSQL = `
(ds.platform_name IS NULL OR TRIM(ds.platform_name) = '')
AND (ds.source_code IS NULL OR TRIM(ds.source_code) = '')
AND (ds.source_kind IS NULL OR TRIM(ds.source_kind) = '' OR ds.source_kind = 'UNKNOWN')
AND TIMESTAMPDIFF(SECOND, COALESCE(ds.latest_seen_at, ds.updated_at, ds.created_at), CURRENT_TIMESTAMP) >= ?
AND NOT EXISTS (
SELECT 1
FROM vehicle_daily_mileage m
WHERE m.source_id = ds.id
LIMIT 1
)
AND NOT EXISTS (
SELECT 1
FROM jt808_registration r
WHERE ds.protocol = 'JT808'
AND r.source_ip = ds.source_ip
LIMIT 1
)`
const pruneUnmanagedDataSourcesCountSQL = `
SELECT COUNT(*)
FROM vehicle_data_source ds
WHERE ` + pruneUnmanagedDataSourcesWhereSQL
const pruneUnmanagedDataSourcesSQL = `
DELETE ds
FROM vehicle_data_source ds
WHERE ` + pruneUnmanagedDataSourcesWhereSQL
const retireStaleUnmanagedDataSourcesWhereSQL = `
ds.enabled = 1
AND (ds.platform_name IS NULL OR TRIM(ds.platform_name) = '')
AND (ds.source_code IS NULL OR TRIM(ds.source_code) = '')
AND (ds.source_kind IS NULL OR TRIM(ds.source_kind) = '' OR ds.source_kind = 'UNKNOWN')
AND TIMESTAMPDIFF(SECOND, COALESCE(ds.latest_seen_at, ds.updated_at, ds.created_at), CURRENT_TIMESTAMP) >= ?
AND NOT EXISTS (
SELECT 1
FROM jt808_registration r
WHERE ds.protocol = 'JT808'
AND r.source_ip = ds.source_ip
LIMIT 1
)`
const retireStaleUnmanagedDataSourcesCountSQL = `
SELECT COUNT(*)
FROM vehicle_data_source ds
WHERE ` + retireStaleUnmanagedDataSourcesWhereSQL
const retireStaleUnmanagedDataSourcesSQL = `
UPDATE vehicle_data_source ds
SET enabled = 0,
remark = CASE
WHEN remark IS NULL OR TRIM(remark) = ''
THEN 'auto-retired: stale unmanaged source without registration evidence'
ELSE remark
END,
updated_at = CURRENT_TIMESTAMP
WHERE ` + retireStaleUnmanagedDataSourcesWhereSQL
func fail(err error) {
_, _ = fmt.Fprintln(os.Stderr, err)
os.Exit(1)
}

View File

@@ -0,0 +1,426 @@
package main
import (
"context"
"database/sql/driver"
"os"
"path/filepath"
"strings"
"testing"
"time"
"github.com/DATA-DOG/go-sqlmock"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/identity"
)
func TestReportItemLimitExpandsWhenCSVOutputIsRequested(t *testing.T) {
if got := reportItemLimit(50, "", ""); got != 50 {
t.Fatalf("reportItemLimit without csv = %d, want 50", got)
}
if got := reportItemLimit(50, "unresolved.csv", ""); got != -1 {
t.Fatalf("reportItemLimit with unresolved csv = %d, want -1", got)
}
if got := reportItemLimit(50, "", "conflicts.csv"); got != -1 {
t.Fatalf("reportItemLimit with conflicts csv = %d, want -1", got)
}
}
func TestUnsupportedMappingFilesError(t *testing.T) {
err := unsupportedMappingFilesError(identity.MappingScanReport{})
if err != nil {
t.Fatalf("unsupportedMappingFilesError(empty) = %v", err)
}
err = unsupportedMappingFilesError(identity.MappingScanReport{
UnsupportedFiles: 2,
UnsupportedItems: []identity.MappingUnsupportedFileReport{
{File: "G7s/legacy.xls", Ext: ".xls"},
{File: "信达/legacy.xlsb", Ext: ".xlsb"},
},
})
if err == nil {
t.Fatal("unsupportedMappingFilesError() nil, want error")
}
text := err.Error()
for _, want := range []string{"2 unsupported workbook", "G7s/legacy.xls", "信达/legacy.xlsb", "before -apply"} {
if !strings.Contains(text, want) {
t.Fatalf("error missing %q: %s", want, text)
}
}
}
func TestValidateMappingApplyInputAllowsDryRunButBlocksApply(t *testing.T) {
scan := identity.MappingScanReport{
UnsupportedFiles: 1,
UnsupportedItems: []identity.MappingUnsupportedFileReport{
{File: "G7s/legacy.xls", Ext: ".xls"},
},
}
if err := validateMappingApplyInput(false, scan); err != nil {
t.Fatalf("validateMappingApplyInput(dry-run) = %v", err)
}
if err := validateMappingApplyInput(true, scan); err == nil {
t.Fatal("validateMappingApplyInput(apply) nil, want error")
}
}
func TestWriteUnresolvedCSV(t *testing.T) {
path := filepath.Join(t.TempDir(), "unresolved.csv")
err := writeUnresolvedCSV(path, []identity.MappingRecord{
{
File: "G7s/example.xlsx",
Sheet: "Sheet1",
Row: 2,
SourceCode: "g7s",
SourceName: "G7s",
Protocol: "JT808",
IdentifierType: identity.IdentifierTypeJT808Phone,
IdentifierValue: "13307795425",
RawValue: "013307795425",
Plate: "粤AG18312",
OEM: "G7s",
},
})
if err != nil {
t.Fatalf("writeUnresolvedCSV() error = %v", err)
}
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("ReadFile() error = %v", err)
}
text := string(data)
for _, want := range []string{"identifier_type", "JT808_PHONE", "13307795425", "vin_not_found_in_legacy_binding"} {
if !strings.Contains(text, want) {
t.Fatalf("csv missing %q:\n%s", want, text)
}
}
}
func TestWriteConflictsCSV(t *testing.T) {
path := filepath.Join(t.TempDir(), "conflicts.csv")
err := writeConflictsCSV(path, []identity.MappingConflict{
{
Record: identity.MappingRecord{
File: "source.xlsx",
Sheet: "Sheet1",
Row: 3,
SourceCode: "xinda",
Protocol: "JT808",
IdentifierType: identity.IdentifierTypePlate,
IdentifierValue: "粤AG18312",
Plate: "粤AG18312",
},
ExistingVIN: "VIN001",
NewVIN: "VIN002",
Reason: "identifier already points to another vin",
},
})
if err != nil {
t.Fatalf("writeConflictsCSV() error = %v", err)
}
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("ReadFile() error = %v", err)
}
text := string(data)
for _, want := range []string{"existing_vin", "VIN001", "VIN002", "identifier already points to another vin"} {
if !strings.Contains(text, want) {
t.Fatalf("csv missing %q:\n%s", want, text)
}
}
}
func TestSyncJT808DataSourcesFromIdentifiersPreservesManualPlatformNames(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
mock.ExpectQuery("SELECT COUNT\\(\\*\\)").
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(3))
mock.ExpectQuery("SELECT COUNT\\(\\*\\)").
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(1))
mock.ExpectQuery("SELECT COUNT\\(\\*\\)").
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(2))
mock.ExpectExec("INSERT INTO vehicle_data_source").
WillReturnResult(driver.RowsAffected(3))
mock.ExpectQuery("SELECT COUNT\\(\\*\\)").
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(5))
mock.ExpectExec("UPDATE vehicle_data_source").
WillReturnResult(driver.RowsAffected(5))
mock.ExpectQuery("SELECT DISTINCT s.vin, s.stat_date, s.protocol").
WillReturnRows(sqlmock.NewRows([]string{"vin", "stat_date", "protocol"}).
AddRow("LNXNEGRRXSR319449", "2026-07-13", "JT808"))
mock.ExpectBegin()
mock.ExpectExec("INSERT INTO vehicle_daily_mileage").
WillReturnResult(driver.RowsAffected(1))
mock.ExpectExec("UPDATE vehicle_daily_mileage_source s").
WillReturnResult(driver.RowsAffected(1))
mock.ExpectExec("DELETE FROM vehicle_daily_mileage").
WillReturnResult(driver.RowsAffected(0))
mock.ExpectCommit()
report, err := syncJT808DataSourcesFromIdentifiers(context.Background(), db, true)
if err != nil {
t.Fatalf("syncJT808DataSourcesFromIdentifiers() error = %v", err)
}
if !report.Apply || report.CandidateSources != 3 || report.SkippedSources != 1 || report.ConflictingSources != 2 || report.Synced != 3 || report.PlatformKindCandidates != 5 || report.PlatformKindClassified != 5 || report.ReprojectDailyMileageTargets != 1 || report.ReprojectedDailyMileageTargets != 1 {
t.Fatalf("report = %#v", report)
}
for _, want := range []string{
"COUNT(DISTINCT",
"vehicle_identifier vi",
"TRIM(r.source_ip) AS source_ip",
"GROUP BY TRIM(r.source_ip)",
"inferred.source_code",
"'PLATFORM'",
"inferred.source_code_count = 1",
"vehicle_data_source.source_code IS NULL OR TRIM(vehicle_data_source.source_code) = ''",
"ELSE vehicle_data_source.source_code",
"vehicle_data_source.source_kind IS NULL OR TRIM(vehicle_data_source.source_kind) = '' OR vehicle_data_source.source_kind = 'UNKNOWN'",
"ELSE vehicle_data_source.source_kind",
"vehicle_data_source.enabled = 0",
"vehicle_data_source.remark LIKE 'auto-retired:%'",
"vehicle_data_source.remark = 'auto-reenabled: source evidence restored'",
"THEN 'auto-reenabled: source evidence restored'",
"THEN 1",
"LEFT JOIN vehicle_data_source ds",
"ds.id IS NULL",
} {
if !strings.Contains(syncJT808DataSourcesSQL, want) {
t.Fatalf("sync sql missing %q:\n%s", want, syncJT808DataSourcesSQL)
}
}
if strings.Index(syncJT808DataSourcesSQL, "enabled = CASE") < 0 ||
strings.Index(syncJT808DataSourcesSQL, "remark = CASE") < 0 ||
strings.Index(syncJT808DataSourcesSQL, "enabled = CASE") > strings.Index(syncJT808DataSourcesSQL, "remark = CASE") {
t.Fatalf("sync sql must restore enabled before updating remark because MySQL evaluates assignments in order:\n%s", syncJT808DataSourcesSQL)
}
for _, want := range []string{
"LEFT JOIN vehicle_data_source ds",
"ds.id IS NULL",
"ds.source_code IS NULL OR TRIM(ds.source_code) = ''",
"ds.enabled = 0 AND (ds.remark LIKE 'auto-retired:%'",
"ds.remark = 'auto-reenabled: source evidence restored'",
} {
if !strings.Contains(syncJT808DataSourcesCandidateCountSQL, want) {
t.Fatalf("candidate count sql missing %q:\n%s", want, syncJT808DataSourcesCandidateCountSQL)
}
}
for _, want := range []string{
"JOIN vehicle_data_source ds",
"ds.source_code <> inferred.source_code",
} {
if !strings.Contains(syncJT808DataSourcesConflictCountSQL, want) {
t.Fatalf("conflict count sql missing %q:\n%s", want, syncJT808DataSourcesConflictCountSQL)
}
}
for _, want := range []string{
"vehicle_data_source.platform_name IS NULL OR TRIM(vehicle_data_source.platform_name) = ''",
"ELSE vehicle_data_source.platform_name",
} {
if !strings.Contains(syncJT808DataSourcesSQL, want) {
t.Fatalf("sync sql missing %q:\n%s", want, syncJT808DataSourcesSQL)
}
}
for _, want := range []string{
"source_kind IS NULL OR TRIM(source_kind) = '' OR source_kind = 'UNKNOWN'",
"source_code IS NOT NULL AND TRIM(source_code) <> ''",
"platform_name IS NOT NULL AND TRIM(platform_name) <> ''",
"SET source_kind = 'PLATFORM'",
} {
if !strings.Contains(classifyConfiguredDataSourcesSQL, want) && !strings.Contains(classifyConfiguredDataSourcesCountSQL, want) {
t.Fatalf("configured-source classification sql missing %q:\n%s\n%s", want, classifyConfiguredDataSourcesSQL, classifyConfiguredDataSourcesCountSQL)
}
}
for _, want := range []string{
"SELECT DISTINCT s.vin, s.stat_date, s.protocol",
"vehicle_daily_mileage_source s",
"LEFT JOIN vehicle_data_source ds",
"LEFT JOIN vehicle_daily_mileage m",
"s.protocol = 'JT808'",
"s.quality_status = 'OK'",
"ds.enabled = 1",
"m.vin IS NULL",
"selected.is_selected = 1",
} {
if !strings.Contains(reprojectSelectableDailyMileageSourcesSQL, want) {
t.Fatalf("reproject sql missing %q:\n%s", want, reprojectSelectableDailyMileageSourcesSQL)
}
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}
func TestSyncJT808DataSourcesDryRunDoesNotWrite(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
mock.ExpectQuery("SELECT COUNT\\(\\*\\)").
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(4))
mock.ExpectQuery("SELECT COUNT\\(\\*\\)").
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(2))
mock.ExpectQuery("SELECT COUNT\\(\\*\\)").
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(1))
mock.ExpectQuery("SELECT COUNT\\(\\*\\)").
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(6))
mock.ExpectQuery("SELECT DISTINCT s.vin, s.stat_date, s.protocol").
WillReturnRows(sqlmock.NewRows([]string{"vin", "stat_date", "protocol"}).
AddRow("LNXNEGRRXSR319449", "2026-07-13", "JT808"))
report, err := syncJT808DataSourcesFromIdentifiers(context.Background(), db, false)
if err != nil {
t.Fatalf("syncJT808DataSourcesFromIdentifiers() error = %v", err)
}
if report.Apply || report.CandidateSources != 4 || report.SkippedSources != 2 || report.ConflictingSources != 1 || report.PlatformKindCandidates != 6 || report.PlatformKindClassified != 0 || report.Synced != 0 || report.ReprojectDailyMileageTargets != 1 || report.ReprojectedDailyMileageTargets != 0 {
t.Fatalf("report = %#v", report)
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}
func TestMaintenanceModeNamesAllSourceMaintenanceSteps(t *testing.T) {
tests := []struct {
name string
apply bool
sync bool
prune bool
retire bool
want string
}{
{name: "dry run", want: "dry_run"},
{name: "apply", apply: true, want: "apply"},
{name: "sync dry run", sync: true, want: "sync_dry_run"},
{name: "sync prune retire apply", apply: true, sync: true, prune: true, retire: true, want: "sync_prune_retire_apply"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
if got := maintenanceMode(tt.apply, tt.sync, tt.prune, tt.retire); got != tt.want {
t.Fatalf("maintenanceMode() = %q, want %q", got, tt.want)
}
})
}
}
func TestPruneUnmanagedDataSourcesDryRunDoesNotDelete(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
mock.ExpectQuery("SELECT COUNT\\(\\*\\)").
WithArgs(int64(3600)).
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(339))
report, err := pruneUnmanagedDataSourcesWithoutEvidence(context.Background(), db, false, time.Hour)
if err != nil {
t.Fatalf("pruneUnmanagedDataSourcesWithoutEvidence() error = %v", err)
}
if report.Apply || report.MinAgeSeconds != 3600 || report.CandidateSources != 339 || report.Pruned != 0 {
t.Fatalf("report = %#v", report)
}
for _, want := range []string{
"vehicle_data_source ds",
"vehicle_daily_mileage m",
"m.source_id = ds.id",
"jt808_registration r",
"r.source_ip = ds.source_ip",
"ds.source_kind IS NULL OR TRIM(ds.source_kind) = '' OR ds.source_kind = 'UNKNOWN'",
"TIMESTAMPDIFF(SECOND",
} {
if !strings.Contains(pruneUnmanagedDataSourcesCountSQL, want) {
t.Fatalf("prune count sql missing %q:\n%s", want, pruneUnmanagedDataSourcesCountSQL)
}
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}
func TestRetireStaleUnmanagedDataSourcesDryRunDoesNotDisable(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
mock.ExpectQuery("SELECT COUNT\\(\\*\\)").
WithArgs(int64(86400)).
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(83))
report, err := retireStaleUnmanagedDataSourcesWithoutEvidence(context.Background(), db, false, 24*time.Hour)
if err != nil {
t.Fatalf("retireStaleUnmanagedDataSourcesWithoutEvidence() error = %v", err)
}
if report.Apply || report.MinAgeSeconds != 86400 || report.CandidateSources != 83 || report.Retired != 0 {
t.Fatalf("report = %#v", report)
}
for _, want := range []string{
"vehicle_data_source ds",
"ds.enabled = 1",
"jt808_registration r",
"r.source_ip = ds.source_ip",
"ds.source_kind IS NULL OR TRIM(ds.source_kind) = '' OR ds.source_kind = 'UNKNOWN'",
"TIMESTAMPDIFF(SECOND",
} {
if !strings.Contains(retireStaleUnmanagedDataSourcesCountSQL, want) {
t.Fatalf("retire count sql missing %q:\n%s", want, retireStaleUnmanagedDataSourcesCountSQL)
}
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}
func TestRetireStaleUnmanagedDataSourcesApplyDisablesOnlyCandidates(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
mock.ExpectQuery("SELECT COUNT\\(\\*\\)").
WithArgs(int64(7200)).
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(7))
mock.ExpectExec("UPDATE vehicle_data_source ds").
WithArgs(int64(7200)).
WillReturnResult(driver.RowsAffected(7))
report, err := retireStaleUnmanagedDataSourcesWithoutEvidence(context.Background(), db, true, 2*time.Hour)
if err != nil {
t.Fatalf("retireStaleUnmanagedDataSourcesWithoutEvidence() error = %v", err)
}
if !report.Apply || report.MinAgeSeconds != 7200 || report.CandidateSources != 7 || report.Retired != 7 {
t.Fatalf("report = %#v", report)
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}
func TestPruneUnmanagedDataSourcesApplyDeletesOnlyCandidates(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
mock.ExpectQuery("SELECT COUNT\\(\\*\\)").
WithArgs(int64(1800)).
WillReturnRows(sqlmock.NewRows([]string{"count"}).AddRow(12))
mock.ExpectExec("DELETE ds\\s+FROM vehicle_data_source ds").
WithArgs(int64(1800)).
WillReturnResult(driver.RowsAffected(12))
report, err := pruneUnmanagedDataSourcesWithoutEvidence(context.Background(), db, true, 30*time.Minute)
if err != nil {
t.Fatalf("pruneUnmanagedDataSourcesWithoutEvidence() error = %v", err)
}
if !report.Apply || report.MinAgeSeconds != 1800 || report.CandidateSources != 12 || report.Pruned != 12 {
t.Fatalf("report = %#v", report)
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}

View File

@@ -0,0 +1,512 @@
package main
import (
"context"
"database/sql"
"encoding/json"
"errors"
"fmt"
"log/slog"
"os"
"os/signal"
"strconv"
"strings"
"sync"
"syscall"
"time"
_ "github.com/go-sql-driver/mysql"
"github.com/segmentio/kafka-go"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/health"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/identity"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/observability"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/topics"
)
const identityBatchOperationTimeout = 30 * time.Second
func main() {
logger := observability.NewLogger("vehicle-identity-writer")
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
cfg := loadConfig()
if err := cfg.Validate(); err != nil {
logger.Error("invalid identity writer config", "error", err)
os.Exit(1)
}
db, err := sql.Open("mysql", cfg.MySQLDSN)
if err != nil {
logger.Error("mysql open failed", "error", err)
os.Exit(1)
}
defer db.Close()
db.SetMaxOpenConns(cfg.MySQLMaxOpenConns)
db.SetMaxIdleConns(cfg.MySQLMaxIdleConns)
db.SetConnMaxLifetime(cfg.MySQLConnMaxLifetime)
if err := db.PingContext(ctx); err != nil {
logger.Error("mysql ping failed", "error", err)
os.Exit(1)
}
if cfg.EnsureSchema {
if err := identity.EnsureJT808RegistrationSchema(ctx, db); err != nil {
logger.Error("jt808 registration schema bootstrap failed", "error", err)
os.Exit(1)
}
}
registry := metrics.NewRegistry()
metrics.RegisterKafkaConsumerInfo(registry, "vehicle-identity-writer", cfg.KafkaGroup, []string{cfg.KafkaTopic})
registry.SetGauge("vehicle_identity_writer_config", metrics.Labels{"setting": "batch_size"}, float64(cfg.BatchSize))
registry.SetGauge("vehicle_identity_writer_config", metrics.Labels{"setting": "batch_wait_ms"}, float64(cfg.BatchWait.Milliseconds()))
registry.SetGauge("vehicle_identity_writer_config", metrics.Labels{"setting": "location_touch_interval_seconds"}, cfg.LocationTouchInterval.Seconds())
registry.SetGauge("vehicle_identity_writer_config", metrics.Labels{"setting": "workers"}, float64(cfg.Workers))
health.Start(ctx, logger, health.NewServer(cfg.HealthAddr, "vehicle-identity-writer", []health.Check{
{Name: "mysql", Check: db.PingContext},
}, registry))
store := identity.NewJT808RegistrationStore(db)
logger.Info("identity writer started",
"group", cfg.KafkaGroup,
"topic", cfg.KafkaTopic,
"workers", cfg.Workers,
"batch_size", cfg.BatchSize,
"batch_wait_ms", cfg.BatchWait.Milliseconds(),
"location_touch_interval_seconds", cfg.LocationTouchInterval.Seconds())
var workers sync.WaitGroup
for workerID := 1; workerID <= cfg.Workers; workerID++ {
workers.Add(1)
go func(id int) {
defer workers.Done()
runIdentityConsumer(ctx, logger, registry, store, cfg, id)
}(workerID)
}
workers.Wait()
}
func runIdentityConsumer(
ctx context.Context,
logger *slog.Logger,
registry *metrics.Registry,
store registrationBatchStore,
cfg config,
workerID int,
) {
reader := kafka.NewReader(kafka.ReaderConfig{
Brokers: cfg.KafkaBrokers,
GroupID: cfg.KafkaGroup,
GroupTopics: []string{cfg.KafkaTopic},
MinBytes: 1,
MaxBytes: 10e6,
StartOffset: cfg.StartOffset,
})
defer reader.Close()
// Kafka keeps one phone on one partition. Per-worker throttling avoids a
// global hot lock; a rebalance can only cause one harmless idempotent touch.
projector := identity.NewJT808RegistrationProjector(cfg.Location, cfg.LocationTouchInterval)
workerLabels := metrics.Labels{"worker": strconv.Itoa(workerID)}
registry.SetGauge("vehicle_identity_writer_worker_active", workerLabels, 1)
defer registry.SetGauge("vehicle_identity_writer_worker_active", workerLabels, 0)
logger.Info("identity kafka consumer started", "worker", workerID, "topic", cfg.KafkaTopic)
for {
first, err := reader.FetchMessage(ctx)
if err != nil {
if ctx.Err() != nil {
return
}
logger.Error("kafka fetch failed", "error", err)
if !waitForRetry(ctx, cfg.RetryDelay) {
return
}
continue
}
batch := collectIdentityBatch(ctx, reader, first, cfg.BatchSize, cfg.BatchWait)
if !processIdentityBatchReliablyForWorker(ctx, logger, registry, projector, store, reader, batch, cfg.RetryDelay, workerLabels) {
return
}
}
}
type registrationBatchStore interface {
UpsertBatch(context.Context, []identity.JT808RegistrationFact) error
}
type registrationProjector interface {
ProjectBatch([]envelope.FrameEnvelope) []identity.JT808RegistrationFact
MarkPersisted([]identity.JT808RegistrationFact)
}
type kafkaMessageFetcher interface {
FetchMessage(context.Context) (kafka.Message, error)
}
type kafkaMessageCommitter interface {
CommitMessages(context.Context, ...kafka.Message) error
}
const (
identityFailureWrite = "write_error"
identityFailureCommit = "commit_error"
)
type identityBatchFailure struct {
reason string
err error
}
func (e *identityBatchFailure) Error() string { return e.err.Error() }
func (e *identityBatchFailure) Unwrap() error { return e.err }
func collectIdentityBatch(ctx context.Context, fetcher kafkaMessageFetcher, first kafka.Message, maxSize int, maxWait time.Duration) []kafka.Message {
if maxSize <= 1 {
return []kafka.Message{first}
}
if maxWait <= 0 {
maxWait = 20 * time.Millisecond
}
batch := []kafka.Message{first}
deadline := time.Now().Add(maxWait)
for len(batch) < maxSize {
remaining := time.Until(deadline)
if remaining <= 0 {
break
}
fetchCtx, cancel := context.WithTimeout(ctx, remaining)
message, err := fetcher.FetchMessage(fetchCtx)
cancel()
if err != nil {
break
}
batch = append(batch, message)
}
return batch
}
func processIdentityBatch(
ctx context.Context,
logger *slog.Logger,
registry *metrics.Registry,
projector registrationProjector,
store registrationBatchStore,
committer kafkaMessageCommitter,
messages []kafka.Message,
) error {
return processIdentityBatchAttemptForWorker(ctx, logger, registry, projector, store, committer, messages, true, nil)
}
func processIdentityBatchAttempt(
ctx context.Context,
logger *slog.Logger,
registry *metrics.Registry,
projector registrationProjector,
store registrationBatchStore,
committer kafkaMessageCommitter,
messages []kafka.Message,
recordReceived bool,
) error {
return processIdentityBatchAttemptForWorker(ctx, logger, registry, projector, store, committer, messages, recordReceived, nil)
}
func processIdentityBatchAttemptForWorker(
ctx context.Context,
logger *slog.Logger,
registry *metrics.Registry,
projector registrationProjector,
store registrationBatchStore,
committer kafkaMessageCommitter,
messages []kafka.Message,
recordReceived bool,
workerLabels metrics.Labels,
) error {
if len(messages) == 0 {
return nil
}
operationCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), identityBatchOperationTimeout)
defer cancel()
registry.SetGauge("vehicle_identity_writer_batch_pending_messages", workerLabels, float64(len(messages)))
defer registry.SetGauge("vehicle_identity_writer_batch_pending_messages", workerLabels, 0)
envelopes := make([]envelope.FrameEnvelope, 0, len(messages))
for _, message := range messages {
if recordReceived {
recordIdentityMessage(registry, message, "received")
registry.SetKafkaLag("vehicle_identity_writer_kafka_lag", message.Topic, message.Partition, message.Offset, message.HighWaterMark)
}
var env envelope.FrameEnvelope
if err := json.Unmarshal(message.Value, &env); err != nil {
if recordReceived {
recordIdentityMessage(registry, message, "invalid_json")
logger.Warn("skip invalid identity raw envelope", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "error", err)
}
continue
}
if status, err := topics.ValidateRawEnvelope(message.Topic, env); err != nil {
if recordReceived {
recordIdentityMessage(registry, message, status)
logger.Warn("skip mismatched identity raw envelope", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "protocol", env.Protocol, "event_id", env.StableEventID(), "error", err)
}
continue
}
envelopes = append(envelopes, env)
}
facts := projector.ProjectBatch(envelopes)
registry.SetGauge("vehicle_identity_writer_batch_pending_facts", workerLabels, float64(len(facts)))
defer registry.SetGauge("vehicle_identity_writer_batch_pending_facts", workerLabels, 0)
if len(facts) > 0 {
started := time.Now()
if err := store.UpsertBatch(operationCtx, facts); err != nil {
recordIdentityWrite(registry, "error", len(facts), time.Since(started))
return &identityBatchFailure{reason: identityFailureWrite, err: fmt.Errorf("upsert jt808 registration facts: %w", err)}
}
projector.MarkPersisted(facts)
recordIdentityWrite(registry, "ok", len(facts), time.Since(started))
}
if err := committer.CommitMessages(operationCtx, messages...); err != nil {
for _, message := range messages {
recordIdentityCommit(registry, message, "error")
}
return &identityBatchFailure{reason: identityFailureCommit, err: fmt.Errorf("commit identity kafka batch: %w", err)}
}
for _, message := range messages {
recordIdentityCommit(registry, message, "ok")
}
return nil
}
func processIdentityBatchReliably(
ctx context.Context,
logger *slog.Logger,
registry *metrics.Registry,
projector registrationProjector,
store registrationBatchStore,
committer kafkaMessageCommitter,
messages []kafka.Message,
retryDelay time.Duration,
) bool {
return processIdentityBatchReliablyForWorker(ctx, logger, registry, projector, store, committer, messages, retryDelay, nil)
}
func processIdentityBatchReliablyForWorker(
ctx context.Context,
logger *slog.Logger,
registry *metrics.Registry,
projector registrationProjector,
store registrationBatchStore,
committer kafkaMessageCommitter,
messages []kafka.Message,
retryDelay time.Duration,
workerLabels metrics.Labels,
) bool {
defer registry.SetGauge("vehicle_identity_writer_retry_pending_messages", workerLabels, 0)
recordReceived := true
for {
err := processIdentityBatchAttemptForWorker(ctx, logger, registry, projector, store, committer, messages, recordReceived, workerLabels)
if err == nil {
return true
}
if ctx.Err() != nil {
return false
}
reason := identityFailureReason(err)
registry.SetGauge("vehicle_identity_writer_retry_pending_messages", workerLabels, float64(len(messages)))
registry.IncCounter("vehicle_identity_writer_batch_retries_total", metrics.Labels{"reason": reason})
logger.Error("identity batch failed; retrying without fetching newer offsets", "messages", len(messages), "reason", reason, "error", err)
if reason == identityFailureCommit {
return retryIdentityCommit(ctx, logger, registry, committer, messages, retryDelay)
}
if !waitForRetry(ctx, retryDelay) {
return false
}
recordReceived = false
}
}
func retryIdentityCommit(
ctx context.Context,
logger *slog.Logger,
registry *metrics.Registry,
committer kafkaMessageCommitter,
messages []kafka.Message,
retryDelay time.Duration,
) bool {
for {
if !waitForRetry(ctx, retryDelay) {
return false
}
operationCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), identityBatchOperationTimeout)
err := committer.CommitMessages(operationCtx, messages...)
cancel()
if err == nil {
for _, message := range messages {
recordIdentityCommit(registry, message, "ok")
}
return true
}
for _, message := range messages {
recordIdentityCommit(registry, message, "error")
}
registry.IncCounter("vehicle_identity_writer_batch_retries_total", metrics.Labels{"reason": identityFailureCommit})
logger.Error("identity kafka commit retry failed", "messages", len(messages), "error", err)
}
}
func identityFailureReason(err error) string {
var failure *identityBatchFailure
if errors.As(err, &failure) && failure.reason != "" {
return failure.reason
}
return identityFailureWrite
}
var identityWriteDurationBucketsMS = []float64{1, 5, 10, 25, 50, 100, 250, 500, 1000, 5000}
func recordIdentityMessage(registry *metrics.Registry, message kafka.Message, status string) {
labels := metrics.Labels{"topic": message.Topic, "status": status}
registry.IncCounter("vehicle_identity_writer_kafka_messages_total", labels)
metrics.RecordLastActivity(registry, "vehicle_identity_writer_last_message_unix_seconds", labels)
}
func recordIdentityWrite(registry *metrics.Registry, status string, facts int, elapsed time.Duration) {
labels := metrics.Labels{"status": status}
registry.IncCounter("vehicle_identity_writer_batches_total", labels)
registry.AddCounter("vehicle_identity_writer_facts_total", labels, float64(facts))
registry.ObserveHistogram("vehicle_identity_writer_write_duration_ms_histogram", labels, identityWriteDurationBucketsMS, float64(elapsed.Milliseconds()))
metrics.RecordLastActivity(registry, "vehicle_identity_writer_last_write_unix_seconds", labels)
}
func recordIdentityCommit(registry *metrics.Registry, message kafka.Message, status string) {
labels := metrics.Labels{"topic": message.Topic, "status": status}
registry.IncCounter("vehicle_identity_writer_kafka_commits_total", labels)
metrics.RecordLastActivity(registry, "vehicle_identity_writer_last_commit_unix_seconds", labels)
}
func waitForRetry(ctx context.Context, delay time.Duration) bool {
if delay <= 0 {
delay = time.Second
}
timer := time.NewTimer(delay)
defer timer.Stop()
select {
case <-ctx.Done():
return false
case <-timer.C:
return true
}
}
type config struct {
KafkaBrokers []string
KafkaTopic string
KafkaGroup string
MySQLDSN string
MySQLMaxOpenConns int
MySQLMaxIdleConns int
MySQLConnMaxLifetime time.Duration
EnsureSchema bool
HealthAddr string
Location *time.Location
LocationTouchInterval time.Duration
BatchSize int
BatchWait time.Duration
RetryDelay time.Duration
StartOffset int64
Workers int
}
func loadConfig() config {
location, err := time.LoadLocation(env("LOCAL_TZ", "Asia/Shanghai"))
if err != nil {
location = time.FixedZone("Asia/Shanghai", 8*60*60)
}
startOffset := kafka.LastOffset
if strings.EqualFold(env("KAFKA_START_OFFSET", "last"), "first") {
startOffset = kafka.FirstOffset
}
return config{
KafkaBrokers: splitCSV(env("KAFKA_BROKERS", "127.0.0.1:9092")),
KafkaTopic: env("KAFKA_TOPIC", topics.RawJT808),
KafkaGroup: env("KAFKA_GROUP", "go-identity-writer"),
MySQLDSN: env("MYSQL_DSN", ""),
MySQLMaxOpenConns: envInt("MYSQL_MAX_OPEN_CONNS", 8),
MySQLMaxIdleConns: envInt("MYSQL_MAX_IDLE_CONNS", 4),
MySQLConnMaxLifetime: time.Duration(envInt("MYSQL_CONN_MAX_LIFETIME_SECONDS", 300)) * time.Second,
EnsureSchema: envBool("MYSQL_ENSURE_SCHEMA", true),
HealthAddr: env("HEALTH_ADDR", "127.0.0.1:20217"),
Location: location,
LocationTouchInterval: time.Duration(envInt("JT808_REGISTRATION_LOCATION_TOUCH_INTERVAL_SECONDS", 600)) * time.Second,
BatchSize: envInt("IDENTITY_WRITER_BATCH_SIZE", 500),
BatchWait: time.Duration(envInt("IDENTITY_WRITER_BATCH_WAIT_MS", 20)) * time.Millisecond,
RetryDelay: time.Duration(envInt("IDENTITY_WRITER_RETRY_DELAY_MS", 1000)) * time.Millisecond,
StartOffset: startOffset,
Workers: envInt("IDENTITY_WRITER_WORKERS", 3),
}
}
func (c config) Validate() error {
if len(c.KafkaBrokers) == 0 {
return fmt.Errorf("KAFKA_BROKERS is required")
}
if strings.TrimSpace(c.MySQLDSN) == "" {
return fmt.Errorf("MYSQL_DSN is required")
}
protocol, ok := topics.ProtocolForKnownRawTopic(c.KafkaTopic)
if !ok || protocol != string(envelope.ProtocolJT808) {
return fmt.Errorf("identity writer consumes JT808 raw topic only, got %q", c.KafkaTopic)
}
if c.BatchSize <= 0 {
return fmt.Errorf("IDENTITY_WRITER_BATCH_SIZE must be positive")
}
if c.Workers <= 0 {
return fmt.Errorf("IDENTITY_WRITER_WORKERS must be positive")
}
return nil
}
func env(key string, fallback string) string {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
return value
}
func envInt(key string, fallback int) int {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
parsed, err := strconv.Atoi(value)
if err != nil {
return fallback
}
return parsed
}
func envBool(key string, fallback bool) bool {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
parsed, err := strconv.ParseBool(value)
if err != nil {
return fallback
}
return parsed
}
func splitCSV(value string) []string {
parts := strings.Split(value, ",")
out := make([]string, 0, len(parts))
for _, item := range parts {
if trimmed := strings.TrimSpace(item); trimmed != "" {
out = append(out, trimmed)
}
}
return out
}

View File

@@ -0,0 +1,314 @@
package main
import (
"context"
"errors"
"io"
"log/slog"
"strings"
"testing"
"time"
"github.com/segmentio/kafka-go"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/identity"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/topics"
)
type fakeRegistrationProjector struct {
facts []identity.JT808RegistrationFact
envelopes []envelope.FrameEnvelope
markedFacts []identity.JT808RegistrationFact
}
func (p *fakeRegistrationProjector) ProjectBatch(envs []envelope.FrameEnvelope) []identity.JT808RegistrationFact {
p.envelopes = append(p.envelopes, envs...)
return p.facts
}
func (p *fakeRegistrationProjector) MarkPersisted(facts []identity.JT808RegistrationFact) {
p.markedFacts = append(p.markedFacts, facts...)
}
type fakeRegistrationStore struct {
facts []identity.JT808RegistrationFact
err error
count int
failOnCount int
}
func (s *fakeRegistrationStore) UpsertBatch(_ context.Context, facts []identity.JT808RegistrationFact) error {
s.count++
s.facts = append(s.facts, facts...)
if s.err != nil && (s.failOnCount == 0 || s.count == s.failOnCount) {
return s.err
}
return nil
}
type fakeCommitter struct {
messages []kafka.Message
err error
count int
failOnCount int
}
func (c *fakeCommitter) CommitMessages(_ context.Context, messages ...kafka.Message) error {
c.count++
c.messages = append(c.messages, messages...)
if c.err != nil && (c.failOnCount == 0 || c.count == c.failOnCount) {
return c.err
}
return nil
}
func TestProcessIdentityBatchDoesNotCommitOrThrottleOnStoreFailure(t *testing.T) {
wantErr := errors.New("mysql unavailable")
projector := &fakeRegistrationProjector{facts: []identity.JT808RegistrationFact{{
Phone: "13307795425",
SeenAt: time.Now(),
}}}
store := &fakeRegistrationStore{err: wantErr}
committer := &fakeCommitter{}
err := processIdentityBatch(
context.Background(),
slog.New(slog.NewTextHandler(io.Discard, nil)),
metrics.NewRegistry(),
projector,
store,
committer,
[]kafka.Message{validIdentityMessage(t, 1)},
)
if !errors.Is(err, wantErr) {
t.Fatalf("processIdentityBatch() error = %v, want %v", err, wantErr)
}
if len(committer.messages) != 0 {
t.Fatalf("committed messages = %d, want 0", len(committer.messages))
}
if len(projector.markedFacts) != 0 {
t.Fatalf("marked facts = %d, want 0", len(projector.markedFacts))
}
}
func TestProcessIdentityBatchMarksOnlyAfterStoreAndCommits(t *testing.T) {
fact := identity.JT808RegistrationFact{Phone: "13307795425", SeenAt: time.Now()}
projector := &fakeRegistrationProjector{facts: []identity.JT808RegistrationFact{fact}}
store := &fakeRegistrationStore{}
committer := &fakeCommitter{}
messages := []kafka.Message{validIdentityMessage(t, 1), validIdentityMessage(t, 2)}
err := processIdentityBatch(
context.Background(),
slog.New(slog.NewTextHandler(io.Discard, nil)),
metrics.NewRegistry(),
projector,
store,
committer,
messages,
)
if err != nil {
t.Fatalf("processIdentityBatch() error = %v", err)
}
if len(store.facts) != 1 || len(projector.markedFacts) != 1 {
t.Fatalf("store facts = %d marked facts = %d, want 1/1", len(store.facts), len(projector.markedFacts))
}
if len(committer.messages) != len(messages) {
t.Fatalf("committed messages = %d, want %d", len(committer.messages), len(messages))
}
}
func TestProcessIdentityBatchCommitsPoisonEnvelopeWithoutProjection(t *testing.T) {
projector := &fakeRegistrationProjector{}
store := &fakeRegistrationStore{}
committer := &fakeCommitter{}
message := kafka.Message{Topic: topics.RawJT808, Partition: 1, Offset: 7, Value: []byte("not-json")}
err := processIdentityBatch(
context.Background(),
slog.New(slog.NewTextHandler(io.Discard, nil)),
metrics.NewRegistry(),
projector,
store,
committer,
[]kafka.Message{message},
)
if err != nil {
t.Fatalf("processIdentityBatch() error = %v", err)
}
if len(store.facts) != 0 || len(projector.envelopes) != 0 {
t.Fatalf("poison message reached projector/store: envs=%d facts=%d", len(projector.envelopes), len(store.facts))
}
if len(committer.messages) != 1 {
t.Fatalf("committed messages = %d, want 1", len(committer.messages))
}
}
func TestProcessIdentityBatchReliablyRetriesCommitWithoutRewritingMySQL(t *testing.T) {
fact := identity.JT808RegistrationFact{Phone: "13307795425", SeenAt: time.Now()}
projector := &fakeRegistrationProjector{facts: []identity.JT808RegistrationFact{fact}}
store := &fakeRegistrationStore{}
committer := &fakeCommitter{err: errors.New("commit failed"), failOnCount: 1}
registry := metrics.NewRegistry()
messages := []kafka.Message{validIdentityMessage(t, 1), validIdentityMessage(t, 2)}
ok := processIdentityBatchReliably(
context.Background(),
slog.New(slog.NewTextHandler(io.Discard, nil)),
registry,
projector,
store,
committer,
messages,
time.Nanosecond,
)
if !ok {
t.Fatal("reliable identity batch returned false")
}
if store.count != 1 {
t.Fatalf("mysql writes = %d, want 1 after commit-only retry", store.count)
}
if committer.count != 2 {
t.Fatalf("commit attempts = %d, want initial failure and one retry", committer.count)
}
text := registry.Render()
for _, want := range []string{
`vehicle_identity_writer_batch_retries_total{reason="commit_error"} 1`,
`vehicle_identity_writer_retry_pending_messages 0`,
} {
if !strings.Contains(text, want) {
t.Fatalf("missing metric %s:\n%s", want, text)
}
}
}
func TestProcessIdentityBatchReliablyRetriesStoreBeforeCommit(t *testing.T) {
fact := identity.JT808RegistrationFact{Phone: "13307795425", SeenAt: time.Now()}
projector := &fakeRegistrationProjector{facts: []identity.JT808RegistrationFact{fact}}
store := &fakeRegistrationStore{err: errors.New("mysql unavailable"), failOnCount: 1}
committer := &fakeCommitter{}
registry := metrics.NewRegistry()
messages := []kafka.Message{validIdentityMessage(t, 1)}
ok := processIdentityBatchReliably(
context.Background(),
slog.New(slog.NewTextHandler(io.Discard, nil)),
registry,
projector,
store,
committer,
messages,
time.Nanosecond,
)
if !ok {
t.Fatal("reliable identity batch returned false")
}
if store.count != 2 || committer.count != 1 {
t.Fatalf("mysql writes=%d commits=%d, want 2/1", store.count, committer.count)
}
text := registry.Render()
for _, want := range []string{
`vehicle_identity_writer_batch_retries_total{reason="write_error"} 1`,
`vehicle_identity_writer_kafka_messages_total{status="received",topic="vehicle.raw.go.jt808.v1"} 1`,
} {
if !strings.Contains(text, want) {
t.Fatalf("missing metric %s:\n%s", want, text)
}
}
}
func TestConfigRejectsNonJT808Topic(t *testing.T) {
cfg := config{
KafkaBrokers: []string{"127.0.0.1:9092"},
KafkaTopic: topics.RawGB32960,
MySQLDSN: "user:pass@tcp(localhost:3306)/db",
BatchSize: 100,
Workers: 3,
}
if err := cfg.Validate(); err == nil {
t.Fatal("Validate() error = nil, want non-JT808 topic error")
}
cfg.KafkaTopic = topics.RawJT808
if err := cfg.Validate(); err != nil {
t.Fatalf("Validate() error = %v", err)
}
}
func TestLoadConfigDefaultsAndOverridesWorkers(t *testing.T) {
t.Setenv("IDENTITY_WRITER_WORKERS", "")
if got := loadConfig().Workers; got != 3 {
t.Fatalf("default workers = %d, want 3", got)
}
t.Setenv("IDENTITY_WRITER_WORKERS", "5")
if got := loadConfig().Workers; got != 5 {
t.Fatalf("configured workers = %d, want 5", got)
}
}
func TestConfigRejectsNonPositiveWorkers(t *testing.T) {
cfg := config{
KafkaBrokers: []string{"127.0.0.1:9092"},
KafkaTopic: topics.RawJT808,
MySQLDSN: "user:pass@tcp(localhost:3306)/db",
BatchSize: 100,
}
if err := cfg.Validate(); err == nil || !strings.Contains(err.Error(), "IDENTITY_WRITER_WORKERS") {
t.Fatalf("Validate() error = %v, want workers error", err)
}
}
func TestProcessIdentityBatchForWorkerLabelsPendingMetrics(t *testing.T) {
registry := metrics.NewRegistry()
workerLabels := metrics.Labels{"worker": "2"}
err := processIdentityBatchAttemptForWorker(
context.Background(),
slog.New(slog.NewTextHandler(io.Discard, nil)),
registry,
&fakeRegistrationProjector{},
&fakeRegistrationStore{},
&fakeCommitter{},
[]kafka.Message{validIdentityMessage(t, 1)},
true,
workerLabels,
)
if err != nil {
t.Fatalf("processIdentityBatchAttemptForWorker() error = %v", err)
}
text := registry.Render()
for _, want := range []string{
`vehicle_identity_writer_batch_pending_messages{worker="2"} 0`,
`vehicle_identity_writer_batch_pending_facts{worker="2"} 0`,
} {
if !strings.Contains(text, want) {
t.Fatalf("missing metric %s:\n%s", want, text)
}
}
}
func validIdentityMessage(t *testing.T, offset int64) kafka.Message {
t.Helper()
env := envelope.FrameEnvelope{
Protocol: envelope.ProtocolJT808,
MessageID: identity.JT808LocationMessageID,
EventKind: envelope.EventKindRaw,
Phone: "13307795425",
VIN: "LTESTVIN000000001",
ReceivedAtMS: time.Now().UnixMilli(),
ParseStatus: envelope.ParseOK,
}
payload, err := env.MarshalJSONBytes()
if err != nil {
t.Fatal(err)
}
return kafka.Message{
Topic: topics.RawJT808,
Partition: 1,
Offset: offset,
HighWaterMark: offset + 1,
Value: payload,
}
}

View File

@@ -0,0 +1,104 @@
package main
import (
"context"
"database/sql"
"flag"
"fmt"
"log"
"os"
"os/signal"
"strings"
"syscall"
_ "github.com/go-sql-driver/mysql"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/loadsim"
)
func main() {
log.SetFlags(log.LstdFlags | log.Lmicroseconds)
flagCfg := loadsim.RegisterFlags(flag.CommandLine)
cleanupRegistration := flag.Bool("cleanup-registration", false, "delete this run's loopback JT808 registration rows after the simulation")
cleanupOnly := flag.Bool("cleanup-only", false, "skip the simulation and only clean the configured JT808 phone range")
mysqlDSN := flag.String("mysql-dsn", strings.TrimSpace(os.Getenv("MYSQL_DSN")), "MySQL DSN used only by JT808 registration cleanup")
if err := flag.CommandLine.Parse(os.Args[1:]); err != nil {
log.Fatal(err)
}
cfg, err := flagCfg.Build()
if err != nil {
log.Fatal(err)
}
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
if *cleanupOnly {
deleted, err := cleanupJT808Registrations(ctx, cfg, *mysqlDSN)
if err != nil {
log.Fatal(err)
}
log.Printf("jt808 synthetic registration cleanup completed rows_deleted=%d", deleted)
return
}
log.Printf("load simulation started protocol=%s addr=%s connections=%d connect_rate=%d send_interval=%s duration=%s template=%s",
cfg.Protocol, cfg.Addr, cfg.Connections, cfg.ConnectRatePerSecond, cfg.SendInterval, cfg.Duration, cfg.Template)
stats, err := (loadsim.Runner{}).Run(ctx, cfg)
if err != nil {
log.Fatal(err)
}
log.Print(formatStats(stats))
if *cleanupRegistration {
deleted, err := cleanupJT808Registrations(ctx, cfg, *mysqlDSN)
if err != nil {
log.Fatal(err)
}
log.Printf("jt808 synthetic registration cleanup completed rows_deleted=%d", deleted)
}
}
func cleanupJT808Registrations(ctx context.Context, cfg loadsim.Config, dsn string) (int64, error) {
if cfg.Protocol != loadsim.ProtocolJT808 {
return 0, fmt.Errorf("registration cleanup only supports jt808")
}
if strings.TrimSpace(dsn) == "" {
return 0, fmt.Errorf("mysql-dsn or MYSQL_DSN is required for registration cleanup")
}
db, err := sql.Open("mysql", dsn)
if err != nil {
return 0, fmt.Errorf("open mysql for registration cleanup: %w", err)
}
defer db.Close()
return cleanupJT808RegistrationsWithDB(ctx, db, cfg)
}
type cleanupExecer interface {
ExecContext(context.Context, string, ...any) (sql.Result, error)
}
func cleanupJT808RegistrationsWithDB(ctx context.Context, exec cleanupExecer, cfg loadsim.Config) (int64, error) {
firstPhone := fmt.Sprintf("%012d", cfg.JT808PhoneBase)
lastPhone := fmt.Sprintf("%012d", cfg.JT808PhoneBase+int64(cfg.Connections)-1)
result, err := exec.ExecContext(ctx, `DELETE FROM jt808_registration
WHERE phone BETWEEN ? AND ?
AND source_ip IN ('127.0.0.1', '::1')`, firstPhone, lastPhone)
if err != nil {
return 0, fmt.Errorf("delete loopback jt808 registrations: %w", err)
}
deleted, err := result.RowsAffected()
if err != nil {
return 0, fmt.Errorf("read registration cleanup result: %w", err)
}
return deleted, nil
}
func formatStats(stats loadsim.Stats) string {
return fmt.Sprintf("connections_opened=%d connections_failed=%d frames_written=%d write_errors=%d response_bytes=%d read_errors=%d",
stats.ConnectionsOpened,
stats.ConnectionsFailed,
stats.FramesWritten,
stats.WriteErrors,
stats.ResponseBytes,
stats.ReadErrors,
)
}

View File

@@ -0,0 +1,61 @@
package main
import (
"context"
"strings"
"testing"
"github.com/DATA-DOG/go-sqlmock"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/loadsim"
)
func TestFormatStatsIncludesCapacityCounters(t *testing.T) {
out := formatStats(loadsim.Stats{
ConnectionsOpened: 10,
ConnectionsFailed: 2,
FramesWritten: 300,
WriteErrors: 1,
ResponseBytes: 2048,
ReadErrors: 0,
})
for _, want := range []string{
"connections_opened=10",
"connections_failed=2",
"frames_written=300",
"write_errors=1",
"response_bytes=2048",
"read_errors=0",
} {
if !strings.Contains(out, want) {
t.Fatalf("formatStats() = %q, missing %q", out, want)
}
}
}
func TestCleanupJT808RegistrationsUsesBoundedLoopbackDelete(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatal(err)
}
defer db.Close()
mock.ExpectExec(`DELETE FROM jt808_registration`).
WithArgs("139000000000", "139000000999").
WillReturnResult(sqlmock.NewResult(0, 1000))
deleted, err := cleanupJT808RegistrationsWithDB(context.Background(), db, loadsim.Config{
Protocol: loadsim.ProtocolJT808,
Connections: 1000,
JT808PhoneBase: loadsim.DefaultJT808PhoneBase,
})
if err != nil {
t.Fatalf("cleanupJT808RegistrationsWithDB() error = %v", err)
}
if deleted != 1000 {
t.Fatalf("deleted = %d, want 1000", deleted)
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatal(err)
}
}

View File

@@ -0,0 +1,964 @@
package main
import (
"context"
"database/sql"
"encoding/json"
"errors"
"fmt"
"log/slog"
"os"
"os/signal"
"strings"
"syscall"
"time"
"github.com/nats-io/nats.go"
"github.com/redis/go-redis/v9"
_ "github.com/taosdata/driver-go/v3/taosWS"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/health"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/history"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/observability"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/realtime"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/topics"
)
func main() {
logger := observability.NewLogger("nats-fast-writer")
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
cfg := loadConfig()
conn, err := nats.Connect(cfg.NATSURL, nats.Name(cfg.NATSClientName), nats.Timeout(5*time.Second))
if err != nil {
logger.Error("nats connect failed", "error", err)
os.Exit(1)
}
defer conn.Close()
js, err := conn.JetStream()
if err != nil {
logger.Error("nats jetstream init failed", "error", err)
os.Exit(1)
}
if err := ensureStream(js, cfg); err != nil {
logger.Error("nats stream ensure failed", "stream", cfg.NATSStream, "error", err)
os.Exit(1)
}
sub, err := js.PullSubscribe(
cfg.NATSFilter,
cfg.NATSDurable,
nats.BindStream(cfg.NATSStream),
nats.ManualAck(),
nats.AckWait(cfg.AckWait),
nats.MaxDeliver(-1),
)
if err != nil {
logger.Error("nats pull consumer init failed", "stream", cfg.NATSStream, "durable", cfg.NATSDurable, "error", err)
os.Exit(1)
}
var historyWriter fastAppender
var tdCheck health.Check
if cfg.TDengineEnabled {
tdDB, err := sql.Open(cfg.TDengineDriver, cfg.TDengineDSN)
if err != nil {
logger.Error("tdengine open failed", "error", err)
os.Exit(1)
}
defer tdDB.Close()
tdDB.SetMaxOpenConns(cfg.TDengineMaxOpenConns)
tdDB.SetMaxIdleConns(cfg.TDengineMaxIdleConns)
if err := tdDB.PingContext(ctx); err != nil {
logger.Error("tdengine ping failed", "error", err)
os.Exit(1)
}
writer := history.NewWriterWithDatabase(tdDB, cfg.TDengineDatabase)
if cfg.TDengineEnsureSchema {
if err := writer.EnsureSchema(ctx, cfg.TDengineDatabase); err != nil {
logger.Error("tdengine schema bootstrap failed", "error", err)
os.Exit(1)
}
}
historyWriter = writer
tdCheck = health.Check{Name: "tdengine", Check: tdDB.PingContext}
} else {
logger.Info("nats fast writer tdengine stage disabled")
}
redisClient := redis.NewClient(&redis.Options{
Addr: cfg.RedisAddr,
Username: cfg.RedisUsername,
Password: cfg.RedisPassword,
DB: cfg.RedisDB,
})
defer redisClient.Close()
if err := redisClient.Ping(ctx).Err(); err != nil {
logger.Error("redis ping failed", "error", err)
os.Exit(1)
}
realtimeRepo := realtime.NewRepository(redisClient, realtime.Config{OnlineTTL: cfg.OnlineTTL})
registry := metrics.NewRegistry()
recordFastWriterConfigMetrics(registry, cfg)
registry.SetGauge("vehicle_fast_writer_tdengine_enabled", nil, boolGauge(cfg.TDengineEnabled))
healthChecks := []health.Check{
{Name: "nats", Check: func(context.Context) error {
if conn.Status() != nats.CONNECTED {
return fmt.Errorf("nats status is %s", conn.Status().String())
}
return nil
}},
{Name: "redis", Check: func(ctx context.Context) error { return redisClient.Ping(ctx).Err() }},
}
if tdCheck.Name != "" {
healthChecks = append(healthChecks, tdCheck)
}
health.Start(ctx, logger, health.NewServer(env("HEALTH_ADDR", ""), "nats-fast-writer", healthChecks, registry))
logger.Info("nats fast writer started",
"stream", cfg.NATSStream,
"durable", cfg.NATSDurable,
"filter", cfg.NATSFilter,
"batch_size", cfg.BatchSize,
"fetch_wait_ms", cfg.FetchWait.Milliseconds(),
"workers", cfg.Workers,
"tdengine_enabled", cfg.TDengineEnabled,
"tdengine_max_open_conns", cfg.TDengineMaxOpenConns,
"tdengine_max_idle_conns", cfg.TDengineMaxIdleConns,
"operation_timeout_ms", cfg.OperationWait.Milliseconds())
for i := 0; i < cfg.Workers; i++ {
go runFastWorker(ctx, logger, registry, js, sub, historyWriter, realtimeRepo, cfg)
}
<-ctx.Done()
}
type config struct {
NATSURL string
NATSClientName string
NATSStream string
NATSDurable string
NATSFilter string
NATSSubjects []string
BatchSize int
FetchWait time.Duration
OperationWait time.Duration
AckWait time.Duration
StreamMaxAge time.Duration
StreamMaxBytes int64
StreamEnsureWait time.Duration
Workers int
TDengineEnabled bool
TDengineDriver string
TDengineDSN string
TDengineDatabase string
TDengineEnsureSchema bool
TDengineMaxOpenConns int
TDengineMaxIdleConns int
RedisAddr string
RedisUsername string
RedisPassword string
RedisDB int
OnlineTTL time.Duration
}
func loadConfig() config {
subjects := splitCSV(env("NATS_STREAM_SUBJECTS", strings.Join([]string{
env("NATS_SUBJECT_GB32960_RAW", env("KAFKA_TOPIC_GB32960_RAW", topics.RawGB32960)),
env("NATS_SUBJECT_JT808_RAW", env("KAFKA_TOPIC_JT808_RAW", topics.RawJT808)),
env("NATS_SUBJECT_YUTONG_MQTT_RAW", env("KAFKA_TOPIC_YUTONG_MQTT_RAW", topics.RawYutongMQTT)),
env("NATS_SUBJECT_GB32960_FIELDS", env("KAFKA_TOPIC_GB32960_FIELDS", topics.FieldsGB32960)),
env("NATS_SUBJECT_JT808_FIELDS", env("KAFKA_TOPIC_JT808_FIELDS", topics.FieldsJT808)),
env("NATS_SUBJECT_YUTONG_MQTT_FIELDS", env("KAFKA_TOPIC_YUTONG_MQTT_FIELDS", topics.FieldsYutongMQTT)),
}, ",")))
workers := envInt("FAST_WRITER_WORKERS", 8)
maxOpenConns := envInt("FAST_WRITER_TDENGINE_MAX_OPEN_CONNS", 1)
maxIdleConns := envInt("FAST_WRITER_TDENGINE_MAX_IDLE_CONNS", maxOpenConns)
return config{
NATSURL: env("NATS_URL", "nats://127.0.0.1:4222"),
NATSClientName: env("NATS_CLIENT_NAME", "lingniu-nats-fast-writer"),
NATSStream: env("NATS_STREAM", "VEHICLE_INGEST"),
NATSDurable: env("NATS_DURABLE", "vehicle-fast-writer"),
NATSFilter: env("NATS_FILTER", "vehicle.raw.go.>"),
NATSSubjects: subjects,
BatchSize: envInt("FAST_WRITER_BATCH_SIZE", 100),
FetchWait: time.Duration(envInt("FAST_WRITER_FETCH_WAIT_MS", 20)) * time.Millisecond,
OperationWait: time.Duration(envInt("FAST_WRITER_OPERATION_TIMEOUT_MS", 1000)) * time.Millisecond,
AckWait: time.Duration(envInt("NATS_ACK_WAIT_SECONDS", 30)) * time.Second,
StreamMaxAge: time.Duration(envInt("NATS_STREAM_MAX_AGE_HOURS", 24)) * time.Hour,
StreamMaxBytes: envInt64("NATS_STREAM_MAX_BYTES", 20*1024*1024*1024),
StreamEnsureWait: time.Duration(envInt("NATS_STREAM_ENSURE_TIMEOUT_SECONDS", 60)) * time.Second,
Workers: workers,
TDengineEnabled: envBool("FAST_WRITER_TDENGINE_ENABLED", false),
TDengineDriver: env("TDENGINE_DRIVER", "taosWS"),
TDengineDSN: env("TDENGINE_DSN", ""),
TDengineDatabase: env("TDENGINE_DATABASE", history.DefaultDatabase),
TDengineEnsureSchema: env("TDENGINE_ENSURE_SCHEMA", "true") != "false",
TDengineMaxOpenConns: maxOpenConns,
TDengineMaxIdleConns: maxIdleConns,
RedisAddr: env("REDIS_ADDR", "127.0.0.1:6379"),
RedisUsername: env("REDIS_USERNAME", ""),
RedisPassword: env("REDIS_PASSWORD", ""),
RedisDB: envInt("REDIS_DB", 0),
OnlineTTL: time.Duration(envInt("REALTIME_ONLINE_TTL_SECONDS", 60)) * time.Second,
}
}
type fastAppender interface {
AppendAll(context.Context, envelope.FrameEnvelope) error
AppendAllBatch(context.Context, []envelope.FrameEnvelope) error
}
type fastUpdater interface {
FastUpdate(context.Context, envelope.FrameEnvelope) error
}
type fastResultUpdater interface {
FastUpdateWithResult(context.Context, envelope.FrameEnvelope) (realtime.FastUpdateResult, error)
}
type fastBatchUpdater interface {
FastUpdateBatch(context.Context, []envelope.FrameEnvelope) error
}
type fastBatchResultUpdater interface {
FastUpdateBatchWithResult(context.Context, []envelope.FrameEnvelope) (realtime.FastUpdateResult, error)
}
type fastMessage struct {
subject string
data []byte
ack func() error
}
func runFastWorker(ctx context.Context, logger *slog.Logger, registry *metrics.Registry, infoReader natsConsumerInfoReader, sub natsPullSubscription, appender fastAppender, updater fastUpdater, cfg config) {
var lastConsumerInfoAt time.Time
for {
if ctx.Err() != nil {
return
}
if time.Since(lastConsumerInfoAt) >= time.Duration(envInt("NATS_CONSUMER_METRICS_INTERVAL_SECONDS", 10))*time.Second {
lastConsumerInfoAt = time.Now()
if infoReader != nil {
info, err := infoReader.ConsumerInfo(cfg.NATSStream, cfg.NATSDurable, nats.Context(ctx))
if err != nil {
logger.Warn("nats consumer info failed", "stream", cfg.NATSStream, "durable", cfg.NATSDurable, "error", err)
} else {
recordFastNATSConsumerInfoMetrics(registry, cfg, info)
}
}
}
msgs, err := sub.Fetch(cfg.BatchSize, nats.MaxWait(cfg.FetchWait))
if err != nil {
if isFastWorkerShutdownFetchError(ctx, err) {
return
}
if errors.Is(err, nats.ErrTimeout) {
continue
}
if isTransientFastFetchError(err) {
logger.Warn("nats fetch interrupted", "error", err)
time.Sleep(time.Second)
continue
}
logger.Error("nats fetch failed", "error", err)
time.Sleep(time.Second)
continue
}
fastMessages := make([]*fastMessage, 0, len(msgs))
for _, msg := range msgs {
natsMsg := msg
fastMessages = append(fastMessages, &fastMessage{
subject: natsMsg.Subject,
data: natsMsg.Data,
ack: func() error {
return natsMsg.Ack()
},
})
}
operationCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), cfg.OperationWait)
err = processFastBatch(operationCtx, registry, appender, updater, fastMessages)
cancel()
if err != nil {
addFastMetric(registry, fastBatchSubject(fastMessages), "error")
logger.Error("fast write batch failed", "messages", len(fastMessages), "error", err)
continue
}
}
}
func isFastWorkerShutdownFetchError(ctx context.Context, err error) bool {
if err == nil {
return false
}
if ctx.Err() != nil {
return true
}
return errors.Is(err, nats.ErrConnectionClosed)
}
func isTransientFastFetchError(err error) bool {
if err == nil {
return false
}
text := strings.ToLower(strings.TrimSpace(err.Error()))
return strings.Contains(text, "disconnected during fetch") ||
strings.Contains(text, "connection closed") ||
strings.Contains(text, "connection reset") ||
strings.Contains(text, "broken pipe") ||
strings.Contains(text, "temporary") ||
strings.Contains(text, "temporarily") ||
strings.Contains(text, "timeout")
}
type natsPullSubscription interface {
Fetch(int, ...nats.PullOpt) ([]*nats.Msg, error)
}
type natsConsumerInfoReader interface {
ConsumerInfo(stream string, name string, opts ...nats.JSOpt) (*nats.ConsumerInfo, error)
}
var fastWriterStageDurationBucketsMS = []float64{1, 5, 10, 25, 50, 100, 250, 500, 1000, 5000}
var fastWriterRedisE2EDurationBucketsMS = []float64{10, 25, 50, 100, 250, 500, 1000, 2500, 5000, 10000}
var fastWriterRedisE2ERecent = metrics.NewRecentLatencyByKey(512)
var fastBatchPending = metrics.PendingPairGauge{}
func processFastBatch(ctx context.Context, registry *metrics.Registry, appender fastAppender, updater fastUpdater, messages []*fastMessage) error {
if len(messages) == 0 {
return nil
}
for _, msg := range messages {
addFastMetric(registry, msg.subject, "received")
}
envelopes := make([]envelope.FrameEnvelope, 0, len(messages))
validMessages := make([]*fastMessage, 0, len(messages))
for _, msg := range messages {
var env envelope.FrameEnvelope
if err := json.Unmarshal(msg.data, &env); err != nil {
addFastMetric(registry, msg.subject, "invalid_json")
if msg.ack != nil {
started := time.Now()
err := msg.ack()
recordFastWriterStageDuration(registry, msg.subject, "ack", statusFromError(err), time.Since(started))
if err != nil {
addFastMetric(registry, msg.subject, "ack_error")
return fmt.Errorf("nats ack invalid json: %w", err)
}
}
continue
}
if status, err := topics.ValidateRawEnvelope(msg.subject, env); err != nil {
addFastMetric(registry, msg.subject, status)
if msg.ack != nil {
started := time.Now()
err := msg.ack()
recordFastWriterStageDuration(registry, msg.subject, "ack", statusFromError(err), time.Since(started))
if err != nil {
addFastMetric(registry, msg.subject, "ack_error")
return fmt.Errorf("nats ack mismatched raw envelope: %w", err)
}
}
continue
}
envelopes = append(envelopes, env)
validMessages = append(validMessages, msg)
}
if len(envelopes) == 0 {
return nil
}
addFastBatchPending(registry, len(messages), len(envelopes))
defer addFastBatchPending(registry, -len(messages), -len(envelopes))
subject := fastBatchSubject(validMessages)
if appender != nil {
started := time.Now()
err := appender.AppendAllBatch(ctx, envelopes)
recordFastWriterStageDuration(registry, subject, "tdengine", statusFromError(err), time.Since(started))
if err != nil {
if shouldFallbackFastBatchError(err) {
if fallbackErr := processFastMessagesIndividually(ctx, registry, appender, updater, validMessages); fallbackErr != nil {
addFastBatchFallbackMetric(registry, "tdengine", "error")
return fmt.Errorf("tdengine batch fallback: %w", fallbackErr)
}
addFastBatchFallbackMetric(registry, "tdengine", "ok")
return nil
}
addFastBatchFallbackMetric(registry, "tdengine", "skipped_transient")
if catchupErr := updateFastRedisBatchWithoutAck(ctx, registry, updater, validMessages, envelopes); catchupErr != nil {
addFastDecoupledUpdateMetric(registry, "tdengine_transient", "error")
return fmt.Errorf("tdengine batch append: %w; redis catchup: %v", err, catchupErr)
}
addFastDecoupledUpdateMetric(registry, "tdengine_transient", "ok")
return fmt.Errorf("tdengine batch append: %w", err)
}
}
if batchUpdater, ok := updater.(fastBatchResultUpdater); ok {
for _, group := range fastSubjectGroups(validMessages, envelopes) {
started := time.Now()
result, err := batchUpdater.FastUpdateBatchWithResult(ctx, group.envelopes)
recordFastWriterStageDuration(registry, group.subject, "redis", statusFromError(err), time.Since(started))
if err != nil {
if shouldFallbackFastBatchError(err) {
if fallbackErr := processFastMessagesIndividually(ctx, registry, nil, updater, validMessages); fallbackErr != nil {
addFastBatchFallbackMetric(registry, "redis", "error")
return fmt.Errorf("redis fast batch fallback: %w", fallbackErr)
}
addFastBatchFallbackMetric(registry, "redis", "ok")
return nil
}
addFastBatchFallbackMetric(registry, "redis", "skipped_transient")
return fmt.Errorf("redis fast batch update: %w", err)
}
recordFastWriterRedisEnvelopeMetrics(registry, group.subject, result)
recordFastWriterRedisFieldMetrics(registry, group.subject, result)
recordFastWriterRedisE2EDurationMessages(registry, group.messages, group.envelopes, group.subject)
}
for _, msg := range validMessages {
if msg.ack != nil {
started := time.Now()
err := msg.ack()
recordFastWriterStageDuration(registry, msg.subject, "ack", statusFromError(err), time.Since(started))
if err != nil {
addFastMetric(registry, msg.subject, "ack_error")
return fmt.Errorf("nats ack: %w", err)
}
}
addFastMetric(registry, msg.subject, "ok")
}
return nil
}
if batchUpdater, ok := updater.(fastBatchUpdater); ok {
for _, group := range fastSubjectGroups(validMessages, envelopes) {
started := time.Now()
err := batchUpdater.FastUpdateBatch(ctx, group.envelopes)
recordFastWriterStageDuration(registry, group.subject, "redis", statusFromError(err), time.Since(started))
if err != nil {
if shouldFallbackFastBatchError(err) {
if fallbackErr := processFastMessagesIndividually(ctx, registry, nil, updater, validMessages); fallbackErr != nil {
addFastBatchFallbackMetric(registry, "redis", "error")
return fmt.Errorf("redis fast batch fallback: %w", fallbackErr)
}
addFastBatchFallbackMetric(registry, "redis", "ok")
return nil
}
addFastBatchFallbackMetric(registry, "redis", "skipped_transient")
return fmt.Errorf("redis fast batch update: %w", err)
}
recordFastWriterRedisE2EDurationMessages(registry, group.messages, group.envelopes, group.subject)
}
for _, msg := range validMessages {
if msg.ack != nil {
started := time.Now()
err := msg.ack()
recordFastWriterStageDuration(registry, msg.subject, "ack", statusFromError(err), time.Since(started))
if err != nil {
addFastMetric(registry, msg.subject, "ack_error")
return fmt.Errorf("nats ack: %w", err)
}
}
addFastMetric(registry, msg.subject, "ok")
}
return nil
}
for i, env := range envelopes {
msg := validMessages[i]
started := time.Now()
var result realtime.FastUpdateResult
var err error
if resultUpdater, ok := updater.(fastResultUpdater); ok {
result, err = resultUpdater.FastUpdateWithResult(ctx, env)
} else {
err = updater.FastUpdate(ctx, env)
}
recordFastWriterStageDuration(registry, msg.subject, "redis", statusFromError(err), time.Since(started))
if err != nil {
return fmt.Errorf("redis fast update: %w", err)
}
recordFastWriterRedisEnvelopeMetrics(registry, msg.subject, result)
recordFastWriterRedisFieldMetrics(registry, msg.subject, result)
recordFastWriterRedisE2EDuration(registry, msg.subject, env)
if msg.ack != nil {
started := time.Now()
err := msg.ack()
recordFastWriterStageDuration(registry, msg.subject, "ack", statusFromError(err), time.Since(started))
if err != nil {
addFastMetric(registry, msg.subject, "ack_error")
return fmt.Errorf("nats ack: %w", err)
}
}
addFastMetric(registry, msg.subject, "ok")
}
return nil
}
func processFastMessagesIndividually(ctx context.Context, registry *metrics.Registry, appender fastAppender, updater fastUpdater, messages []*fastMessage) error {
for _, msg := range messages {
if err := processFastMessageWithReceived(ctx, registry, appender, updater, msg, false); err != nil {
return err
}
}
return nil
}
func processFastMessage(ctx context.Context, registry *metrics.Registry, appender fastAppender, updater fastUpdater, msg *fastMessage) error {
return processFastMessageWithReceived(ctx, registry, appender, updater, msg, true)
}
func processFastMessageWithReceived(ctx context.Context, registry *metrics.Registry, appender fastAppender, updater fastUpdater, msg *fastMessage, recordReceived bool) error {
if recordReceived {
addFastMetric(registry, msg.subject, "received")
}
var env envelope.FrameEnvelope
if err := json.Unmarshal(msg.data, &env); err != nil {
addFastMetric(registry, msg.subject, "invalid_json")
if msg.ack != nil {
started := time.Now()
err := msg.ack()
recordFastWriterStageDuration(registry, msg.subject, "ack", statusFromError(err), time.Since(started))
if err != nil {
addFastMetric(registry, msg.subject, "ack_error")
return fmt.Errorf("nats ack invalid json: %w", err)
}
}
return nil
}
if status, err := topics.ValidateRawEnvelope(msg.subject, env); err != nil {
addFastMetric(registry, msg.subject, status)
if msg.ack != nil {
started := time.Now()
err := msg.ack()
recordFastWriterStageDuration(registry, msg.subject, "ack", statusFromError(err), time.Since(started))
if err != nil {
addFastMetric(registry, msg.subject, "ack_error")
return fmt.Errorf("nats ack mismatched raw envelope: %w", err)
}
}
return nil
}
if appender != nil {
started := time.Now()
err := appender.AppendAll(ctx, env)
recordFastWriterStageDuration(registry, msg.subject, "tdengine", statusFromError(err), time.Since(started))
if err != nil {
if isTransientFastBatchError(err) {
if catchupErr := updateFastRedisSingleWithoutAck(ctx, registry, updater, msg.subject, env); catchupErr != nil {
addFastDecoupledUpdateMetric(registry, "tdengine_transient", "error")
return fmt.Errorf("tdengine append: %w; redis catchup: %v", err, catchupErr)
}
addFastDecoupledUpdateMetric(registry, "tdengine_transient", "ok")
}
return fmt.Errorf("tdengine append: %w", err)
}
}
started := time.Now()
var result realtime.FastUpdateResult
var err error
if resultUpdater, ok := updater.(fastResultUpdater); ok {
result, err = resultUpdater.FastUpdateWithResult(ctx, env)
} else {
err = updater.FastUpdate(ctx, env)
}
recordFastWriterStageDuration(registry, msg.subject, "redis", statusFromError(err), time.Since(started))
if err != nil {
return fmt.Errorf("redis fast update: %w", err)
}
recordFastWriterRedisEnvelopeMetrics(registry, msg.subject, result)
recordFastWriterRedisFieldMetrics(registry, msg.subject, result)
recordFastWriterRedisE2EDuration(registry, msg.subject, env)
if msg.ack != nil {
started = time.Now()
err = msg.ack()
recordFastWriterStageDuration(registry, msg.subject, "ack", statusFromError(err), time.Since(started))
if err != nil {
addFastMetric(registry, msg.subject, "ack_error")
return fmt.Errorf("nats ack: %w", err)
}
}
addFastMetric(registry, msg.subject, "ok")
return nil
}
func updateFastRedisBatchWithoutAck(ctx context.Context, registry *metrics.Registry, updater fastUpdater, messages []*fastMessage, envelopes []envelope.FrameEnvelope) error {
if len(envelopes) == 0 {
return nil
}
subject := fastBatchSubject(messages)
if batchUpdater, ok := updater.(fastBatchResultUpdater); ok {
started := time.Now()
result, err := batchUpdater.FastUpdateBatchWithResult(ctx, envelopes)
recordFastWriterStageDuration(registry, subject, "redis", statusFromError(err), time.Since(started))
if err == nil {
recordFastWriterRedisEnvelopeMetrics(registry, subject, result)
recordFastWriterRedisFieldMetrics(registry, subject, result)
recordFastWriterRedisE2EDurationMessages(registry, messages, envelopes, subject)
return nil
}
if shouldFallbackFastBatchError(err) {
addFastBatchFallbackMetric(registry, "redis_catchup", "attempted")
return updateFastRedisSinglesWithoutAck(ctx, registry, updater, messages, envelopes)
}
return err
}
if batchUpdater, ok := updater.(fastBatchUpdater); ok {
started := time.Now()
err := batchUpdater.FastUpdateBatch(ctx, envelopes)
recordFastWriterStageDuration(registry, subject, "redis", statusFromError(err), time.Since(started))
if err == nil {
recordFastWriterRedisE2EDurationMessages(registry, messages, envelopes, subject)
return nil
}
if shouldFallbackFastBatchError(err) {
addFastBatchFallbackMetric(registry, "redis_catchup", "attempted")
return updateFastRedisSinglesWithoutAck(ctx, registry, updater, messages, envelopes)
}
return err
}
return updateFastRedisSinglesWithoutAck(ctx, registry, updater, messages, envelopes)
}
func updateFastRedisSinglesWithoutAck(ctx context.Context, registry *metrics.Registry, updater fastUpdater, messages []*fastMessage, envelopes []envelope.FrameEnvelope) error {
for index, env := range envelopes {
subject := fastMessageSubject(messages, index, fastBatchSubject(messages))
if err := updateFastRedisSingleWithoutAck(ctx, registry, updater, subject, env); err != nil {
return err
}
}
return nil
}
func updateFastRedisSingleWithoutAck(ctx context.Context, registry *metrics.Registry, updater fastUpdater, subject string, env envelope.FrameEnvelope) error {
started := time.Now()
var result realtime.FastUpdateResult
var err error
if resultUpdater, ok := updater.(fastResultUpdater); ok {
result, err = resultUpdater.FastUpdateWithResult(ctx, env)
} else {
err = updater.FastUpdate(ctx, env)
}
recordFastWriterStageDuration(registry, subject, "redis", statusFromError(err), time.Since(started))
if err != nil {
return err
}
recordFastWriterRedisEnvelopeMetrics(registry, subject, result)
recordFastWriterRedisFieldMetrics(registry, subject, result)
recordFastWriterRedisE2EDuration(registry, subject, env)
return nil
}
func shouldFallbackFastBatchError(err error) bool {
if err == nil || errors.Is(err, context.Canceled) || errors.Is(err, context.DeadlineExceeded) {
return false
}
return !isTransientFastBatchError(err)
}
func isTransientFastBatchError(err error) bool {
if err == nil {
return false
}
text := strings.ToLower(strings.TrimSpace(err.Error()))
return strings.Contains(text, "timeout") ||
strings.Contains(text, "temporary") ||
strings.Contains(text, "temporarily") ||
strings.Contains(text, "connection refused") ||
strings.Contains(text, "connection reset") ||
strings.Contains(text, "connection closed") ||
strings.Contains(text, "broken pipe") ||
strings.Contains(text, "bad connection") ||
strings.Contains(text, "i/o timeout") ||
text == "eof" ||
strings.Contains(text, "unexpected eof") ||
strings.Contains(text, "server is down") ||
strings.Contains(text, "network is unreachable") ||
strings.Contains(text, "no route to host")
}
func recordFastWriterConfigMetrics(registry *metrics.Registry, cfg config) {
if registry == nil {
return
}
registry.SetGauge("vehicle_fast_writer_config", metrics.Labels{"setting": "workers"}, float64(cfg.Workers))
registry.SetGauge("vehicle_fast_writer_config", metrics.Labels{"setting": "batch_size"}, float64(cfg.BatchSize))
registry.SetGauge("vehicle_fast_writer_config", metrics.Labels{"setting": "fetch_wait_ms"}, float64(cfg.FetchWait.Milliseconds()))
registry.SetGauge("vehicle_fast_writer_config", metrics.Labels{"setting": "operation_timeout_ms"}, float64(cfg.OperationWait.Milliseconds()))
}
func boolGauge(value bool) float64 {
if value {
return 1
}
return 0
}
type fastSubjectGroup struct {
subject string
messages []*fastMessage
envelopes []envelope.FrameEnvelope
}
func fastSubjectGroups(messages []*fastMessage, envelopes []envelope.FrameEnvelope) []fastSubjectGroup {
if len(envelopes) == 0 {
return nil
}
fallback := fastBatchSubject(messages)
groups := make([]fastSubjectGroup, 0, len(envelopes))
indexBySubject := make(map[string]int, len(envelopes))
for index, env := range envelopes {
subject := fastMessageSubject(messages, index, fallback)
groupIndex, exists := indexBySubject[subject]
if !exists {
groupIndex = len(groups)
indexBySubject[subject] = groupIndex
groups = append(groups, fastSubjectGroup{subject: subject})
}
groups[groupIndex].envelopes = append(groups[groupIndex].envelopes, env)
if index >= 0 && index < len(messages) {
groups[groupIndex].messages = append(groups[groupIndex].messages, messages[index])
}
}
return groups
}
func fastBatchSubject(messages []*fastMessage) string {
if len(messages) == 0 {
return "unknown"
}
subject := messages[0].subject
for _, msg := range messages[1:] {
if msg.subject != subject {
return "mixed"
}
}
if strings.TrimSpace(subject) == "" {
return "unknown"
}
return subject
}
func fastMessageSubject(messages []*fastMessage, index int, fallback string) string {
if index >= 0 && index < len(messages) && strings.TrimSpace(messages[index].subject) != "" {
return messages[index].subject
}
if strings.TrimSpace(fallback) != "" {
return fallback
}
return "unknown"
}
func ensureStream(js nats.JetStreamContext, cfg config) error {
ctx, cancel := context.WithTimeout(context.Background(), cfg.StreamEnsureWait)
defer cancel()
opts := []nats.JSOpt{nats.Context(ctx)}
stream := &nats.StreamConfig{
Name: cfg.NATSStream,
Subjects: cfg.NATSSubjects,
Storage: nats.FileStorage,
Retention: nats.LimitsPolicy,
MaxAge: cfg.StreamMaxAge,
MaxBytes: cfg.StreamMaxBytes,
Duplicates: 2 * time.Minute,
}
if _, err := js.StreamInfo(cfg.NATSStream, opts...); err == nil {
_, err = js.UpdateStream(stream, opts...)
return err
}
_, err := js.AddStream(stream, opts...)
if err == nil {
return nil
}
_, updateErr := js.UpdateStream(stream, opts...)
if updateErr == nil {
return nil
}
return err
}
func addFastMetric(registry *metrics.Registry, subject string, status string) {
if registry == nil {
return
}
labels := metrics.Labels{"subject": subject, "status": status}
registry.IncCounter("vehicle_fast_writer_messages_total", labels)
metrics.RecordLastActivity(registry, "vehicle_fast_writer_last_message_unix_seconds", labels)
}
func addFastBatchFallbackMetric(registry *metrics.Registry, stage string, status string) {
if registry == nil {
return
}
registry.IncCounter("vehicle_fast_writer_batch_fallback_total", metrics.Labels{
"stage": stage,
"status": status,
})
}
func addFastDecoupledUpdateMetric(registry *metrics.Registry, reason string, status string) {
if registry == nil {
return
}
registry.IncCounter("vehicle_fast_writer_decoupled_updates_total", metrics.Labels{
"reason": reason,
"status": status,
})
}
func addFastBatchPending(registry *metrics.Registry, messages int, envelopes int) {
if registry == nil {
return
}
fastBatchPending.Add(registry, "vehicle_fast_writer_batch_pending_messages", "vehicle_fast_writer_batch_pending_envelopes", messages, envelopes)
}
func recordFastNATSConsumerInfoMetrics(registry *metrics.Registry, cfg config, info *nats.ConsumerInfo) {
if registry == nil || info == nil {
return
}
labels := metrics.Labels{"stream": cfg.NATSStream, "consumer": cfg.NATSDurable}
registry.SetGauge("vehicle_fast_writer_nats_consumer_pending", labels, float64(info.NumPending))
registry.SetGauge("vehicle_fast_writer_nats_consumer_ack_pending", labels, float64(info.NumAckPending))
registry.SetGauge("vehicle_fast_writer_nats_consumer_waiting", labels, float64(info.NumWaiting))
}
func recordFastWriterStageDuration(registry *metrics.Registry, subject string, stage string, status string, elapsed time.Duration) {
if registry == nil {
return
}
labels := metrics.Labels{
"subject": subject,
"stage": stage,
"status": status,
}
registry.ObserveHistogram("vehicle_fast_writer_stage_duration_ms_histogram", labels, fastWriterStageDurationBucketsMS, float64(elapsed.Milliseconds()))
metrics.RecordLastActivity(registry, "vehicle_fast_writer_last_stage_unix_seconds", labels)
}
func recordFastWriterRedisE2EDurationMessages(registry *metrics.Registry, messages []*fastMessage, envelopes []envelope.FrameEnvelope, fallback string) {
for index, env := range envelopes {
recordFastWriterRedisE2EDuration(registry, fastMessageSubject(messages, index, fallback), env)
}
}
func recordFastWriterRedisE2EDuration(registry *metrics.Registry, subject string, env envelope.FrameEnvelope) {
if registry == nil || env.ReceivedAtMS <= 0 {
return
}
elapsedMS := time.Since(time.UnixMilli(env.ReceivedAtMS)).Milliseconds()
if elapsedMS < 0 {
elapsedMS = 0
}
labels := metrics.Labels{
"subject": subject,
}
registry.ObserveHistogram("vehicle_fast_writer_redis_e2e_duration_ms_histogram", labels, fastWriterRedisE2EDurationBucketsMS, float64(elapsedMS))
p99, samples := fastWriterRedisE2ERecent.Observe(subject, float64(elapsedMS))
registry.SetGauge("vehicle_fast_writer_redis_e2e_recent_p99_ms", labels, p99)
registry.SetGauge("vehicle_fast_writer_redis_e2e_recent_samples", labels, float64(samples))
metrics.RecordLastActivity(registry, "vehicle_fast_writer_last_redis_e2e_unix_seconds", labels)
}
func recordFastWriterRedisFieldMetrics(registry *metrics.Registry, subject string, result realtime.FastUpdateResult) {
if registry == nil || result.FieldsSeen == 0 {
return
}
addFastWriterRedisFieldMetric(registry, subject, "seen", result.FieldsSeen)
addFastWriterRedisFieldMetric(registry, subject, "written", result.FieldsWritten)
addFastWriterRedisFieldMetric(registry, subject, "skipped_stale", result.FieldsSkippedStale)
}
func recordFastWriterRedisEnvelopeMetrics(registry *metrics.Registry, subject string, result realtime.FastUpdateResult) {
if registry == nil || result.EnvelopesSeen == 0 {
return
}
addFastWriterRedisEnvelopeMetric(registry, subject, "seen", result.EnvelopesSeen)
addFastWriterRedisEnvelopeMetric(registry, subject, "updated", result.EnvelopesUpdated)
addFastWriterRedisEnvelopeMetric(registry, subject, "skipped_non_realtime", result.EnvelopesSkippedNonRealtime)
addFastWriterRedisEnvelopeMetric(registry, subject, "skipped_missing_vin", result.EnvelopesSkippedMissingVIN)
addFastWriterRedisEnvelopeMetric(registry, subject, "skipped_missing_vehicle_key", result.EnvelopesSkippedMissingVehicleKey)
addFastWriterRedisEnvelopeMetric(registry, subject, "skipped_missing_fields", result.EnvelopesSkippedMissingFields)
}
func addFastWriterRedisEnvelopeMetric(registry *metrics.Registry, subject string, status string, value int) {
if value <= 0 {
return
}
registry.AddCounter("vehicle_fast_writer_redis_envelopes_total", metrics.Labels{
"subject": subject,
"status": status,
}, float64(value))
}
func addFastWriterRedisFieldMetric(registry *metrics.Registry, subject string, status string, value int) {
if value <= 0 {
return
}
registry.AddCounter("vehicle_fast_writer_redis_fields_total", metrics.Labels{
"subject": subject,
"status": status,
}, float64(value))
}
func statusFromError(err error) string {
if err != nil {
return "error"
}
return "ok"
}
func env(key string, fallback string) string {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
return value
}
func envInt(key string, fallback int) int {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
var parsed int
if _, err := fmt.Sscanf(value, "%d", &parsed); err != nil {
return fallback
}
return parsed
}
func envInt64(key string, fallback int64) int64 {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
var parsed int64
if _, err := fmt.Sscanf(value, "%d", &parsed); err != nil || parsed <= 0 {
return fallback
}
return parsed
}
func envBool(key string, fallback bool) bool {
value := strings.ToLower(strings.TrimSpace(os.Getenv(key)))
switch value {
case "":
return fallback
case "1", "true", "yes", "y", "on":
return true
case "0", "false", "no", "n", "off":
return false
default:
return fallback
}
}
func splitCSV(value string) []string {
var out []string
for _, item := range strings.Split(value, ",") {
item = strings.TrimSpace(item)
if item != "" {
out = append(out, item)
}
}
return out
}

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,851 @@
package main
import (
"context"
"encoding/json"
"errors"
"fmt"
"log/slog"
"os"
"os/signal"
"sort"
"strconv"
"strings"
"syscall"
"time"
"github.com/nats-io/nats.go"
"github.com/segmentio/kafka-go"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/health"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/observability"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/realtime"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/topics"
)
func main() {
logger := observability.NewLogger("nats-kafka-bridge")
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
cfg := loadConfig()
if err := cfg.Validate(); err != nil {
logger.Error("invalid bridge config", "error", err)
os.Exit(1)
}
conn, err := nats.Connect(cfg.NATSURL, nats.Name(cfg.NATSClientName), nats.Timeout(5*time.Second))
if err != nil {
logger.Error("nats connect failed", "error", err)
os.Exit(1)
}
defer conn.Close()
registry := metrics.NewRegistry()
recordBridgeConfigMetrics(registry, cfg)
health.Start(ctx, logger, health.NewServer(env("HEALTH_ADDR", ""), "nats-kafka-bridge", []health.Check{
{Name: "nats", Check: func(context.Context) error {
if conn.Status() != nats.CONNECTED {
return fmt.Errorf("nats status is %s", conn.Status().String())
}
return nil
}},
}, registry))
js, err := conn.JetStream()
if err != nil {
logger.Error("nats jetstream init failed", "error", err)
os.Exit(1)
}
if err := ensureStream(js, cfg); err != nil {
logger.Error("nats stream ensure failed", "stream", cfg.NATSStream, "error", err)
os.Exit(1)
}
sub, err := js.PullSubscribe(
cfg.NATSFilter,
cfg.NATSDurable,
nats.BindStream(cfg.NATSStream),
nats.ManualAck(),
nats.AckWait(cfg.AckWait),
nats.MaxDeliver(-1),
)
if err != nil {
logger.Error("nats pull consumer init failed", "stream", cfg.NATSStream, "durable", cfg.NATSDurable, "error", err)
os.Exit(1)
}
logger.Info("nats kafka bridge started",
"nats_url", cfg.NATSURL,
"stream", cfg.NATSStream,
"durable", cfg.NATSDurable,
"filter", cfg.NATSFilter,
"kafka_brokers", strings.Join(cfg.KafkaBrokers, ","),
"kafka_batch_timeout_ms", cfg.KafkaBatchTimeout.Milliseconds(),
"kafka_write_concurrency", cfg.KafkaWriteConcurrency,
"fetch_wait_ms", cfg.FetchWait.Milliseconds(),
"batch_size", cfg.BatchSize,
"workers", cfg.Workers,
"derive_fields_from_raw_enabled", cfg.DeriveFieldsFromRaw)
for i := 0; i < cfg.Workers; i++ {
writer := newKafkaWriter(cfg)
defer writer.Close()
go runBridge(ctx, logger.With("worker", i), registry, js, sub, writer, cfg)
}
<-ctx.Done()
}
type config struct {
NATSURL string
NATSClientName string
NATSStream string
NATSDurable string
NATSFilter string
NATSSubjects []string
KafkaBrokers []string
RawRoutes []subjectRoute
FieldsRoutes []subjectRoute
Route map[string]string
RawFieldRoutes map[string]fieldsProjectionRoute
DeriveFieldsFromRaw bool
KafkaBatchTimeout time.Duration
KafkaWriteConcurrency int
BatchSize int
FetchWait time.Duration
OperationWait time.Duration
AckWait time.Duration
StreamMaxAge time.Duration
StreamMaxBytes int64
StreamEnsureWait time.Duration
Workers int
}
type subjectRoute struct {
Protocol envelope.Protocol
Subject string
Topic string
}
type fieldsProjectionRoute struct {
Protocol envelope.Protocol
Topic string
}
func loadConfig() config {
rawRoutes := []subjectRoute{
{Protocol: envelope.ProtocolGB32960, Subject: env("NATS_SUBJECT_GB32960_RAW", env("KAFKA_TOPIC_GB32960_RAW", topics.RawGB32960)), Topic: env("KAFKA_TOPIC_GB32960_RAW", topics.RawGB32960)},
{Protocol: envelope.ProtocolJT808, Subject: env("NATS_SUBJECT_JT808_RAW", env("KAFKA_TOPIC_JT808_RAW", topics.RawJT808)), Topic: env("KAFKA_TOPIC_JT808_RAW", topics.RawJT808)},
{Protocol: envelope.ProtocolYutongMQTT, Subject: env("NATS_SUBJECT_YUTONG_MQTT_RAW", env("KAFKA_TOPIC_YUTONG_MQTT_RAW", topics.RawYutongMQTT)), Topic: env("KAFKA_TOPIC_YUTONG_MQTT_RAW", topics.RawYutongMQTT)},
}
fieldsRoutes := []subjectRoute{
{Protocol: envelope.ProtocolGB32960, Subject: env("NATS_SUBJECT_GB32960_FIELDS", env("KAFKA_TOPIC_GB32960_FIELDS", topics.FieldsGB32960)), Topic: env("KAFKA_TOPIC_GB32960_FIELDS", topics.FieldsGB32960)},
{Protocol: envelope.ProtocolJT808, Subject: env("NATS_SUBJECT_JT808_FIELDS", env("KAFKA_TOPIC_JT808_FIELDS", topics.FieldsJT808)), Topic: env("KAFKA_TOPIC_JT808_FIELDS", topics.FieldsJT808)},
{Protocol: envelope.ProtocolYutongMQTT, Subject: env("NATS_SUBJECT_YUTONG_MQTT_FIELDS", env("KAFKA_TOPIC_YUTONG_MQTT_FIELDS", topics.FieldsYutongMQTT)), Topic: env("KAFKA_TOPIC_YUTONG_MQTT_FIELDS", topics.FieldsYutongMQTT)},
}
route := routeMap(rawRoutes, fieldsRoutes)
if unifiedSubject, unifiedTopic, ok := unifiedRouteFromEnv(); ok {
route[unifiedSubject] = unifiedTopic
}
workers := envInt("BRIDGE_WORKERS", 4)
if workers < 1 {
workers = 1
}
return config{
NATSURL: env("NATS_URL", "nats://127.0.0.1:4222"),
NATSClientName: env("NATS_CLIENT_NAME", "lingniu-nats-kafka-bridge"),
NATSStream: env("NATS_STREAM", "VEHICLE_INGEST"),
NATSDurable: env("NATS_DURABLE", "vehicle-kafka-bridge"),
NATSFilter: env("NATS_FILTER", "vehicle.>"),
NATSSubjects: splitCSV(env("NATS_STREAM_SUBJECTS", strings.Join(mapKeys(route), ","))),
KafkaBrokers: splitCSV(env("KAFKA_BROKERS", "127.0.0.1:9092")),
RawRoutes: rawRoutes,
FieldsRoutes: fieldsRoutes,
Route: route,
RawFieldRoutes: rawFieldProjectionRoutes(rawRoutes, fieldsRoutes),
DeriveFieldsFromRaw: envBool("BRIDGE_DERIVE_FIELDS_FROM_RAW_ENABLED", true),
KafkaBatchTimeout: time.Duration(envInt("BRIDGE_KAFKA_BATCH_TIMEOUT_MS", 20)) * time.Millisecond,
KafkaWriteConcurrency: envInt("BRIDGE_KAFKA_WRITE_CONCURRENCY", 6),
BatchSize: envInt("BRIDGE_BATCH_SIZE", 500),
FetchWait: time.Duration(envInt("BRIDGE_FETCH_WAIT_MS", 20)) * time.Millisecond,
OperationWait: time.Duration(envInt("BRIDGE_OPERATION_TIMEOUT_MS", 30000)) * time.Millisecond,
AckWait: time.Duration(envInt("NATS_ACK_WAIT_SECONDS", 60)) * time.Second,
StreamMaxAge: time.Duration(envInt("NATS_STREAM_MAX_AGE_HOURS", 24)) * time.Hour,
StreamMaxBytes: envInt64("NATS_STREAM_MAX_BYTES", 20*1024*1024*1024),
StreamEnsureWait: time.Duration(envInt("NATS_STREAM_ENSURE_TIMEOUT_SECONDS", 60)) * time.Second,
Workers: workers,
}
}
func recordBridgeConfigMetrics(registry *metrics.Registry, cfg config) {
if registry == nil {
return
}
registry.SetGauge("vehicle_bridge_config", metrics.Labels{"setting": "workers"}, float64(cfg.Workers))
registry.SetGauge("vehicle_bridge_config", metrics.Labels{"setting": "batch_size"}, float64(cfg.BatchSize))
registry.SetGauge("vehicle_bridge_config", metrics.Labels{"setting": "fetch_wait_ms"}, float64(cfg.FetchWait.Milliseconds()))
registry.SetGauge("vehicle_bridge_config", metrics.Labels{"setting": "kafka_batch_timeout_ms"}, float64(cfg.KafkaBatchTimeout.Milliseconds()))
registry.SetGauge("vehicle_bridge_config", metrics.Labels{"setting": "kafka_write_concurrency"}, float64(cfg.KafkaWriteConcurrency))
registry.SetGauge("vehicle_bridge_config", metrics.Labels{"setting": "derive_fields_from_raw_enabled"}, boolMetric(cfg.DeriveFieldsFromRaw))
}
func newKafkaWriter(cfg config) *kafka.Writer {
return &kafka.Writer{
Addr: kafka.TCP(cfg.KafkaBrokers...),
Balancer: &kafka.Hash{},
AllowAutoTopicCreation: false,
RequiredAcks: kafka.RequireAll,
BatchTimeout: cfg.KafkaBatchTimeout,
Async: false,
}
}
func routeMap(rawRoutes []subjectRoute, fieldsRoutes []subjectRoute) map[string]string {
route := make(map[string]string, len(rawRoutes)+len(fieldsRoutes))
for _, item := range rawRoutes {
if item.Subject != "" {
route[item.Subject] = item.Topic
}
}
for _, item := range fieldsRoutes {
if item.Subject != "" {
route[item.Subject] = item.Topic
}
}
return route
}
func rawFieldProjectionRoutes(rawRoutes []subjectRoute, fieldsRoutes []subjectRoute) map[string]fieldsProjectionRoute {
fieldsByProtocol := make(map[envelope.Protocol]string, len(fieldsRoutes))
for _, route := range fieldsRoutes {
if route.Protocol != "" && strings.TrimSpace(route.Topic) != "" {
fieldsByProtocol[route.Protocol] = strings.TrimSpace(route.Topic)
}
}
out := make(map[string]fieldsProjectionRoute, len(rawRoutes))
for _, route := range rawRoutes {
subject := strings.TrimSpace(route.Subject)
topic := fieldsByProtocol[route.Protocol]
if subject == "" || topic == "" {
continue
}
out[subject] = fieldsProjectionRoute{Protocol: route.Protocol, Topic: topic}
}
return out
}
func (c config) Validate() error {
rawSubjects, rawTopics := routeSubjectsAndTopics(c.RawRoutes)
fieldsSubjects, fieldsTopics := routeSubjectsAndTopics(c.FieldsRoutes)
if err := topics.ValidateKafkaRawFields(rawTopics, fieldsTopics); err != nil {
return err
}
return topics.ValidateRawFieldsDisjoint(rawSubjects, fieldsSubjects, "nats subject")
}
func routeSubjectsAndTopics(routes []subjectRoute) (map[string]string, map[string]string) {
subjects := make(map[string]string, len(routes))
kafkaTopics := make(map[string]string, len(routes))
for _, route := range routes {
name := strings.TrimSpace(route.Subject)
if name == "" {
name = strings.TrimSpace(route.Topic)
}
if name == "" {
name = "unknown"
}
subjects[name] = strings.TrimSpace(route.Subject)
topicName := routeProtocolName(route)
if topicName == "" {
topicName = name
}
kafkaTopics[topicName] = strings.TrimSpace(route.Topic)
}
return subjects, kafkaTopics
}
func routeProtocolName(route subjectRoute) string {
if protocol := strings.TrimSpace(string(route.Protocol)); protocol != "" {
return protocol
}
if protocol, ok := topics.ProtocolForKnownRawTopic(route.Topic); ok {
return protocol
}
if protocol, ok := topics.ProtocolForKnownFieldsTopic(route.Topic); ok {
return protocol
}
return ""
}
func unifiedRouteFromEnv() (string, string, bool) {
subject, hasSubject := envOptional("NATS_SUBJECT_UNIFIED")
topic, hasTopic := envOptional("KAFKA_TOPIC_UNIFIED")
if !hasSubject && !hasTopic {
return "", "", false
}
if subject == "" {
subject = topic
}
if topic == "" {
topic = subject
}
if subject == "" || topic == "" {
return "", "", false
}
return subject, topic, true
}
type natsPullSubscription interface {
Fetch(int, ...nats.PullOpt) ([]*nats.Msg, error)
}
type natsConsumerInfoReader interface {
ConsumerInfo(stream string, name string, opts ...nats.JSOpt) (*nats.ConsumerInfo, error)
}
type kafkaBatchWriter interface {
WriteMessages(context.Context, ...kafka.Message) error
}
type bridgeMessage struct {
subject string
data []byte
ack func() error
}
type routedBridgeMessage struct {
sourceIndex int
kafka kafka.Message
receivedAtMS int64
}
type bridgeSourceState struct {
source bridgeMessage
routed bool
failed bool
}
func runBridge(ctx context.Context, logger *slog.Logger, registry *metrics.Registry, infoReader natsConsumerInfoReader, sub natsPullSubscription, writer kafkaBatchWriter, cfg config) {
var lastConsumerInfoAt time.Time
for {
if ctx.Err() != nil {
return
}
if time.Since(lastConsumerInfoAt) >= time.Duration(envInt("NATS_CONSUMER_METRICS_INTERVAL_SECONDS", 10))*time.Second {
lastConsumerInfoAt = time.Now()
if infoReader != nil {
info, err := infoReader.ConsumerInfo(cfg.NATSStream, cfg.NATSDurable, nats.Context(ctx))
if err != nil {
logger.Warn("nats consumer info failed", "stream", cfg.NATSStream, "durable", cfg.NATSDurable, "error", err)
} else {
recordNATSConsumerInfoMetrics(registry, cfg, info)
}
}
}
msgs, err := sub.Fetch(cfg.BatchSize, nats.MaxWait(cfg.FetchWait))
if err != nil {
if isBridgeShutdownFetchError(ctx, err) {
return
}
if errors.Is(err, nats.ErrTimeout) {
continue
}
if isTransientBridgeFetchError(err) {
logger.Warn("nats fetch interrupted", "error", err)
time.Sleep(time.Second)
continue
}
logger.Error("nats fetch failed", "error", err)
time.Sleep(time.Second)
continue
}
bridgeMessages := make([]bridgeMessage, 0, len(msgs))
for _, msg := range msgs {
msg := msg
bridgeMessages = append(bridgeMessages, bridgeMessage{
subject: msg.Subject,
data: msg.Data,
ack: func() error {
return msg.Ack()
},
})
}
operationCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), cfg.OperationWait)
err = bridgeBatchWithProjectionConcurrency(operationCtx, registry, writer, bridgeMessages, cfg.Route, cfg.RawFieldRoutes, cfg.DeriveFieldsFromRaw, cfg.KafkaWriteConcurrency)
cancel()
if err != nil {
logger.Error("bridge batch failed", "count", len(bridgeMessages), "error", err)
continue
}
}
}
func isBridgeShutdownFetchError(ctx context.Context, err error) bool {
if err == nil {
return false
}
if ctx.Err() != nil {
return true
}
return errors.Is(err, nats.ErrConnectionClosed)
}
func isTransientBridgeFetchError(err error) bool {
if err == nil {
return false
}
text := strings.ToLower(strings.TrimSpace(err.Error()))
return strings.Contains(text, "disconnected during fetch") ||
strings.Contains(text, "connection closed") ||
strings.Contains(text, "connection reset") ||
strings.Contains(text, "broken pipe") ||
strings.Contains(text, "temporary") ||
strings.Contains(text, "temporarily") ||
strings.Contains(text, "timeout")
}
func ensureStream(js nats.JetStreamContext, cfg config) error {
ctx, cancel := context.WithTimeout(context.Background(), cfg.StreamEnsureWait)
defer cancel()
opts := []nats.JSOpt{nats.Context(ctx)}
stream := &nats.StreamConfig{
Name: cfg.NATSStream,
Subjects: cfg.NATSSubjects,
Storage: nats.FileStorage,
Retention: nats.LimitsPolicy,
MaxAge: cfg.StreamMaxAge,
MaxBytes: cfg.StreamMaxBytes,
Duplicates: 2 * time.Minute,
}
if _, err := js.StreamInfo(cfg.NATSStream, opts...); err == nil {
_, err = js.UpdateStream(stream, opts...)
return err
}
_, err := js.AddStream(stream, opts...)
if err == nil {
return nil
}
_, updateErr := js.UpdateStream(stream, opts...)
if updateErr == nil {
return nil
}
return err
}
func bridgeBatch(ctx context.Context, registry *metrics.Registry, writer kafkaBatchWriter, messages []bridgeMessage, route map[string]string) error {
return bridgeBatchWithProjectionConcurrency(ctx, registry, writer, messages, route, nil, false, 6)
}
func bridgeBatchWithProjection(
ctx context.Context,
registry *metrics.Registry,
writer kafkaBatchWriter,
messages []bridgeMessage,
route map[string]string,
rawFieldRoutes map[string]fieldsProjectionRoute,
deriveFieldsFromRaw bool,
) error {
return bridgeBatchWithProjectionConcurrency(ctx, registry, writer, messages, route, rawFieldRoutes, deriveFieldsFromRaw, 6)
}
func bridgeBatchWithProjectionConcurrency(
ctx context.Context,
registry *metrics.Registry,
writer kafkaBatchWriter,
messages []bridgeMessage,
route map[string]string,
rawFieldRoutes map[string]fieldsProjectionRoute,
deriveFieldsFromRaw bool,
kafkaWriteConcurrency int,
) error {
if len(messages) == 0 {
return nil
}
started := time.Now()
status := "ok"
defer func() {
recordBridgeBatchDuration(registry, status, time.Since(started))
}()
routed := make([]routedBridgeMessage, 0, len(messages)*2)
sources := make([]bridgeSourceState, len(messages))
for sourceIndex, message := range messages {
sources[sourceIndex].source = message
addBridgeSubjectMetric(registry, "vehicle_bridge_messages_total", message.subject, "received")
topic, ok := route[message.subject]
if !ok || topic == "" {
addBridgeSubjectMetric(registry, "vehicle_bridge_messages_total", message.subject, "route_error")
if message.ack != nil {
if err := message.ack(); err != nil {
addBridgeSubjectMetric(registry, "vehicle_bridge_nats_acks_total", message.subject, "error")
status = "error"
return fmt.Errorf("ack unrouted nats subject %q: %w", message.subject, err)
}
addBridgeSubjectMetric(registry, "vehicle_bridge_nats_acks_total", message.subject, "dropped_route_error")
}
continue
}
kafkaMessage, receivedAtMS, decoded, decodeErr := kafkaMessageWithEnvelope(topic, message.data)
sources[sourceIndex].routed = true
routed = append(routed, routedBridgeMessage{
sourceIndex: sourceIndex,
kafka: kafkaMessage,
receivedAtMS: receivedAtMS,
})
if !deriveFieldsFromRaw {
continue
}
projection, ok := rawFieldRoutes[message.subject]
if !ok {
continue
}
fieldsMessage, fieldsReceivedAtMS, fieldCount, projectionStatus, projected := projectRawFields(message.subject, decoded, decodeErr, projection)
recordBridgeFieldsProjection(registry, projection.Protocol, projectionStatus, fieldCount)
if !projected {
continue
}
routed = append(routed, routedBridgeMessage{
sourceIndex: sourceIndex,
kafka: fieldsMessage,
receivedAtMS: fieldsReceivedAtMS,
})
}
if len(routed) == 0 {
addBridgeBatchPending(registry, 0, 0)
return nil
}
addBridgeBatchPending(registry, len(messages), len(routed))
defer addBridgeBatchPending(registry, -len(messages), -len(routed))
groups := groupRoutedBridgeMessagesByTopic(routed)
results := writeBridgeKafkaGroups(ctx, writer, groups, kafkaWriteConcurrency)
var firstErr error
for _, result := range results {
group := result.group
err := result.err
recordBridgeKafkaWriteDuration(registry, group[0].kafka.Topic, statusFromError(err), result.elapsed)
if err != nil {
for _, item := range group {
addBridgeTopicMetric(registry, "vehicle_bridge_kafka_writes_total", item.kafka.Topic, "error")
sources[item.sourceIndex].failed = true
}
status = "error"
if firstErr == nil {
firstErr = fmt.Errorf("kafka write topic %s: %w", group[0].kafka.Topic, err)
}
continue
}
for _, item := range group {
addBridgeTopicMetric(registry, "vehicle_bridge_kafka_writes_total", item.kafka.Topic, "ok")
recordBridgeKafkaE2EDuration(registry, item.kafka.Topic, item.receivedAtMS)
}
}
for i := range sources {
source := &sources[i]
if !source.routed || source.failed || source.source.ack == nil {
continue
}
if err := source.source.ack(); err != nil {
addBridgeSubjectMetric(registry, "vehicle_bridge_nats_acks_total", source.source.subject, "error")
status = "error"
return err
}
addBridgeSubjectMetric(registry, "vehicle_bridge_nats_acks_total", source.source.subject, "ok")
}
return firstErr
}
type bridgeKafkaWriteResult struct {
group []routedBridgeMessage
elapsed time.Duration
err error
}
func writeBridgeKafkaGroups(ctx context.Context, writer kafkaBatchWriter, groups [][]routedBridgeMessage, concurrency int) []bridgeKafkaWriteResult {
if len(groups) == 0 {
return nil
}
if concurrency < 1 {
concurrency = 1
}
if concurrency > len(groups) {
concurrency = len(groups)
}
semaphore := make(chan struct{}, concurrency)
results := make(chan bridgeKafkaWriteResult, len(groups))
for _, group := range groups {
group := group
semaphore <- struct{}{}
go func() {
defer func() { <-semaphore }()
messages := make([]kafka.Message, 0, len(group))
for _, item := range group {
messages = append(messages, item.kafka)
}
started := time.Now()
err := writer.WriteMessages(ctx, messages...)
results <- bridgeKafkaWriteResult{group: group, elapsed: time.Since(started), err: err}
}()
}
out := make([]bridgeKafkaWriteResult, 0, len(groups))
for range groups {
out = append(out, <-results)
}
return out
}
func projectRawFields(subject string, raw envelope.FrameEnvelope, decodeErr error, projection fieldsProjectionRoute) (kafka.Message, int64, int, string, bool) {
if decodeErr != nil {
return kafka.Message{}, 0, 0, "invalid_json", false
}
if _, err := topics.ValidateRawEnvelope(subject, raw); err != nil {
return kafka.Message{}, 0, 0, "invalid_envelope", false
}
if raw.Protocol != projection.Protocol {
return kafka.Message{}, 0, 0, "protocol_mismatch", false
}
if !envelope.IsRealtimeTelemetryFrame(raw) {
return kafka.Message{}, 0, 0, "skipped_non_realtime", false
}
if len(raw.ParsedFields) == 0 {
return kafka.Message{}, 0, 0, "skipped_missing_fields", false
}
fields, ok := realtime.BuildFieldsEnvelope(raw)
if !ok || len(fields.Fields) == 0 {
return kafka.Message{}, 0, 0, "skipped_missing_fields", false
}
if _, err := topics.ValidateFieldsEnvelope(projection.Topic, fields); err != nil {
return kafka.Message{}, 0, 0, "invalid_fields_envelope", false
}
payload, err := fields.MarshalJSONBytes()
if err != nil {
return kafka.Message{}, 0, 0, "marshal_error", false
}
result := kafka.Message{Topic: projection.Topic, Key: fields.KafkaKey(), Value: payload}
return result, fields.ReceivedAtMS, len(fields.Fields), "published", true
}
func groupRoutedBridgeMessagesByTopic(messages []routedBridgeMessage) [][]routedBridgeMessage {
if len(messages) == 0 {
return nil
}
groupsByTopic := make(map[string][]routedBridgeMessage)
for _, message := range messages {
groupsByTopic[message.kafka.Topic] = append(groupsByTopic[message.kafka.Topic], message)
}
topics := make([]string, 0, len(groupsByTopic))
for topic := range groupsByTopic {
topics = append(topics, topic)
}
sort.Strings(topics)
groups := make([][]routedBridgeMessage, 0, len(topics))
for _, topic := range topics {
groups = append(groups, groupsByTopic[topic])
}
return groups
}
var bridgeBatchDurationBucketsMS = []float64{1, 5, 10, 25, 50, 100, 250, 500, 1000, 5000}
var bridgeKafkaWriteDurationBucketsMS = []float64{1, 5, 10, 25, 50, 100, 250, 500, 1000, 5000}
var bridgeKafkaE2EDurationBucketsMS = []float64{10, 25, 50, 100, 250, 500, 1000, 2500, 5000, 10000}
var bridgeKafkaE2ERecent = metrics.NewRecentLatencyByKey(512)
var bridgeBatchPending = metrics.PendingPairGauge{}
func addBridgeSubjectMetric(registry *metrics.Registry, name string, subject string, status string) {
if registry == nil {
return
}
labels := metrics.Labels{"subject": subject, "status": status}
registry.IncCounter(name, labels)
if name == "vehicle_bridge_messages_total" {
metrics.RecordLastActivity(registry, "vehicle_bridge_last_message_unix_seconds", labels)
return
}
if name == "vehicle_bridge_nats_acks_total" {
metrics.RecordLastActivity(registry, "vehicle_bridge_last_ack_unix_seconds", labels)
}
}
func addBridgeTopicMetric(registry *metrics.Registry, name string, topic string, status string) {
if registry == nil {
return
}
labels := metrics.Labels{"topic": topic, "status": status}
registry.IncCounter(name, labels)
if name == "vehicle_bridge_kafka_writes_total" {
metrics.RecordLastActivity(registry, "vehicle_bridge_last_kafka_write_unix_seconds", labels)
}
}
func recordBridgeFieldsProjection(registry *metrics.Registry, protocol envelope.Protocol, status string, fieldCount int) {
if registry == nil {
return
}
protocolLabel := strings.TrimSpace(string(protocol))
if protocolLabel == "" {
protocolLabel = "unknown"
}
if strings.TrimSpace(status) == "" {
status = "unknown"
}
labels := metrics.Labels{"protocol": protocolLabel, "status": status}
registry.IncCounter("vehicle_bridge_fields_projection_total", labels)
metrics.RecordLastActivity(registry, "vehicle_bridge_last_fields_projection_unix_seconds", labels)
if status == "published" {
registry.SetGauge("vehicle_bridge_fields_projection_count", labels, float64(fieldCount))
}
}
func addBridgeBatchPending(registry *metrics.Registry, messages int, kafkaMessages int) {
if registry == nil {
return
}
bridgeBatchPending.Add(registry, "vehicle_bridge_batch_pending_messages", "vehicle_bridge_batch_pending_kafka_messages", messages, kafkaMessages)
}
func recordBridgeBatchDuration(registry *metrics.Registry, status string, elapsed time.Duration) {
if registry == nil {
return
}
registry.ObserveHistogram("vehicle_bridge_batch_duration_ms_histogram", metrics.Labels{
"status": status,
}, bridgeBatchDurationBucketsMS, float64(elapsed.Milliseconds()))
}
func recordBridgeKafkaWriteDuration(registry *metrics.Registry, topic string, status string, elapsed time.Duration) {
if registry == nil {
return
}
registry.ObserveHistogram("vehicle_bridge_kafka_write_duration_ms_histogram", metrics.Labels{
"topic": topic,
"status": status,
}, bridgeKafkaWriteDurationBucketsMS, float64(elapsed.Milliseconds()))
}
func recordBridgeKafkaE2EDuration(registry *metrics.Registry, topic string, receivedAtMS int64) {
if registry == nil || receivedAtMS <= 0 {
return
}
elapsedMS := time.Since(time.UnixMilli(receivedAtMS)).Milliseconds()
if elapsedMS < 0 {
elapsedMS = 0
}
labels := metrics.Labels{
"topic": topic,
}
registry.ObserveHistogram("vehicle_bridge_kafka_e2e_duration_ms_histogram", labels, bridgeKafkaE2EDurationBucketsMS, float64(elapsedMS))
p99, samples := bridgeKafkaE2ERecent.Observe(topic, float64(elapsedMS))
registry.SetGauge("vehicle_bridge_kafka_e2e_recent_p99_ms", labels, p99)
registry.SetGauge("vehicle_bridge_kafka_e2e_recent_samples", labels, float64(samples))
metrics.RecordLastActivity(registry, "vehicle_bridge_last_kafka_e2e_unix_seconds", labels)
}
func statusFromError(err error) string {
if err != nil {
return "error"
}
return "ok"
}
func recordNATSConsumerInfoMetrics(registry *metrics.Registry, cfg config, info *nats.ConsumerInfo) {
if registry == nil || info == nil {
return
}
labels := metrics.Labels{"stream": cfg.NATSStream, "consumer": cfg.NATSDurable}
registry.SetGauge("vehicle_bridge_nats_consumer_pending", labels, float64(info.NumPending))
registry.SetGauge("vehicle_bridge_nats_consumer_ack_pending", labels, float64(info.NumAckPending))
registry.SetGauge("vehicle_bridge_nats_consumer_waiting", labels, float64(info.NumWaiting))
}
func kafkaMessage(topic string, data []byte) kafka.Message {
message, _ := kafkaMessageWithReceivedAt(topic, data)
return message
}
func kafkaMessageWithReceivedAt(topic string, data []byte) (kafka.Message, int64) {
message, receivedAtMS, _, _ := kafkaMessageWithEnvelope(topic, data)
return message, receivedAtMS
}
func kafkaMessageWithEnvelope(topic string, data []byte) (kafka.Message, int64, envelope.FrameEnvelope, error) {
var env envelope.FrameEnvelope
message := kafka.Message{Topic: topic, Value: data}
if err := json.Unmarshal(data, &env); err != nil {
return message, 0, envelope.FrameEnvelope{}, err
}
message.Key = env.KafkaKey()
return message, env.ReceivedAtMS, env, nil
}
func env(key string, fallback string) string {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
return value
}
func envBool(key string, fallback bool) bool {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
parsed, err := strconv.ParseBool(value)
if err != nil {
return fallback
}
return parsed
}
func boolMetric(value bool) float64 {
if value {
return 1
}
return 0
}
func envOptional(key string) (string, bool) {
value, ok := os.LookupEnv(key)
if !ok {
return "", false
}
return strings.TrimSpace(value), true
}
func envInt(key string, fallback int) int {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
parsed, err := strconv.Atoi(value)
if err != nil || parsed <= 0 {
return fallback
}
return parsed
}
func envInt64(key string, fallback int64) int64 {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
parsed, err := strconv.ParseInt(value, 10, 64)
if err != nil || parsed <= 0 {
return fallback
}
return parsed
}
func splitCSV(value string) []string {
var out []string
for _, item := range strings.Split(value, ",") {
item = strings.TrimSpace(item)
if item != "" {
out = append(out, item)
}
}
return out
}
func mapKeys(values map[string]string) []string {
out := make([]string, 0, len(values))
for key := range values {
out = append(out, key)
}
return out
}

View File

@@ -0,0 +1,850 @@
package main
import (
"context"
"encoding/json"
"errors"
"strings"
"sync"
"sync/atomic"
"testing"
"time"
"github.com/nats-io/nats.go"
"github.com/segmentio/kafka-go"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/topics"
)
func TestBridgeBatchWritesKafkaThenAcks(t *testing.T) {
env := envelope.FrameEnvelope{Protocol: envelope.ProtocolJT808, Phone: "13307795425", MessageID: "0x0200"}
payload, err := json.Marshal(env)
if err != nil {
t.Fatal(err)
}
writer := &recordingBridgeWriter{}
acked := 0
err = bridgeBatch(context.Background(), nil, writer, []bridgeMessage{
{subject: "vehicle.raw.jt808.v1", data: payload, ack: func() error {
acked++
return nil
}},
}, map[string]string{"vehicle.raw.jt808.v1": "vehicle.raw.jt808.v1"})
if err != nil {
t.Fatalf("bridgeBatch() error = %v", err)
}
if len(writer.messages) != 1 {
t.Fatalf("kafka writes = %d, want 1", len(writer.messages))
}
if got, want := writer.messages[0].Topic, "vehicle.raw.jt808.v1"; got != want {
t.Fatalf("topic = %q, want %q", got, want)
}
if got, want := string(writer.messages[0].Key), "JT808:13307795425"; got != want {
t.Fatalf("key = %q, want %q", got, want)
}
if acked != 1 {
t.Fatalf("acks = %d, want 1", acked)
}
}
func TestBridgeBatchDoesNotAckWhenKafkaFails(t *testing.T) {
writer := &recordingBridgeWriter{err: errors.New("kafka unavailable")}
acked := 0
err := bridgeBatch(context.Background(), nil, writer, []bridgeMessage{
{subject: "vehicle.event.unified.v1", data: []byte(`{"protocol":"JT808"}`), ack: func() error {
acked++
return nil
}},
}, map[string]string{"vehicle.event.unified.v1": "vehicle.event.unified.v1"})
if err == nil {
t.Fatal("bridgeBatch() error is nil, want kafka error")
}
if acked != 0 {
t.Fatalf("acks = %d, want 0", acked)
}
}
func TestBridgeBatchProjectsFieldsFromCanonicalRawAndAcksOnce(t *testing.T) {
raw := envelope.FrameEnvelope{
EventID: "raw-event-1",
EventKind: envelope.EventKindRaw,
Protocol: envelope.ProtocolJT808,
MessageID: "0x0200",
VIN: "LTEST000000000001",
EventTimeMS: 1_700_000_000_000,
ReceivedAtMS: 1_700_000_000_100,
ParseStatus: envelope.ParseOK,
ParsedFields: map[string]any{
"jt808.location.speed_kmh": 52.3,
"jt808.location.total_mileage_km": 12345.6,
},
}
payload, err := raw.MarshalJSONBytes()
if err != nil {
t.Fatal(err)
}
acked := 0
writer := &recordingBridgeWriter{}
registry := metrics.NewRegistry()
err = bridgeBatchWithProjection(
context.Background(),
registry,
writer,
[]bridgeMessage{{subject: topics.RawJT808, data: payload, ack: func() error { acked++; return nil }}},
map[string]string{topics.RawJT808: topics.RawJT808},
map[string]fieldsProjectionRoute{topics.RawJT808: {Protocol: envelope.ProtocolJT808, Topic: topics.FieldsJT808}},
true,
)
if err != nil {
t.Fatalf("bridgeBatchWithProjection() error = %v", err)
}
if acked != 1 {
t.Fatalf("acks = %d, want exactly one ack for raw plus derived fields", acked)
}
if len(writer.messages) != 2 {
t.Fatalf("kafka messages = %d, want raw and fields", len(writer.messages))
}
byTopic := map[string]kafka.Message{}
for _, message := range writer.messages {
byTopic[message.Topic] = message
}
if len(byTopic[topics.RawJT808].Value) == 0 || len(byTopic[topics.FieldsJT808].Value) == 0 {
t.Fatalf("projected topics = %#v", byTopic)
}
var fields envelope.FrameEnvelope
if err := json.Unmarshal(byTopic[topics.FieldsJT808].Value, &fields); err != nil {
t.Fatal(err)
}
if fields.EventKind != envelope.EventKindFields || fields.SourceEventID != raw.EventID {
t.Fatalf("fields envelope = %#v", fields)
}
if got := fields.Fields["jt808.location.total_mileage_km"]; got != 12345.6 {
t.Fatalf("projected mileage = %#v", got)
}
metricText := registry.Render()
for _, want := range []string{
`vehicle_bridge_fields_projection_total{protocol="JT808",status="published"} 1`,
`vehicle_bridge_fields_projection_count{protocol="JT808",status="published"} 2`,
`vehicle_bridge_nats_acks_total{status="ok",subject="vehicle.raw.go.jt808.v1"} 1`,
} {
if !strings.Contains(metricText, want) {
t.Fatalf("projection metric missing %s:\n%s", want, metricText)
}
}
}
func TestBridgeBatchDoesNotAckRawWhenDerivedFieldsWriteFails(t *testing.T) {
raw := envelope.FrameEnvelope{
EventKind: envelope.EventKindRaw,
Protocol: envelope.ProtocolJT808,
MessageID: "0x0200",
VIN: "LTEST000000000001",
EventTimeMS: 1_700_000_000_000,
ReceivedAtMS: 1_700_000_000_100,
ParseStatus: envelope.ParseOK,
ParsedFields: map[string]any{"jt808.location.total_mileage_km": 12345.6},
}
payload, err := raw.MarshalJSONBytes()
if err != nil {
t.Fatal(err)
}
acked := 0
writer := &recordingBridgeWriter{topicErr: map[string]error{topics.FieldsJT808: errors.New("fields unavailable")}}
err = bridgeBatchWithProjection(
context.Background(),
nil,
writer,
[]bridgeMessage{{subject: topics.RawJT808, data: payload, ack: func() error { acked++; return nil }}},
map[string]string{topics.RawJT808: topics.RawJT808},
map[string]fieldsProjectionRoute{topics.RawJT808: {Protocol: envelope.ProtocolJT808, Topic: topics.FieldsJT808}},
true,
)
if err == nil {
t.Fatal("bridgeBatchWithProjection() error = nil, want fields Kafka error")
}
if acked != 0 {
t.Fatalf("acks = %d, want raw left pending until both Kafka outputs succeed", acked)
}
if len(writer.messages) != 1 || writer.messages[0].Topic != topics.RawJT808 {
t.Fatalf("successful kafka messages = %#v, want only raw before replay", writer.messages)
}
}
func TestBridgeBatchSkipsFieldsProjectionForNonRealtimeRaw(t *testing.T) {
raw := envelope.FrameEnvelope{
EventKind: envelope.EventKindRaw,
Protocol: envelope.ProtocolJT808,
MessageID: "0x0100",
Phone: "13307795425",
EventTimeMS: 1_700_000_000_000,
ReceivedAtMS: 1_700_000_000_100,
ParseStatus: envelope.ParseOK,
}
payload, err := raw.MarshalJSONBytes()
if err != nil {
t.Fatal(err)
}
acked := 0
writer := &recordingBridgeWriter{}
registry := metrics.NewRegistry()
err = bridgeBatchWithProjection(
context.Background(),
registry,
writer,
[]bridgeMessage{{subject: topics.RawJT808, data: payload, ack: func() error { acked++; return nil }}},
map[string]string{topics.RawJT808: topics.RawJT808},
map[string]fieldsProjectionRoute{topics.RawJT808: {Protocol: envelope.ProtocolJT808, Topic: topics.FieldsJT808}},
true,
)
if err != nil {
t.Fatalf("bridgeBatchWithProjection() error = %v", err)
}
if acked != 1 || len(writer.messages) != 1 || writer.messages[0].Topic != topics.RawJT808 {
t.Fatalf("acks=%d messages=%#v", acked, writer.messages)
}
if text := registry.Render(); !strings.Contains(text, `vehicle_bridge_fields_projection_total{protocol="JT808",status="skipped_non_realtime"} 1`) {
t.Fatalf("non-realtime projection metric missing:\n%s", text)
}
}
func TestBridgeBatchAcksSuccessfulTopicWhenAnotherTopicFails(t *testing.T) {
rawEnv := envelope.FrameEnvelope{Protocol: envelope.ProtocolJT808, Phone: "13307795425", MessageID: "0x0200"}
fieldsEnv := envelope.FrameEnvelope{Protocol: envelope.ProtocolJT808, Phone: "13307795425", MessageID: "0x0200"}
rawPayload, err := json.Marshal(rawEnv)
if err != nil {
t.Fatal(err)
}
fieldsPayload, err := json.Marshal(fieldsEnv)
if err != nil {
t.Fatal(err)
}
registry := metrics.NewRegistry()
writer := &recordingBridgeWriter{
topicErr: map[string]error{
"vehicle.fields.go.jt808.v1": errors.New("fields topic unavailable"),
},
}
acks := map[string]int{}
err = bridgeBatch(context.Background(), registry, writer, []bridgeMessage{
{subject: "vehicle.fields.go.jt808.v1", data: fieldsPayload, ack: func() error {
acks["fields"]++
return nil
}},
{subject: "vehicle.raw.go.jt808.v1", data: rawPayload, ack: func() error {
acks["raw"]++
return nil
}},
}, map[string]string{
"vehicle.fields.go.jt808.v1": "vehicle.fields.go.jt808.v1",
"vehicle.raw.go.jt808.v1": "vehicle.raw.go.jt808.v1",
})
if err == nil {
t.Fatal("bridgeBatch() error = nil, want failed fields topic")
}
if len(writer.calls) != 2 {
t.Fatalf("writer calls = %d, want one call per topic", len(writer.calls))
}
if len(writer.messages) != 1 || writer.messages[0].Topic != "vehicle.raw.go.jt808.v1" {
t.Fatalf("successful kafka messages = %#v, want only raw topic", writer.messages)
}
if acks["raw"] != 1 || acks["fields"] != 0 {
t.Fatalf("acks = %#v, want raw acked and fields left unacked", acks)
}
text := registry.Render()
for _, want := range []string{
`vehicle_bridge_kafka_writes_total{status="error",topic="vehicle.fields.go.jt808.v1"} 1`,
`vehicle_bridge_kafka_writes_total{status="ok",topic="vehicle.raw.go.jt808.v1"} 1`,
`vehicle_bridge_nats_acks_total{status="ok",subject="vehicle.raw.go.jt808.v1"} 1`,
} {
if !strings.Contains(text, want) {
t.Fatalf("partial bridge metric missing %s:\n%s", want, text)
}
}
if strings.Contains(text, `vehicle_bridge_nats_acks_total{status="ok",subject="vehicle.fields.go.jt808.v1"}`) {
t.Fatalf("failed topic should not be acked:\n%s", text)
}
}
func TestBridgeBatchAcksAndDropsUnroutedSubject(t *testing.T) {
writer := &recordingBridgeWriter{}
acked := 0
registry := metrics.NewRegistry()
err := bridgeBatch(context.Background(), registry, writer, []bridgeMessage{
{subject: "vehicle.unconfigured.v1", data: []byte(`{"protocol":"JT808"}`), ack: func() error {
acked++
return nil
}},
}, map[string]string{"vehicle.raw.go.jt808.v1": "vehicle.raw.go.jt808.v1"})
if err != nil {
t.Fatalf("bridgeBatch() error = %v", err)
}
if acked != 1 {
t.Fatalf("acks = %d, want unrouted message acked", acked)
}
if len(writer.messages) != 0 {
t.Fatalf("kafka writes = %d, want 0", len(writer.messages))
}
text := registry.Render()
for _, want := range []string{
`vehicle_bridge_messages_total{status="route_error",subject="vehicle.unconfigured.v1"} 1`,
`vehicle_bridge_nats_acks_total{status="dropped_route_error",subject="vehicle.unconfigured.v1"} 1`,
`vehicle_bridge_batch_pending_kafka_messages 0`,
} {
if !strings.Contains(text, want) {
t.Fatalf("metrics missing %s:\n%s", want, text)
}
}
}
func TestBridgeBatchKeepsValidMessagesWhenUnroutedSubjectIsPresent(t *testing.T) {
env := envelope.FrameEnvelope{Protocol: envelope.ProtocolJT808, Phone: "13307795425", MessageID: "0x0200"}
payload, err := json.Marshal(env)
if err != nil {
t.Fatal(err)
}
writer := &recordingBridgeWriter{}
acked := map[string]int{}
err = bridgeBatch(context.Background(), nil, writer, []bridgeMessage{
{subject: "vehicle.unconfigured.v1", data: []byte(`{"protocol":"JT808"}`), ack: func() error {
acked["unknown"]++
return nil
}},
{subject: "vehicle.raw.go.jt808.v1", data: payload, ack: func() error {
acked["valid"]++
return nil
}},
}, map[string]string{"vehicle.raw.go.jt808.v1": "vehicle.raw.go.jt808.v1"})
if err != nil {
t.Fatalf("bridgeBatch() error = %v", err)
}
if len(writer.messages) != 1 {
t.Fatalf("kafka writes = %d, want 1 valid message", len(writer.messages))
}
if writer.messages[0].Topic != "vehicle.raw.go.jt808.v1" {
t.Fatalf("topic = %q", writer.messages[0].Topic)
}
if acked["unknown"] != 1 || acked["valid"] != 1 {
t.Fatalf("acks = %#v, want both messages acked", acked)
}
}
func TestBridgeBatchReturnsErrorWhenUnroutedAckFails(t *testing.T) {
writer := &recordingBridgeWriter{}
err := bridgeBatch(context.Background(), nil, writer, []bridgeMessage{
{subject: "vehicle.unconfigured.v1", data: []byte(`{"protocol":"JT808"}`), ack: func() error {
return errors.New("nats ack unavailable")
}},
}, map[string]string{"vehicle.raw.go.jt808.v1": "vehicle.raw.go.jt808.v1"})
if err == nil {
t.Fatal("bridgeBatch() error = nil, want ack error")
}
if len(writer.messages) != 0 {
t.Fatalf("kafka writes = %d, want 0", len(writer.messages))
}
}
func TestBridgeBatchRecordsMetrics(t *testing.T) {
env := envelope.FrameEnvelope{
Protocol: envelope.ProtocolJT808,
Phone: "13307795425",
MessageID: "0x0200",
ReceivedAtMS: time.Now().Add(-20 * time.Millisecond).UnixMilli(),
}
payload, err := json.Marshal(env)
if err != nil {
t.Fatal(err)
}
registry := metrics.NewRegistry()
writer := &recordingBridgeWriter{}
err = bridgeBatch(context.Background(), registry, writer, []bridgeMessage{
{subject: "vehicle.raw.jt808.v1", data: payload, ack: func() error { return nil }},
}, map[string]string{"vehicle.raw.jt808.v1": "vehicle.raw.jt808.v1"})
if err != nil {
t.Fatalf("bridgeBatch() error = %v", err)
}
text := registry.Render()
for _, want := range []string{
`vehicle_bridge_messages_total{status="received",subject="vehicle.raw.jt808.v1"} 1`,
`vehicle_bridge_last_message_unix_seconds{status="received",subject="vehicle.raw.jt808.v1"} `,
`vehicle_bridge_kafka_writes_total{status="ok",topic="vehicle.raw.jt808.v1"} 1`,
`vehicle_bridge_last_kafka_write_unix_seconds{status="ok",topic="vehicle.raw.jt808.v1"} `,
`vehicle_bridge_kafka_write_duration_ms_histogram_count{status="ok",topic="vehicle.raw.jt808.v1"} 1`,
`vehicle_bridge_kafka_e2e_duration_ms_histogram_count{topic="vehicle.raw.jt808.v1"} 1`,
`vehicle_bridge_kafka_e2e_recent_p99_ms{topic="vehicle.raw.jt808.v1"} `,
`vehicle_bridge_kafka_e2e_recent_samples{topic="vehicle.raw.jt808.v1"} `,
`vehicle_bridge_last_kafka_e2e_unix_seconds{topic="vehicle.raw.jt808.v1"} `,
`vehicle_bridge_nats_acks_total{status="ok",subject="vehicle.raw.jt808.v1"} 1`,
`vehicle_bridge_last_ack_unix_seconds{status="ok",subject="vehicle.raw.jt808.v1"} `,
} {
if !strings.Contains(text, want) {
t.Fatalf("metrics missing %s:\n%s", want, text)
}
}
}
func TestRecordBridgeKafkaE2EDurationSkipsMissingReceiveTime(t *testing.T) {
registry := metrics.NewRegistry()
recordBridgeKafkaE2EDuration(registry, "vehicle.raw.go.jt808.v1", 0)
if text := registry.Render(); strings.Contains(text, "vehicle_bridge_kafka_e2e_duration_ms_histogram") || strings.Contains(text, "vehicle_bridge_kafka_e2e_recent") {
t.Fatalf("e2e metric should be skipped when received_at_ms is missing:\n%s", text)
}
}
func TestBridgeBatchExposesPendingAndDurationMetrics(t *testing.T) {
env := envelope.FrameEnvelope{Protocol: envelope.ProtocolGB32960, VIN: "VIN001", MessageID: "0x02"}
payload, err := json.Marshal(env)
if err != nil {
t.Fatal(err)
}
registry := metrics.NewRegistry()
writer := &recordingBridgeWriter{
onWrite: func() {
text := registry.Render()
for _, want := range []string{
`vehicle_bridge_batch_pending_messages 2`,
`vehicle_bridge_batch_pending_kafka_messages 2`,
} {
if !strings.Contains(text, want) {
t.Fatalf("pending bridge metric missing %s during write:\n%s", want, text)
}
}
},
}
err = bridgeBatch(context.Background(), registry, writer, []bridgeMessage{
{subject: "vehicle.raw.go.gb32960.v1", data: payload, ack: func() error { return nil }},
{subject: "vehicle.raw.go.gb32960.v1", data: payload, ack: func() error { return nil }},
}, map[string]string{"vehicle.raw.go.gb32960.v1": "vehicle.raw.go.gb32960.v1"})
if err != nil {
t.Fatalf("bridgeBatch() error = %v", err)
}
text := registry.Render()
for _, want := range []string{
`vehicle_bridge_batch_pending_messages 0`,
`vehicle_bridge_batch_pending_kafka_messages 0`,
`vehicle_bridge_batch_duration_ms_histogram_bucket{le="+Inf",status="ok"} 1`,
`vehicle_bridge_batch_duration_ms_histogram_count{status="ok"} 1`,
`vehicle_bridge_batch_duration_ms_histogram_sum{status="ok"}`,
} {
if !strings.Contains(text, want) {
t.Fatalf("bridge batch metric missing %s:\n%s", want, text)
}
}
}
func TestBridgeBatchPendingAggregatesConcurrentWorkers(t *testing.T) {
env := envelope.FrameEnvelope{Protocol: envelope.ProtocolGB32960, VIN: "VIN001", MessageID: "0x02"}
payload, err := json.Marshal(env)
if err != nil {
t.Fatal(err)
}
registry := metrics.NewRegistry()
firstStarted := make(chan struct{})
secondStarted := make(chan struct{})
release := make(chan struct{})
writerFor := func(started chan struct{}) *recordingBridgeWriter {
return &recordingBridgeWriter{
onWrite: func() {
close(started)
<-release
},
}
}
messages := func(count int) []bridgeMessage {
out := make([]bridgeMessage, 0, count)
for i := 0; i < count; i++ {
out = append(out, bridgeMessage{subject: "vehicle.raw.go.gb32960.v1", data: payload, ack: func() error { return nil }})
}
return out
}
route := map[string]string{"vehicle.raw.go.gb32960.v1": "vehicle.raw.go.gb32960.v1"}
var wg sync.WaitGroup
wg.Add(2)
var firstErr error
var secondErr error
go func() {
defer wg.Done()
firstErr = bridgeBatch(context.Background(), registry, writerFor(firstStarted), messages(2), route)
}()
go func() {
defer wg.Done()
secondErr = bridgeBatch(context.Background(), registry, writerFor(secondStarted), messages(3), route)
}()
<-firstStarted
<-secondStarted
text := registry.Render()
for _, want := range []string{
`vehicle_bridge_batch_pending_messages 5`,
`vehicle_bridge_batch_pending_kafka_messages 5`,
} {
if !strings.Contains(text, want) {
t.Fatalf("aggregate pending bridge metric missing %s during concurrent writes:\n%s", want, text)
}
}
close(release)
wg.Wait()
if firstErr != nil || secondErr != nil {
t.Fatalf("bridgeBatch errors = %v / %v", firstErr, secondErr)
}
text = registry.Render()
for _, want := range []string{
`vehicle_bridge_batch_pending_messages 0`,
`vehicle_bridge_batch_pending_kafka_messages 0`,
} {
if !strings.Contains(text, want) {
t.Fatalf("aggregate pending bridge metric should reset after concurrent writes, missing %s:\n%s", want, text)
}
}
}
func TestRecordNATSConsumerInfoMetrics(t *testing.T) {
registry := metrics.NewRegistry()
cfg := config{NATSStream: "VEHICLE_INGEST", NATSDurable: "vehicle-kafka-bridge"}
recordNATSConsumerInfoMetrics(registry, cfg, &nats.ConsumerInfo{
NumPending: 17,
NumAckPending: 3,
NumWaiting: 2,
})
text := registry.Render()
for _, want := range []string{
`vehicle_bridge_nats_consumer_pending{consumer="vehicle-kafka-bridge",stream="VEHICLE_INGEST"} 17`,
`vehicle_bridge_nats_consumer_ack_pending{consumer="vehicle-kafka-bridge",stream="VEHICLE_INGEST"} 3`,
`vehicle_bridge_nats_consumer_waiting{consumer="vehicle-kafka-bridge",stream="VEHICLE_INGEST"} 2`,
} {
if !strings.Contains(text, want) {
t.Fatalf("metrics missing %s:\n%s", want, text)
}
}
}
func TestIsBridgeShutdownFetchError(t *testing.T) {
if isBridgeShutdownFetchError(context.Background(), errors.New("temporary nats failure")) {
t.Fatal("temporary failure should not be treated as shutdown")
}
if !isBridgeShutdownFetchError(context.Background(), nats.ErrConnectionClosed) {
t.Fatal("connection closed should stop worker without noisy error log")
}
ctx, cancel := context.WithCancel(context.Background())
cancel()
if !isBridgeShutdownFetchError(ctx, errors.New("any fetch error after cancellation")) {
t.Fatal("cancelled context should stop worker without noisy error log")
}
}
func TestIsTransientBridgeFetchError(t *testing.T) {
for _, err := range []error{
errors.New("nats: disconnected during fetch"),
errors.New("nats: connection closed"),
errors.New("read tcp: connection reset by peer"),
errors.New("temporary network unavailable"),
errors.New("i/o timeout"),
} {
if !isTransientBridgeFetchError(err) {
t.Fatalf("isTransientBridgeFetchError(%v) = false, want true", err)
}
}
if isTransientBridgeFetchError(errors.New("permission denied")) {
t.Fatal("non-transient bridge fetch error should stay non-transient")
}
if isTransientBridgeFetchError(nil) {
t.Fatal("nil should not be transient")
}
}
func TestLoadConfigDefaultsToGoSubjectRoutes(t *testing.T) {
cfg := loadConfig()
if err := cfg.Validate(); err != nil {
t.Fatalf("default config Validate() error = %v", err)
}
want := map[string]string{
"vehicle.raw.go.gb32960.v1": "vehicle.raw.go.gb32960.v1",
"vehicle.raw.go.jt808.v1": "vehicle.raw.go.jt808.v1",
"vehicle.raw.go.yutong-mqtt.v1": "vehicle.raw.go.yutong-mqtt.v1",
"vehicle.fields.go.gb32960.v1": "vehicle.fields.go.gb32960.v1",
"vehicle.fields.go.jt808.v1": "vehicle.fields.go.jt808.v1",
"vehicle.fields.go.yutong-mqtt.v1": "vehicle.fields.go.yutong-mqtt.v1",
}
if len(cfg.Route) != len(want) {
t.Fatalf("route len = %d, want %d: %#v", len(cfg.Route), len(want), cfg.Route)
}
for subject, topic := range want {
if got := cfg.Route[subject]; got != topic {
t.Fatalf("route[%q] = %q, want %q; route=%#v", subject, got, topic, cfg.Route)
}
}
if !cfg.DeriveFieldsFromRaw {
t.Fatal("DeriveFieldsFromRaw = false, want canonical raw projection enabled by default")
}
for subject, wantTopic := range map[string]string{
topics.RawGB32960: topics.FieldsGB32960,
topics.RawJT808: topics.FieldsJT808,
topics.RawYutongMQTT: topics.FieldsYutongMQTT,
} {
if got := cfg.RawFieldRoutes[subject].Topic; got != wantTopic {
t.Fatalf("RawFieldRoutes[%q].Topic = %q, want %q", subject, got, wantTopic)
}
}
}
func TestConfigValidateRejectsRawFieldsSubjectOverlap(t *testing.T) {
cfg := config{
RawRoutes: []subjectRoute{
{Subject: "vehicle.same.jt808", Topic: "vehicle.raw.go.jt808.v1"},
},
FieldsRoutes: []subjectRoute{
{Subject: "vehicle.same.jt808", Topic: "vehicle.fields.go.jt808.v1"},
},
}
err := cfg.Validate()
if err == nil {
t.Fatal("Validate() error = nil, want subject overlap rejection")
}
if !strings.Contains(err.Error(), "nats subject") {
t.Fatalf("Validate() error = %q, want nats subject hint", err)
}
}
func TestConfigValidateRejectsKnownProtocolTopicMismatch(t *testing.T) {
cfg := config{
RawRoutes: []subjectRoute{
{Protocol: envelope.ProtocolJT808, Subject: "vehicle.raw.go.jt808.v1", Topic: "vehicle.raw.go.gb32960.v1"},
},
}
err := cfg.Validate()
if err == nil {
t.Fatal("Validate() error = nil, want known protocol topic mismatch")
}
if !strings.Contains(err.Error(), "must match protocol") {
t.Fatalf("Validate() error = %q, want protocol mismatch hint", err)
}
}
func TestConfigValidateRejectsFieldsRouteToRawKafkaTopic(t *testing.T) {
cfg := config{
RawRoutes: []subjectRoute{
{Subject: "vehicle.raw.go.jt808.v1", Topic: "vehicle.raw.go.jt808.v1"},
},
FieldsRoutes: []subjectRoute{
{Subject: "vehicle.fields.go.jt808.v1", Topic: "vehicle.raw.go.jt808.v1"},
},
}
err := cfg.Validate()
if err == nil {
t.Fatal("Validate() error = nil, want fields topic family rejection")
}
if !strings.Contains(err.Error(), "fields kafka topic") {
t.Fatalf("Validate() error = %q, want fields kafka topic hint", err)
}
}
func TestLoadConfigDefaultsStreamMaxBytes(t *testing.T) {
cfg := loadConfig()
if got, want := cfg.StreamMaxBytes, int64(20*1024*1024*1024); got != want {
t.Fatalf("StreamMaxBytes = %d, want %d", got, want)
}
if got, want := cfg.StreamEnsureWait, 60*time.Second; got != want {
t.Fatalf("StreamEnsureWait = %v, want %v", got, want)
}
if got, want := cfg.KafkaBatchTimeout, 20*time.Millisecond; got != want {
t.Fatalf("KafkaBatchTimeout = %v, want %v", got, want)
}
if got, want := cfg.KafkaWriteConcurrency, 6; got != want {
t.Fatalf("KafkaWriteConcurrency = %d, want %d", got, want)
}
if got, want := cfg.FetchWait, 20*time.Millisecond; got != want {
t.Fatalf("FetchWait = %v, want %v", got, want)
}
if got, want := cfg.Workers, 4; got != want {
t.Fatalf("Workers = %d, want %d", got, want)
}
}
func TestLoadConfigReadsStreamMaxBytesOverride(t *testing.T) {
t.Setenv("NATS_STREAM_MAX_BYTES", "1073741824")
t.Setenv("NATS_STREAM_ENSURE_TIMEOUT_SECONDS", "90")
t.Setenv("BRIDGE_KAFKA_BATCH_TIMEOUT_MS", "35")
t.Setenv("BRIDGE_KAFKA_WRITE_CONCURRENCY", "4")
t.Setenv("BRIDGE_FETCH_WAIT_MS", "45")
t.Setenv("BRIDGE_WORKERS", "6")
cfg := loadConfig()
if got, want := cfg.StreamMaxBytes, int64(1073741824); got != want {
t.Fatalf("StreamMaxBytes = %d, want %d", got, want)
}
if got, want := cfg.StreamEnsureWait, 90*time.Second; got != want {
t.Fatalf("StreamEnsureWait = %v, want %v", got, want)
}
if got, want := cfg.KafkaBatchTimeout, 35*time.Millisecond; got != want {
t.Fatalf("KafkaBatchTimeout = %v, want %v", got, want)
}
if got, want := cfg.KafkaWriteConcurrency, 4; got != want {
t.Fatalf("KafkaWriteConcurrency = %d, want %d", got, want)
}
if got, want := cfg.FetchWait, 45*time.Millisecond; got != want {
t.Fatalf("FetchWait = %v, want %v", got, want)
}
if got, want := cfg.Workers, 6; got != want {
t.Fatalf("Workers = %d, want %d", got, want)
}
}
func TestRecordBridgeConfigMetrics(t *testing.T) {
registry := metrics.NewRegistry()
recordBridgeConfigMetrics(registry, config{
KafkaBatchTimeout: 35 * time.Millisecond,
KafkaWriteConcurrency: 4,
BatchSize: 600,
FetchWait: 20 * time.Millisecond,
Workers: 6,
DeriveFieldsFromRaw: true,
})
text := registry.Render()
for _, want := range []string{
`vehicle_bridge_config{setting="batch_size"} 600`,
`vehicle_bridge_config{setting="fetch_wait_ms"} 20`,
`vehicle_bridge_config{setting="kafka_batch_timeout_ms"} 35`,
`vehicle_bridge_config{setting="kafka_write_concurrency"} 4`,
`vehicle_bridge_config{setting="workers"} 6`,
`vehicle_bridge_config{setting="derive_fields_from_raw_enabled"} 1`,
} {
if !strings.Contains(text, want) {
t.Fatalf("bridge config metric missing %s:\n%s", want, text)
}
}
}
func TestNewKafkaWriterUsesConfiguredBatchTimeout(t *testing.T) {
writer := newKafkaWriter(config{
KafkaBrokers: []string{"127.0.0.1:9092"},
KafkaBatchTimeout: 17 * time.Millisecond,
})
defer writer.Close()
if got, want := writer.BatchTimeout, 17*time.Millisecond; got != want {
t.Fatalf("BatchTimeout = %v, want %v", got, want)
}
}
func TestLoadConfigIncludesUnifiedOnlyWhenExplicitlyConfigured(t *testing.T) {
t.Setenv("NATS_SUBJECT_UNIFIED", "vehicle.event.go.unified.v1")
t.Setenv("KAFKA_TOPIC_UNIFIED", "vehicle.event.go.unified.v1")
cfg := loadConfig()
if got := cfg.Route["vehicle.event.go.unified.v1"]; got != "vehicle.event.go.unified.v1" {
t.Fatalf("unified route = %q, route=%#v", got, cfg.Route)
}
}
type recordingBridgeWriter struct {
mu sync.Mutex
messages []kafka.Message
calls [][]kafka.Message
err error
topicErr map[string]error
onWrite func()
}
func (w *recordingBridgeWriter) WriteMessages(_ context.Context, messages ...kafka.Message) error {
if w.onWrite != nil {
w.onWrite()
}
w.mu.Lock()
defer w.mu.Unlock()
call := append([]kafka.Message(nil), messages...)
w.calls = append(w.calls, call)
if len(messages) > 0 && w.topicErr != nil {
if err := w.topicErr[messages[0].Topic]; err != nil {
return err
}
}
if w.err != nil {
return w.err
}
w.messages = append(w.messages, messages...)
return nil
}
type concurrentBridgeWriter struct {
started chan string
release chan struct{}
active atomic.Int32
max atomic.Int32
}
func (w *concurrentBridgeWriter) WriteMessages(_ context.Context, messages ...kafka.Message) error {
active := w.active.Add(1)
defer w.active.Add(-1)
for {
current := w.max.Load()
if active <= current || w.max.CompareAndSwap(current, active) {
break
}
}
w.started <- messages[0].Topic
<-w.release
return nil
}
func TestBridgeWritesIndependentKafkaTopicsConcurrently(t *testing.T) {
writer := &concurrentBridgeWriter{
started: make(chan string, 2),
release: make(chan struct{}),
}
messages := []bridgeMessage{
{subject: topics.RawJT808, data: []byte(`{"protocol":"JT808"}`), ack: func() error { return nil }},
{subject: topics.RawGB32960, data: []byte(`{"protocol":"GB32960"}`), ack: func() error { return nil }},
}
route := map[string]string{
topics.RawJT808: topics.RawJT808,
topics.RawGB32960: topics.RawGB32960,
}
done := make(chan error, 1)
go func() {
done <- bridgeBatchWithProjectionConcurrency(context.Background(), nil, writer, messages, route, nil, false, 2)
}()
for i := 0; i < 2; i++ {
select {
case <-writer.started:
case <-time.After(time.Second):
t.Fatal("independent Kafka topic writes did not start concurrently")
}
}
if got := writer.max.Load(); got != 2 {
t.Fatalf("max concurrent Kafka writes = %d, want 2", got)
}
close(writer.release)
if err := <-done; err != nil {
t.Fatalf("bridgeBatchWithProjectionConcurrency() error = %v", err)
}
}

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -4,9 +4,15 @@ import (
"context"
"database/sql"
"encoding/json"
"errors"
"fmt"
"os"
"os/signal"
"path/filepath"
"sort"
"strconv"
"strings"
"sync"
"syscall"
"time"
@@ -14,8 +20,12 @@ import (
"github.com/segmentio/kafka-go"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/eventbus"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/health"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/metrics"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/observability"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/topics"
)
func main() {
@@ -24,6 +34,10 @@ func main() {
defer stop()
cfg := loadConfig()
if err := cfg.Validate(); err != nil {
logger.Error("invalid stat writer config", "error", err)
os.Exit(1)
}
db, err := sql.Open("mysql", cfg.MySQLDSN)
if err != nil {
logger.Error("mysql open failed", "error", err)
@@ -34,15 +48,68 @@ func main() {
logger.Error("mysql ping failed", "error", err)
os.Exit(1)
}
registry := metrics.NewRegistry()
metrics.RegisterKafkaConsumerInfo(registry, "vehicle-stat-writer", cfg.KafkaGroup, cfg.KafkaTopics)
registry.SetGauge("vehicle_stat_project_interval_seconds", nil, cfg.ProjectInterval.Seconds())
registry.SetGauge("vehicle_stat_source_touch_interval_seconds", nil, cfg.SourceTouchInterval.Seconds())
registry.SetGauge("vehicle_stat_cache_retention_seconds", nil, cfg.CacheRetention.Seconds())
registry.SetGauge("vehicle_stat_cache_cleanup_interval_seconds", nil, cfg.CacheCleanupInterval.Seconds())
registry.SetGauge("vehicle_stat_baseline_miss_ttl_seconds", nil, cfg.BaselineMissTTL.Seconds())
registry.SetGauge("vehicle_stat_baseline_hit_ttl_seconds", nil, cfg.BaselineHitTTL.Seconds())
registry.SetGauge("vehicle_stat_config", metrics.Labels{"setting": "workers"}, float64(cfg.Workers))
health.Start(ctx, logger, health.NewServer(env("HEALTH_ADDR", ""), "vehicle-stat-writer", []health.Check{
{Name: "mysql", Check: db.PingContext},
}, registry))
writer := stats.NewWriter(db, cfg.Location)
writer.SetProjectionInterval(cfg.ProjectInterval)
writer.SetSourceTouchInterval(cfg.SourceTouchInterval)
writer.SetCacheRetention(cfg.CacheRetention)
writer.SetCacheCleanupInterval(cfg.CacheCleanupInterval)
writer.SetBaselineMissTTL(cfg.BaselineMissTTL)
writer.SetBaselineHitTTL(cfg.BaselineHitTTL)
writer.SetMaxCacheEntries(cfg.CacheMaxEntries)
if cfg.EnsureSchema {
if err := writer.EnsureSchema(ctx); err != nil {
logger.Error("mysql schema bootstrap failed", "error", err)
os.Exit(1)
}
}
if cfg.NormalizePlatformSourcesOnStart {
statDate := time.Now().In(cfg.Location).Format("2006-01-02")
normalizeCtx, cancel := context.WithTimeout(ctx, cfg.NormalizePlatformSourcesTimeout)
normalized, err := stats.NormalizePlatformSourceMileageForDate(normalizeCtx, db, statDate, envelope.ProtocolJT808)
cancel()
if err != nil {
logger.Warn("platform source startup normalization failed", "stat_date", statDate, "protocol", envelope.ProtocolJT808, "normalized", normalized, "error", err)
} else if normalized > 0 {
logger.Info("platform source startup normalization finished", "stat_date", statDate, "protocol", envelope.ProtocolJT808, "normalized", normalized)
}
}
var appender statAppender = retryStatAppender{
delegate: writer,
attempts: cfg.RetryAttempts,
delay: cfg.RetryDelay,
registry: registry,
}
quarantiner := fileStatMessageQuarantiner{dir: cfg.QuarantineDir}
logger.Info("stat writer started", "group", cfg.KafkaGroup, "topics", strings.Join(cfg.KafkaTopics, ","), "workers", cfg.Workers, "project_interval_seconds", cfg.ProjectInterval.Seconds(), "source_touch_interval_seconds", cfg.SourceTouchInterval.Seconds(), "cache_retention_seconds", cfg.CacheRetention.Seconds(), "cache_cleanup_interval_seconds", cfg.CacheCleanupInterval.Seconds(), "baseline_miss_ttl_seconds", cfg.BaselineMissTTL.Seconds(), "baseline_hit_ttl_seconds", cfg.BaselineHitTTL.Seconds(), "cache_max_entries", cfg.CacheMaxEntries, "batch_size", cfg.BatchSize, "batch_wait_ms", cfg.BatchWait, "retry_attempts", cfg.RetryAttempts, "retry_delay_ms", cfg.RetryDelay.Milliseconds(), "quarantine_dir", cfg.QuarantineDir)
var workers sync.WaitGroup
for workerID := 1; workerID <= cfg.Workers; workerID++ {
workers.Add(1)
go func(id int) {
defer workers.Done()
runStatConsumer(ctx, logger, registry, appender, quarantiner, cfg, id)
}(workerID)
}
workers.Wait()
}
func runStatConsumer(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender statAppender, quarantiner statMessageQuarantiner, cfg config, workerID int) {
reader := kafka.NewReader(kafka.ReaderConfig{
Brokers: cfg.KafkaBrokers,
GroupID: cfg.KafkaGroup,
@@ -52,7 +119,10 @@ func main() {
})
defer reader.Close()
logger.Info("stat writer started", "group", cfg.KafkaGroup, "topics", strings.Join(cfg.KafkaTopics, ","))
workerLabels := metrics.Labels{"worker": strconv.Itoa(workerID)}
registry.SetGauge("vehicle_stat_worker_active", workerLabels, 1)
defer registry.SetGauge("vehicle_stat_worker_active", workerLabels, 0)
for {
message, err := reader.FetchMessage(ctx)
if err != nil {
@@ -62,29 +132,738 @@ func main() {
logger.Error("kafka fetch failed", "error", err)
continue
}
batch := collectStatBatch(ctx, reader, message, cfg.BatchSize, time.Duration(cfg.BatchWait)*time.Millisecond)
processStatBatchReliablyForWorker(ctx, logger, registry, appender, reader, quarantiner, batch, cfg.RetryDelay, workerLabels)
}
}
const kafkaMessageOperationTimeout = 30 * time.Second
type statAppender interface {
Append(context.Context, envelope.FrameEnvelope) error
}
type statResultAppender interface {
AppendWithResult(context.Context, envelope.FrameEnvelope) (stats.AppendResult, error)
}
type statCacheReporter interface {
CacheStats() stats.CacheStats
}
type kafkaMessageCommitter interface {
CommitMessages(context.Context, ...kafka.Message) error
}
type kafkaMessageFetcher interface {
FetchMessage(context.Context) (kafka.Message, error)
}
type statMessageQuarantiner interface {
Quarantine(context.Context, kafka.Message, error) error
}
type statBatchItem struct {
message kafka.Message
processed bool
}
type statBatchOutcome struct {
commitMessages []kafka.Message
retryMessages []kafka.Message
failedMessage *kafka.Message
writeErr error
}
var statWriteDurationBucketsMS = []float64{1, 5, 10, 25, 50, 100, 250, 500, 1000, 5000}
var statWriteE2EDurationBucketsMS = []float64{10, 25, 50, 100, 250, 500, 1000, 2500, 5000, 10000}
var statWriteE2ERecent = metrics.NewRecentLatencyByKey(512)
func collectStatBatch(ctx context.Context, fetcher kafkaMessageFetcher, first kafka.Message, maxSize int, maxWait time.Duration) []kafka.Message {
if maxSize <= 1 {
return []kafka.Message{first}
}
if maxWait <= 0 {
maxWait = 100 * time.Millisecond
}
batch := []kafka.Message{first}
deadline := time.Now().Add(maxWait)
for len(batch) < maxSize {
remaining := time.Until(deadline)
if remaining <= 0 {
break
}
fetchCtx, cancel := context.WithTimeout(ctx, remaining)
message, err := fetcher.FetchMessage(fetchCtx)
cancel()
if err != nil {
break
}
batch = append(batch, message)
}
return batch
}
func processStatMessage(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender statAppender, committer kafkaMessageCommitter, message kafka.Message) {
processStatBatch(ctx, logger, registry, appender, committer, []kafka.Message{message})
}
func processStatBatch(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender statAppender, committer kafkaMessageCommitter, messages []kafka.Message) statBatchOutcome {
return processStatBatchForWorker(ctx, logger, registry, appender, committer, messages, nil)
}
func processStatBatchForWorker(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender statAppender, committer kafkaMessageCommitter, messages []kafka.Message, workerLabels metrics.Labels) statBatchOutcome {
if len(messages) == 0 {
return statBatchOutcome{}
}
messageCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), kafkaMessageOperationTimeout)
defer cancel()
setStatBatchPendingForWorker(registry, workerLabels, len(messages))
defer setStatBatchPendingForWorker(registry, workerLabels, 0)
defer recordStatCacheMetrics(registry, appender)
items := make([]statBatchItem, len(messages))
for index, message := range messages {
items[index].message = message
}
for itemIndex, message := range messages {
addStatMetric(registry, "vehicle_stat_kafka_messages_total", message, "received")
addStatLagMetric(registry, message)
var env envelope.FrameEnvelope
if err := json.Unmarshal(message.Value, &env); err != nil {
addStatMetric(registry, "vehicle_stat_kafka_messages_total", message, "invalid_json")
logger.Warn("skip invalid envelope json", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "error", err)
_ = reader.CommitMessages(ctx, message)
items[itemIndex].processed = true
continue
}
if err := writer.Append(ctx, env); err != nil {
if status, err := validateStatFieldsEnvelope(message.Topic, env); err != nil {
addStatMetric(registry, "vehicle_stat_kafka_messages_total", message, status)
logger.Warn("skip mismatched stat fields envelope", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "protocol", env.Protocol, "event_id", env.StableEventID(), "event_kind", env.EventKind, "error", err)
items[itemIndex].processed = true
continue
}
started := time.Now()
result, err := appendStatEnvelope(messageCtx, appender, env)
recordStatWriteDuration(registry, message, statusFromError(err), time.Since(started))
recordStatSampleMetrics(registry, message, env.Protocol, result)
recordStatSourceMetrics(registry, message, env.Protocol, result)
recordStatProjectionMetrics(registry, message, env.Protocol, result)
if err != nil {
addStatMetric(registry, "vehicle_stat_writes_total", message, "error")
logger.Error("mysql append failed", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "event_id", env.StableEventID(), "error", err)
continue
committed, commitErr := commitStatProcessedPrefixAfterFailure(messageCtx, logger, registry, committer, items)
failedMessage := message
outcome := statBatchOutcome{
retryMessages: eventbus.MessagesAfterCommittedPrefixes(messages, committed),
failedMessage: &failedMessage,
writeErr: err,
}
if commitErr != nil {
outcome.commitMessages = committed
}
return outcome
}
if err := reader.CommitMessages(ctx, message); err != nil {
logger.Error("kafka commit failed", "topic", message.Topic, "partition", message.Partition, "offset", message.Offset, "error", err)
addStatMetric(registry, "vehicle_stat_writes_total", message, "ok")
recordStatWriteE2EDuration(registry, message, env)
items[itemIndex].processed = true
}
if err := committer.CommitMessages(messageCtx, messages...); err != nil {
for _, message := range messages {
addStatMetric(registry, "vehicle_stat_kafka_commits_total", message, "error")
}
first := messages[0]
logger.Error("kafka commit failed", "topic", first.Topic, "partition", first.Partition, "offset", first.Offset, "messages", len(messages), "error", err)
return statBatchOutcome{commitMessages: messages}
}
for _, message := range messages {
addStatMetric(registry, "vehicle_stat_kafka_commits_total", message, "ok")
}
return statBatchOutcome{}
}
func processStatBatchReliably(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender statAppender, committer kafkaMessageCommitter, messages []kafka.Message, retryDelay time.Duration) {
processStatBatchReliablyForWorker(ctx, logger, registry, appender, committer, nil, messages, retryDelay, nil)
}
func processStatBatchReliablyForWorker(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, appender statAppender, committer kafkaMessageCommitter, quarantiner statMessageQuarantiner, messages []kafka.Message, retryDelay time.Duration, workerLabels metrics.Labels) {
defer registry.SetGauge("vehicle_stat_retry_pending_messages", workerLabels, 0)
pending := messages
for len(pending) > 0 {
outcome := processStatBatchForWorker(ctx, logger, registry, appender, committer, pending, workerLabels)
if len(outcome.commitMessages) > 0 {
registry.IncCounter("vehicle_stat_batch_retries_total", metrics.Labels{"reason": "commit_error"})
if !retryStatCommit(ctx, logger, registry, committer, outcome.commitMessages, retryDelay) {
return
}
}
pending = outcome.retryMessages
if outcome.writeErr != nil && outcome.failedMessage != nil && isQuarantinableStatError(outcome.writeErr) && quarantiner != nil {
failed := *outcome.failedMessage
if err := quarantiner.Quarantine(ctx, failed, outcome.writeErr); err != nil {
logger.Error("stat message quarantine failed", "topic", failed.Topic, "partition", failed.Partition, "offset", failed.Offset, "error", err)
registry.IncCounter("vehicle_stat_quarantine_total", metrics.Labels{"status": "error", "topic": failed.Topic})
} else {
registry.IncCounter("vehicle_stat_quarantine_total", metrics.Labels{"status": "ok", "topic": failed.Topic})
logger.Warn("quarantined permanent stat write failure", "topic", failed.Topic, "partition", failed.Partition, "offset", failed.Offset, "error", outcome.writeErr)
if retryStatCommit(ctx, logger, registry, committer, []kafka.Message{failed}, retryDelay) {
pending = messagesExceptStatMessage(pending, failed)
}
}
}
registry.SetGauge("vehicle_stat_retry_pending_messages", workerLabels, float64(len(pending)))
if len(pending) == 0 {
return
}
registry.IncCounter("vehicle_stat_batch_retries_total", metrics.Labels{"reason": "write_error"})
if !waitForStatRetry(ctx, retryDelay) {
return
}
}
}
type quarantinedStatMessage struct {
QuarantinedAt string `json:"quarantined_at"`
Topic string `json:"topic"`
Partition int `json:"partition"`
Offset int64 `json:"offset"`
Key []byte `json:"key,omitempty"`
Value []byte `json:"value"`
Error string `json:"error"`
}
type fileStatMessageQuarantiner struct {
dir string
}
func (q fileStatMessageQuarantiner) Quarantine(ctx context.Context, message kafka.Message, cause error) error {
if err := ctx.Err(); err != nil {
return err
}
dir := strings.TrimSpace(q.dir)
if dir == "" {
return errors.New("stat quarantine directory is empty")
}
if err := os.MkdirAll(dir, 0o750); err != nil {
return fmt.Errorf("create quarantine directory: %w", err)
}
record := quarantinedStatMessage{
QuarantinedAt: time.Now().UTC().Format(time.RFC3339Nano),
Topic: message.Topic,
Partition: message.Partition,
Offset: message.Offset,
Key: message.Key,
Value: message.Value,
Error: cause.Error(),
}
payload, err := json.Marshal(record)
if err != nil {
return fmt.Errorf("marshal quarantine record: %w", err)
}
name := fmt.Sprintf("%s-%d-%d.json", sanitizeStatQuarantineName(message.Topic), message.Partition, message.Offset)
temporary, err := os.CreateTemp(dir, "."+name+"-*")
if err != nil {
return fmt.Errorf("create quarantine record: %w", err)
}
temporaryName := temporary.Name()
removeTemporary := true
defer func() {
_ = temporary.Close()
if removeTemporary {
_ = os.Remove(temporaryName)
}
}()
if err := temporary.Chmod(0o600); err != nil {
return fmt.Errorf("secure quarantine record: %w", err)
}
if _, err := temporary.Write(payload); err != nil {
return fmt.Errorf("write quarantine record: %w", err)
}
if err := temporary.Sync(); err != nil {
return fmt.Errorf("sync quarantine record: %w", err)
}
if err := temporary.Close(); err != nil {
return fmt.Errorf("close quarantine record: %w", err)
}
if err := os.Rename(temporaryName, filepath.Join(dir, name)); err != nil {
return fmt.Errorf("publish quarantine record: %w", err)
}
removeTemporary = false
return nil
}
func sanitizeStatQuarantineName(value string) string {
value = strings.TrimSpace(value)
if value == "" {
return "unknown-topic"
}
var out strings.Builder
for _, r := range value {
if r >= 'a' && r <= 'z' || r >= 'A' && r <= 'Z' || r >= '0' && r <= '9' || r == '-' || r == '_' || r == '.' {
out.WriteRune(r)
} else {
out.WriteByte('_')
}
}
return out.String()
}
func messagesExceptStatMessage(messages []kafka.Message, excluded kafka.Message) []kafka.Message {
out := make([]kafka.Message, 0, len(messages))
removed := false
for _, message := range messages {
if !removed && message.Topic == excluded.Topic && message.Partition == excluded.Partition && message.Offset == excluded.Offset {
removed = true
continue
}
out = append(out, message)
}
return out
}
func retryStatCommit(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, committer kafkaMessageCommitter, messages []kafka.Message, retryDelay time.Duration) bool {
for len(messages) > 0 {
operationCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), kafkaMessageOperationTimeout)
err := committer.CommitMessages(operationCtx, messages...)
cancel()
if err == nil {
for _, message := range messages {
addStatMetric(registry, "vehicle_stat_kafka_commits_total", message, "ok")
}
return true
}
for _, message := range messages {
addStatMetric(registry, "vehicle_stat_kafka_commits_total", message, "error")
}
first := messages[0]
logger.Error("kafka commit retry failed", "topic", first.Topic, "partition", first.Partition, "offset", first.Offset, "messages", len(messages), "error", err)
registry.IncCounter("vehicle_stat_batch_retries_total", metrics.Labels{"reason": "commit_error"})
if !waitForStatRetry(ctx, retryDelay) {
return false
}
}
return true
}
func waitForStatRetry(ctx context.Context, retryDelay time.Duration) bool {
if retryDelay <= 0 {
retryDelay = 100 * time.Millisecond
}
timer := time.NewTimer(retryDelay)
defer timer.Stop()
select {
case <-ctx.Done():
return false
case <-timer.C:
return true
}
}
func validateStatFieldsEnvelope(topic string, env envelope.FrameEnvelope) (string, error) {
return topics.ValidateFieldsEnvelope(topic, env)
}
func commitStatProcessedPrefixAfterFailure(ctx context.Context, logger interface {
Error(string, ...any)
Warn(string, ...any)
}, registry *metrics.Registry, committer kafkaMessageCommitter, items []statBatchItem) ([]kafka.Message, error) {
committable := processedPrefixMessagesByPartition(items)
if len(committable) == 0 {
return nil, nil
}
if err := committer.CommitMessages(ctx, committable...); err != nil {
for _, message := range committable {
addStatMetric(registry, "vehicle_stat_kafka_commits_total", message, "error")
}
first := committable[0]
logger.Error("kafka processed-prefix commit failed after mysql append failure", "topic", first.Topic, "partition", first.Partition, "offset", first.Offset, "messages", len(committable), "error", err)
return committable, err
}
for _, message := range committable {
addStatMetric(registry, "vehicle_stat_kafka_commits_total", message, "ok")
}
return committable, nil
}
func processedPrefixMessagesByPartition(items []statBatchItem) []kafka.Message {
type partitionKey struct {
topic string
partition int
}
groups := map[partitionKey][]statBatchItem{}
for _, item := range items {
key := partitionKey{topic: item.message.Topic, partition: item.message.Partition}
groups[key] = append(groups[key], item)
}
var keys []partitionKey
for key := range groups {
keys = append(keys, key)
}
sort.Slice(keys, func(i, j int) bool {
if keys[i].topic != keys[j].topic {
return keys[i].topic < keys[j].topic
}
return keys[i].partition < keys[j].partition
})
var out []kafka.Message
for _, key := range keys {
group := groups[key]
sort.Slice(group, func(i, j int) bool {
return group[i].message.Offset < group[j].message.Offset
})
var previousOffset int64
for index, item := range group {
if index > 0 && item.message.Offset != previousOffset+1 {
break
}
if !item.processed {
break
}
out = append(out, item.message)
previousOffset = item.message.Offset
}
}
return out
}
func appendStatEnvelope(ctx context.Context, appender statAppender, env envelope.FrameEnvelope) (stats.AppendResult, error) {
if resultAppender, ok := appender.(statResultAppender); ok {
return resultAppender.AppendWithResult(ctx, env)
}
return stats.AppendResult{}, appender.Append(ctx, env)
}
type retryStatAppender struct {
delegate statAppender
attempts int
delay time.Duration
registry *metrics.Registry
}
func (a retryStatAppender) Append(ctx context.Context, env envelope.FrameEnvelope) error {
_, err := a.AppendWithResult(ctx, env)
return err
}
func (a retryStatAppender) AppendWithResult(ctx context.Context, env envelope.FrameEnvelope) (stats.AppendResult, error) {
if a.delegate == nil {
return stats.AppendResult{}, nil
}
attempts := a.attempts
if attempts <= 0 {
attempts = 1
}
var result stats.AppendResult
var err error
for attempt := 1; attempt <= attempts; attempt++ {
result, err = appendStatEnvelope(ctx, a.delegate, env)
if err == nil || (!isTransientMySQLStatError(err) && !isQuarantinableStatError(err)) {
return result, err
}
if attempt == attempts {
a.recordRetry("single", "exhausted")
return result, err
}
a.recordRetry("single", "retry")
if a.delay <= 0 {
continue
}
timer := time.NewTimer(a.delay)
select {
case <-ctx.Done():
timer.Stop()
return result, ctx.Err()
case <-timer.C:
}
}
return result, err
}
func (a retryStatAppender) recordRetry(operation string, status string) {
if a.registry == nil {
return
}
labels := metrics.Labels{"operation": operation, "status": status}
a.registry.IncCounter("vehicle_stat_write_retries_total", labels)
metrics.RecordLastActivity(a.registry, "vehicle_stat_last_write_retry_unix_seconds", labels)
}
func (a retryStatAppender) CacheStats() stats.CacheStats {
if reporter, ok := a.delegate.(statCacheReporter); ok {
return reporter.CacheStats()
}
return stats.CacheStats{}
}
func isTransientMySQLStatError(err error) bool {
if err == nil || errors.Is(err, context.Canceled) || errors.Is(err, context.DeadlineExceeded) {
return false
}
text := strings.ToLower(strings.TrimSpace(err.Error()))
return strings.Contains(text, "deadlock") ||
strings.Contains(text, "error 1213") ||
strings.Contains(text, "40001") ||
strings.Contains(text, "lock wait timeout") ||
strings.Contains(text, "error 1205") ||
strings.Contains(text, "timeout") ||
strings.Contains(text, "temporary") ||
strings.Contains(text, "temporarily") ||
strings.Contains(text, "connection refused") ||
strings.Contains(text, "connection reset") ||
strings.Contains(text, "connection closed") ||
strings.Contains(text, "broken pipe") ||
strings.Contains(text, "bad connection") ||
strings.Contains(text, "invalid connection") ||
strings.Contains(text, "i/o timeout") ||
text == "eof" ||
strings.Contains(text, "unexpected eof") ||
strings.Contains(text, "server is down") ||
strings.Contains(text, "network is unreachable") ||
strings.Contains(text, "no route to host")
}
func isQuarantinableStatError(err error) bool {
if err == nil || errors.Is(err, context.Canceled) || errors.Is(err, context.DeadlineExceeded) {
return false
}
text := strings.ToLower(strings.TrimSpace(err.Error()))
// Only deterministic row-data failures may advance past a Kafka message
// after the configured finite retries. Schema, permission and unknown
// application errors remain failure-closed so a systemic outage cannot
// silently quarantine an entire stream.
return strings.Contains(text, "error 1048") ||
strings.Contains(text, "cannot be null") ||
strings.Contains(text, "error 1264") ||
strings.Contains(text, "out of range value") ||
strings.Contains(text, "error 1265") ||
strings.Contains(text, "data truncated") ||
strings.Contains(text, "error 1292") ||
strings.Contains(text, "incorrect datetime value") ||
strings.Contains(text, "error 1366") ||
strings.Contains(text, "incorrect decimal value") ||
strings.Contains(text, "incorrect integer value") ||
strings.Contains(text, "error 1406") ||
strings.Contains(text, "data too long for column") ||
strings.Contains(text, "error 3819") ||
strings.Contains(text, "check constraint")
}
func addStatMetric(registry *metrics.Registry, name string, message kafka.Message, status string) {
if registry == nil {
return
}
labels := metrics.Labels{"topic": message.Topic, "status": status}
registry.IncCounter(name, labels)
switch name {
case "vehicle_stat_kafka_messages_total":
metrics.RecordLastActivity(registry, "vehicle_stat_last_message_unix_seconds", labels)
case "vehicle_stat_writes_total":
metrics.RecordLastActivity(registry, "vehicle_stat_last_write_unix_seconds", labels)
case "vehicle_stat_kafka_commits_total":
metrics.RecordLastActivity(registry, "vehicle_stat_last_commit_unix_seconds", labels)
}
}
func recordStatSampleMetrics(registry *metrics.Registry, message kafka.Message, protocol envelope.Protocol, result stats.AppendResult) {
addStatSampleMetric(registry, message, protocol, "found", result.SamplesFound)
addStatSampleMetric(registry, message, protocol, "written", result.SamplesWritten)
addStatSampleMetric(registry, message, protocol, "skipped_missing_fields", result.SamplesSkippedMissingFields)
addStatSampleMetric(registry, message, protocol, "skipped_missing_vin", result.SamplesSkippedMissingVIN)
addStatSampleMetric(registry, message, protocol, "skipped_missing_mileage", result.SamplesSkippedMissingMileage)
addStatSampleMetric(registry, message, protocol, "skipped_non_mileage_frame", result.SamplesSkippedNonMileageFrame)
addStatSampleMetric(registry, message, protocol, "skipped_non_positive_mileage", result.SamplesSkippedNonPositiveMileage)
addStatSampleMetric(registry, message, protocol, "skipped_missing_time", result.SamplesSkippedMissingTime)
addStatSampleMetric(registry, message, protocol, "skipped_same_mileage", result.SamplesSkippedSameMileage)
addStatSampleMetric(registry, message, protocol, "skipped_missing_source", result.SamplesSkippedMissingSource)
addStatSampleMetric(registry, message, protocol, "event_time_future_adjusted", result.SamplesAdjustedFutureEventTime)
}
func addStatSampleMetric(registry *metrics.Registry, message kafka.Message, protocol envelope.Protocol, status string, count int) {
if registry == nil || count == 0 {
return
}
registry.AddCounter("vehicle_stat_samples_total", metrics.Labels{
"topic": message.Topic,
"protocol": statProtocolLabel(protocol),
"status": status,
}, float64(count))
}
func recordStatSourceMetrics(registry *metrics.Registry, message kafka.Message, protocol envelope.Protocol, result stats.AppendResult) {
addStatSourceMetric(registry, message, protocol, "attempted", result.SourceTouchesAttempted)
addStatSourceMetric(registry, message, protocol, "written", result.SourceTouchesWritten)
addStatSourceMetric(registry, message, protocol, "skipped_throttled", result.SourceTouchesSkippedThrottled)
addStatSourceMetric(registry, message, protocol, "skipped_missing_endpoint", result.SourceTouchesSkippedMissing)
addStatSourceMetric(registry, message, protocol, "skipped_unmanaged", result.SourceTouchesSkippedUnmanaged)
}
func addStatSourceMetric(registry *metrics.Registry, message kafka.Message, protocol envelope.Protocol, status string, count int) {
if registry == nil || count == 0 {
return
}
registry.AddCounter("vehicle_stat_sources_total", metrics.Labels{
"topic": message.Topic,
"protocol": statProtocolLabel(protocol),
"status": status,
}, float64(count))
}
func recordStatProjectionMetrics(registry *metrics.Registry, message kafka.Message, protocol envelope.Protocol, result stats.AppendResult) {
addStatProjectionMetric(registry, message, protocol, "attempted", result.ProjectionsAttempted)
addStatProjectionMetric(registry, message, protocol, "written", result.ProjectionsWritten)
addStatProjectionMetric(registry, message, protocol, "skipped_throttled", result.ProjectionsSkippedThrottled)
}
func addStatProjectionMetric(registry *metrics.Registry, message kafka.Message, protocol envelope.Protocol, status string, count int) {
if registry == nil || count == 0 {
return
}
registry.AddCounter("vehicle_stat_projections_total", metrics.Labels{
"topic": message.Topic,
"protocol": statProtocolLabel(protocol),
"status": status,
}, float64(count))
}
func statProtocolLabel(protocol envelope.Protocol) string {
protocolLabel := strings.TrimSpace(string(protocol))
if protocolLabel == "" {
return "UNKNOWN"
}
return protocolLabel
}
func recordStatCacheMetrics(registry *metrics.Registry, appender statAppender) {
if registry == nil {
return
}
reporter, ok := appender.(statCacheReporter)
if !ok {
return
}
stats := reporter.CacheStats()
setStatCacheGauge(registry, "last_total_mileage", stats.LastTotalMileageEntries, stats.MaxEntries, stats.LastCleanupTotalMileage, stats.TotalMileageEvictions)
setStatCacheGauge(registry, "source_seen", stats.LastSourceSeenEntries, stats.MaxEntries, stats.LastCleanupSourceSeen, stats.SourceSeenEvictions)
setStatCacheGauge(registry, "projection", stats.LastProjectionEntries, stats.MaxEntries, stats.LastCleanupProjection, stats.ProjectionEvictions)
setStatCacheGauge(registry, "baseline", stats.BaselineEntries, stats.MaxEntries, stats.LastCleanupBaseline, stats.BaselineEvictions)
if !stats.LastCleanupAt.IsZero() {
registry.SetGauge("vehicle_stat_cache_last_cleanup_unix_seconds", nil, float64(stats.LastCleanupAt.Unix()))
}
}
func setStatCacheGauge(registry *metrics.Registry, cache string, entries int, maxEntries int, lastCleanupDeleted int, evictions int) {
labels := metrics.Labels{"cache": cache}
registry.SetGauge("vehicle_stat_cache_entries", labels, float64(entries))
registry.SetGauge("vehicle_stat_cache_max_entries", labels, float64(maxEntries))
registry.SetGauge("vehicle_stat_cache_last_cleanup_deleted", labels, float64(lastCleanupDeleted))
registry.SetGauge("vehicle_stat_cache_evictions_total", labels, float64(evictions))
}
func addStatLagMetric(registry *metrics.Registry, message kafka.Message) {
if registry == nil {
return
}
registry.SetKafkaLag("vehicle_stat_kafka_lag", message.Topic, message.Partition, message.Offset, message.HighWaterMark)
}
func recordStatWriteDuration(registry *metrics.Registry, message kafka.Message, status string, elapsed time.Duration) {
if registry == nil {
return
}
registry.ObserveHistogram("vehicle_stat_write_duration_ms_histogram", metrics.Labels{
"topic": message.Topic,
"status": status,
}, statWriteDurationBucketsMS, float64(elapsed.Milliseconds()))
}
func recordStatWriteE2EDuration(registry *metrics.Registry, message kafka.Message, env envelope.FrameEnvelope) {
if registry == nil || env.ReceivedAtMS <= 0 {
return
}
elapsed := time.Since(time.UnixMilli(env.ReceivedAtMS)).Milliseconds()
if elapsed < 0 {
elapsed = 0
}
labels := metrics.Labels{"topic": message.Topic}
registry.ObserveHistogram("vehicle_stat_write_e2e_duration_ms_histogram", labels, statWriteE2EDurationBucketsMS, float64(elapsed))
p99, samples := statWriteE2ERecent.Observe(message.Topic, float64(elapsed))
registry.SetGauge("vehicle_stat_write_e2e_recent_p99_ms", labels, p99)
registry.SetGauge("vehicle_stat_write_e2e_recent_samples", labels, float64(samples))
metrics.RecordLastActivity(registry, "vehicle_stat_last_write_e2e_unix_seconds", labels)
}
func setStatBatchPending(registry *metrics.Registry, messages int) {
setStatBatchPendingForWorker(registry, nil, messages)
}
func setStatBatchPendingForWorker(registry *metrics.Registry, workerLabels metrics.Labels, messages int) {
if registry == nil {
return
}
registry.SetGauge("vehicle_stat_batch_pending_messages", workerLabels, float64(messages))
}
func statusFromError(err error) string {
if err != nil {
return "error"
}
return "ok"
}
type config struct {
KafkaBrokers []string
KafkaTopics []string
KafkaGroup string
MySQLDSN string
EnsureSchema bool
Location *time.Location
KafkaBrokers []string
KafkaTopics []string
KafkaGroup string
MySQLDSN string
EnsureSchema bool
Location *time.Location
ProjectInterval time.Duration
SourceTouchInterval time.Duration
CacheRetention time.Duration
CacheCleanupInterval time.Duration
BaselineMissTTL time.Duration
BaselineHitTTL time.Duration
CacheMaxEntries int
Workers int
BatchSize int
BatchWait int
RetryAttempts int
RetryDelay time.Duration
NormalizePlatformSourcesOnStart bool
NormalizePlatformSourcesTimeout time.Duration
QuarantineDir string
}
func (c config) Validate() error {
if len(c.KafkaTopics) == 0 {
return fmt.Errorf("KAFKA_TOPICS must include fields topics")
}
for _, topic := range c.KafkaTopics {
if !strings.HasPrefix(topic, "vehicle.fields.") {
return fmt.Errorf("stat-writer consumes fields topics only, got %q", topic)
}
}
if c.Workers <= 0 {
return fmt.Errorf("STATS_WORKERS must be greater than zero")
}
return nil
}
func loadConfig() config {
@@ -93,12 +872,27 @@ func loadConfig() config {
loc = time.FixedZone("Asia/Shanghai", 8*3600)
}
return config{
KafkaBrokers: splitCSV(env("KAFKA_BROKERS", "127.0.0.1:9092")),
KafkaTopics: splitCSV(env("KAFKA_TOPICS", "vehicle.raw.gb32960.v1,vehicle.raw.jt808.v1")),
KafkaGroup: env("KAFKA_GROUP", "go-stat-writer"),
MySQLDSN: env("MYSQL_DSN", ""),
EnsureSchema: env("MYSQL_ENSURE_SCHEMA", "true") != "false",
Location: loc,
KafkaBrokers: splitCSV(env("KAFKA_BROKERS", "127.0.0.1:9092")),
KafkaTopics: splitCSV(env("KAFKA_TOPICS", strings.Join([]string{topics.FieldsGB32960, topics.FieldsJT808, topics.FieldsYutongMQTT}, ","))),
KafkaGroup: env("KAFKA_GROUP", "go-stat-writer"),
MySQLDSN: env("MYSQL_DSN", ""),
EnsureSchema: env("MYSQL_ENSURE_SCHEMA", "true") != "false",
Location: loc,
ProjectInterval: time.Duration(envInt("STATS_PROJECT_INTERVAL_SECONDS", 15)) * time.Second,
SourceTouchInterval: time.Duration(envInt("STATS_SOURCE_TOUCH_INTERVAL_SECONDS", 60)) * time.Second,
CacheRetention: time.Duration(envInt("STATS_CACHE_RETENTION_HOURS", 72)) * time.Hour,
CacheCleanupInterval: time.Duration(envInt("STATS_CACHE_CLEANUP_INTERVAL_SECONDS", 600)) * time.Second,
BaselineMissTTL: time.Duration(envInt("STATS_BASELINE_MISS_TTL_SECONDS", 60)) * time.Second,
BaselineHitTTL: time.Duration(envInt("STATS_BASELINE_HIT_TTL_SECONDS", 300)) * time.Second,
CacheMaxEntries: envInt("STATS_CACHE_MAX_ENTRIES", 1000000),
Workers: envInt("STATS_WORKERS", 3),
BatchSize: envInt("STATS_BATCH_SIZE", 200),
BatchWait: envInt("STATS_BATCH_WAIT_MS", 20),
RetryAttempts: envInt("STATS_RETRY_ATTEMPTS", 3),
RetryDelay: time.Duration(envInt("STATS_RETRY_DELAY_MS", 20)) * time.Millisecond,
NormalizePlatformSourcesOnStart: env("STATS_NORMALIZE_PLATFORM_SOURCES_ON_START", "true") != "false",
NormalizePlatformSourcesTimeout: time.Duration(envInt("STATS_NORMALIZE_PLATFORM_SOURCES_TIMEOUT_SECONDS", 30)) * time.Second,
QuarantineDir: env("STATS_QUARANTINE_DIR", "/var/lib/lingniu-go-native/stat-writer-quarantine"),
}
}
@@ -110,6 +904,18 @@ func env(key string, fallback string) string {
return value
}
func envInt(key string, fallback int) int {
value := strings.TrimSpace(os.Getenv(key))
if value == "" {
return fallback
}
parsed, err := strconv.Atoi(value)
if err != nil {
return fallback
}
return parsed
}
func splitCSV(value string) []string {
var out []string
for _, item := range strings.Split(value, ",") {

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,954 @@
package main
import (
"context"
"os"
"path/filepath"
"strings"
"testing"
"time"
"github.com/DATA-DOG/go-sqlmock"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/stats"
)
func TestChooseTrustedSourceKeepsContinuingSourceAndRejectsNewJump(t *testing.T) {
previous := []dailySourceLast{{
VIN: "LA9GG64L7PBAF4001",
SourceKey: normalizedSourceKey("JT808", "13307765812", "", "115.231.168.135:42630"),
Phone: "13307765812",
SourceEndpoint: "115.231.168.135:42630",
TotalKM: 4100.8,
}}
current := []dailySourceLast{
{
VIN: "LA9GG64L7PBAF4001",
SourceKey: normalizedSourceKey("JT808", "13307765812", "", "115.159.85.149:28316"),
Phone: "13307765812",
SourceEndpoint: "115.159.85.149:28316",
TotalKM: 42447.2,
},
{
VIN: "LA9GG64L7PBAF4001",
SourceKey: normalizedSourceKey("JT808", "13307765812", "", "115.231.168.135:20215"),
Phone: "13307765812",
SourceEndpoint: "115.231.168.135:20215",
TotalKM: 4123.9,
},
}
chosen, ok := chooseTrustedSource(current, previous)
if !ok {
t.Fatal("chooseTrustedSource() did not choose a source")
}
if chosen.current.SourceKey != normalizedSourceKey("JT808", "13307765812", "", "115.231.168.135:20215") {
t.Fatalf("chosen source = %q", chosen.current.SourceKey)
}
if delta := chosen.current.TotalKM - chosen.previous.TotalKM; delta < 23 || delta > 24 {
t.Fatalf("delta = %v, want about 23.1", delta)
}
}
func TestChooseTrustedSourceAcceptsSmallNegativeMileageJitter(t *testing.T) {
previous := []dailySourceLast{{
VIN: "LB9A32A28R0LS1574",
SourceKey: normalizedSourceKey("JT808", "64115156034", "", "115.159.85.149:53330"),
Phone: "64115156034",
SourceEndpoint: "115.159.85.149:53330",
TotalKM: 15355.4,
}}
current := []dailySourceLast{{
VIN: "LB9A32A28R0LS1574",
SourceKey: normalizedSourceKey("JT808", "64115156034", "", "115.159.85.149:53338"),
Phone: "64115156034",
SourceEndpoint: "115.159.85.149:53338",
TotalKM: 15355.3,
}}
chosen, ok := chooseTrustedSource(current, previous)
if !ok {
t.Fatal("chooseTrustedSource() should accept tiny negative mileage jitter")
}
if chosen.current.SourceKey != current[0].SourceKey {
t.Fatalf("chosen source = %q", chosen.current.SourceKey)
}
}
func TestChooseTrustedSourceAcceptsPlausibleMultiDayFallbackDelta(t *testing.T) {
loc := time.FixedZone("Asia/Shanghai", 8*3600)
sourceKey := normalizedSourceKey("GB32960", "", "", "8.134.95.166:37720")
previous := []dailySourceLast{{
VIN: "LNXNEGRR1SR319498",
SourceKey: sourceKey,
SourceEndpoint: "8.134.95.166:37720",
TS: time.Date(2026, 7, 4, 11, 7, 58, 0, loc),
TotalKM: 8832.1,
}}
current := []dailySourceLast{{
VIN: "LNXNEGRR1SR319498",
SourceKey: sourceKey,
SourceEndpoint: "8.134.95.166:37720",
TS: time.Date(2026, 7, 12, 2, 52, 47, 0, loc),
TotalKM: 16665.6,
}}
chosen, ok := chooseTrustedSource(current, previous)
if !ok {
t.Fatal("chooseTrustedSource() should accept delta within the historical baseline window")
}
if delta := chosen.current.TotalKM - chosen.previous.TotalKM; delta < 7833.4 || delta > 7833.6 {
t.Fatalf("delta = %v, want historical gap delta", delta)
}
}
func TestDailySourceLastBuildsCandidateKeysBySourceIP(t *testing.T) {
sourceA := dailySourceLast{
VIN: "LA9GG64L7PBAF4001",
SourceKey: normalizedSourceKey("JT808", "13307765812", "", "115.231.168.135:20215"),
Phone: "13307765812",
SourceEndpoint: "115.231.168.135:20215",
TotalKM: 4123.9,
}
sourceB := dailySourceLast{
VIN: "LA9GG64L7PBAF4001",
SourceKey: normalizedSourceKey("JT808", "13307765812", "", "115.231.168.135:42630"),
Phone: "13307765812",
SourceEndpoint: "115.231.168.135:42630",
TotalKM: 4100.8,
}
if sourceA.SourceKey != sourceB.SourceKey {
t.Fatalf("same source IP should produce same source key: %q vs %q", sourceA.SourceKey, sourceB.SourceKey)
}
}
func TestAggregateFromDailySourceUsesOlderHistoricalBaseline(t *testing.T) {
current := dailySourceLast{
VIN: "LMRKH9AC2R1004087",
SourceKey: normalizedSourceKey("YUTONG_MQTT", "", "LMRKH9AC2R1004087", "mqtt://yutong/ytforward/shln/3"),
DeviceID: "LMRKH9AC2R1004087",
SourceEndpoint: "mqtt://yutong/ytforward/shln/3",
FirstTS: time.Date(2026, 7, 8, 8, 0, 0, 0, time.FixedZone("Asia/Shanghai", 8*3600)),
TS: time.Date(2026, 7, 8, 17, 0, 0, 0, time.FixedZone("Asia/Shanghai", 8*3600)),
FirstTotalKM: 120778,
TotalKM: 120788,
RawSampleCount: 15,
}
previous := dailySourceLast{
VIN: "LMRKH9AC2R1004087",
SourceKey: current.SourceKey,
DeviceID: "LMRKH9AC2R1004087",
SourceEndpoint: "mqtt://yutong/ytforward/shln/3",
TS: time.Date(2026, 7, 4, 23, 58, 0, 0, time.FixedZone("Asia/Shanghai", 8*3600)),
TotalKM: 120672,
}
agg := aggregateFromDailySource("2026-07-08", envelope.ProtocolYutongMQTT, current, previous, true)
if agg.FirstKM != 120672 || agg.LatestKM != 120788 {
t.Fatalf("km range = %v -> %v", agg.FirstKM, agg.LatestKM)
}
if agg.FirstEventTime != previous.TS || agg.LatestEventTime != current.TS {
t.Fatalf("event range = %v -> %v", agg.FirstEventTime, agg.LatestEventTime)
}
if agg.QualityStatus != stats.QualityOK || agg.QualityReason != stats.QualityReasonHistorical {
t.Fatalf("quality = %s/%s", agg.QualityStatus, agg.QualityReason)
}
if agg.Count != 15 {
t.Fatalf("sample count = %d, want 15", agg.Count)
}
}
func TestAggregateFromDailySourceRejectsHistoricalBaselineJump(t *testing.T) {
loc := time.FixedZone("Asia/Shanghai", 8*3600)
current := dailySourceLast{
VIN: "LNXNEGRR6SR319464",
SourceKey: normalizedSourceKey("GB32960", "", "", "8.134.95.166:49206"),
SourceEndpoint: "8.134.95.166:49206",
FirstTS: time.Date(2026, 7, 12, 8, 5, 42, 0, loc),
TS: time.Date(2026, 7, 12, 10, 44, 56, 0, loc),
FirstTotalKM: 28004.2,
TotalKM: 40009.7,
RawSampleCount: 1938,
}
previous := dailySourceLast{
VIN: current.VIN,
SourceKey: current.SourceKey,
SourceEndpoint: current.SourceEndpoint,
TS: time.Date(2026, 7, 3, 19, 0, 39, 0, loc),
TotalKM: 10009.7,
}
agg := aggregateFromDailySource("2026-07-12", envelope.ProtocolGB32960, current, previous, true)
if agg.FirstKM != previous.TotalKM || agg.LatestKM != current.TotalKM {
t.Fatalf("km range = %v -> %v", agg.FirstKM, agg.LatestKM)
}
if !agg.FirstEventTime.Equal(previous.TS) || !agg.LatestEventTime.Equal(current.TS) {
t.Fatalf("event range = %v -> %v", agg.FirstEventTime, agg.LatestEventTime)
}
if agg.QualityStatus != stats.QualityInvalidDelta || agg.QualityReason != "outside_daily_range" {
t.Fatalf("quality = %s/%s", agg.QualityStatus, agg.QualityReason)
}
}
func TestAggregateFromDailySourceUsesCurrentDayFirstWhenHistoryIsEmpty(t *testing.T) {
current := dailySourceLast{
VIN: "LMRKH9AC2R1004087",
SourceKey: normalizedSourceKey("YUTONG_MQTT", "", "LMRKH9AC2R1004087", "mqtt://yutong/ytforward/shln/3"),
DeviceID: "LMRKH9AC2R1004087",
SourceEndpoint: "mqtt://yutong/ytforward/shln/3",
FirstTS: time.Date(2026, 7, 8, 8, 0, 0, 0, time.FixedZone("Asia/Shanghai", 8*3600)),
TS: time.Date(2026, 7, 8, 17, 0, 0, 0, time.FixedZone("Asia/Shanghai", 8*3600)),
FirstTotalKM: 120778,
TotalKM: 120788,
RawSampleCount: 15,
}
agg := aggregateFromDailySource("2026-07-08", envelope.ProtocolYutongMQTT, current, dailySourceLast{}, false)
if agg.FirstKM != 120778 || agg.LatestKM != 120788 {
t.Fatalf("km range = %v -> %v", agg.FirstKM, agg.LatestKM)
}
if agg.FirstEventTime != current.FirstTS || agg.LatestEventTime != current.TS {
t.Fatalf("event range = %v -> %v", agg.FirstEventTime, agg.LatestEventTime)
}
if agg.QualityStatus != stats.QualityOK || agg.QualityReason != stats.QualityReasonCurrentDayFirst {
t.Fatalf("quality = %s/%s", agg.QualityStatus, agg.QualityReason)
}
}
func TestBuildLastDiffAggregatesCarriesNearestHistoryAcrossEmptyDays(t *testing.T) {
tdDB, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer tdDB.Close()
mock.MatchExpectationsInOrder(true)
loc := time.FixedZone("Asia/Shanghai", 8*3600)
vin := "LMRKH9AC2R1004087"
endpoint := "mqtt://yutong/ytforward/shln/3"
dayOneTS := time.Date(2026, 7, 9, 23, 50, 0, 0, loc)
dayThreeTS := time.Date(2026, 7, 11, 17, 30, 0, 0, loc)
currentRows := func() *sqlmock.Rows {
return sqlmock.NewRows([]string{
"vin", "phone", "device_id", "source_endpoint", "FIRST(event_time)", "FIRST(parsed_json)", "FIRST(raw_text)", "LAST(event_time)", "LAST(parsed_json)", "LAST(raw_text)", "COUNT(*)",
})
}
previousRows := func() *sqlmock.Rows {
return sqlmock.NewRows([]string{
"vin", "phone", "device_id", "source_endpoint", "LAST(event_time)", "LAST(parsed_json)", "LAST(raw_text)", "COUNT(*)",
})
}
mock.ExpectQuery("(?s)SELECT vin.*LAST\\(event_time\\).*event_time < '2026-07-09 00:00:00'.*protocol = 'YUTONG_MQTT'").
WillReturnRows(previousRows())
mock.ExpectQuery("(?s)SELECT vin.*FIRST\\(event_time\\).*event_time >= '2026-07-09 00:00:00'.*protocol = 'YUTONG_MQTT'").
WillReturnRows(currentRows().AddRow(vin, "", vin, endpoint, dayOneTS, `{"data":{"TOTAL_MILEAGE":100000}}`, "", dayOneTS, `{"data":{"TOTAL_MILEAGE":100000}}`, "", int64(4)))
mock.ExpectQuery("(?s)SELECT vin.*FIRST\\(event_time\\).*event_time >= '2026-07-10 00:00:00'.*protocol = 'YUTONG_MQTT'").
WillReturnRows(currentRows())
mock.ExpectQuery("(?s)SELECT vin.*FIRST\\(event_time\\).*event_time >= '2026-07-11 00:00:00'.*protocol = 'YUTONG_MQTT'").
WillReturnRows(currentRows().AddRow(vin, "", vin, endpoint, dayThreeTS, `{"data":{"TOTAL_MILEAGE":120000}}`, "", dayThreeTS, `{"data":{"TOTAL_MILEAGE":120000}}`, "", int64(6)))
aggregates, err := buildLastDiffAggregates(context.Background(), nil, tdDB, config{
TDengineDatabase: "lingniu_vehicle_ts",
DateFrom: "2026-07-09",
DateTo: "2026-07-11",
Protocols: []envelope.Protocol{envelope.ProtocolYutongMQTT},
Location: loc,
})
if err != nil {
t.Fatalf("buildLastDiffAggregates() error = %v", err)
}
sourceKey := stats.SourceKeyForSource(envelope.ProtocolYutongMQTT, "", vin, "mqtt", "PLATFORM", "yutong")
agg := aggregates[vin+"|2026-07-11|YUTONG_MQTT|"+sourceKey]
if agg == nil {
t.Fatalf("missing day-three aggregate; keys=%v", aggregateKeys(aggregates))
}
if agg.FirstKM != 100 || agg.LatestKM != 120 {
t.Fatalf("km range = %v -> %v, want 100 -> 120", agg.FirstKM, agg.LatestKM)
}
if !agg.FirstEventTime.Equal(dayOneTS) || agg.QualityReason != stats.QualityReasonHistorical {
t.Fatalf("baseline = %v reason=%q, want day-one historical baseline", agg.FirstEventTime, agg.QualityReason)
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}
func aggregateKeys(aggregates map[string]*metricAgg) []string {
keys := make([]string, 0, len(aggregates))
for key := range aggregates {
keys = append(keys, key)
}
return keys
}
func TestQueryRealtimeLocationLastRowsBuildsYutongSourceFromPeer(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
loc := time.FixedZone("Asia/Shanghai", 8*3600)
eventTime := time.Date(2026, 7, 12, 5, 54, 50, 0, loc)
mock.ExpectQuery("(?s)FROM vehicle_realtime_location l.*LEFT JOIN vehicle_realtime_snapshot s").
WithArgs("YUTONG_MQTT", "2026-07-12", "2026-07-12").
WillReturnRows(sqlmock.NewRows([]string{"vin", "peer", "total_mileage_event_time", "total_mileage_km"}).
AddRow("LMRKH9AC0R1004086", "mqtt://yutong/ytforward/shln/4", eventTime, 11578.0))
rows, err := queryRealtimeLocationLastRows(context.Background(), db, config{Location: loc}, envelope.ProtocolYutongMQTT, "2026-07-12")
if err != nil {
t.Fatalf("queryRealtimeLocationLastRows() error = %v", err)
}
sourceRows := rows["LMRKH9AC0R1004086"]
if len(sourceRows) != 1 {
t.Fatalf("rows = %d, want 1", len(sourceRows))
}
row := sourceRows[0]
wantSourceKey := stats.SourceKeyForSource(envelope.ProtocolYutongMQTT, "", "LMRKH9AC0R1004086", "mqtt", "PLATFORM", "yutong")
if row.SourceKey != wantSourceKey {
t.Fatalf("source key = %q", row.SourceKey)
}
if row.SourceCode != "yutong" || row.PlatformName != "宇通" || row.SourceKind != "PLATFORM" {
t.Fatalf("source metadata = code:%q platform:%q kind:%q", row.SourceCode, row.PlatformName, row.SourceKind)
}
if row.TotalKM != 11578 || row.DeviceID != "LMRKH9AC0R1004086" || row.RawSampleCount != 1 {
t.Fatalf("unexpected realtime fallback row: %#v", row)
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}
func TestAddRealtimeLocationFallbackAggregatesUsesHistoricalBaseline(t *testing.T) {
mysqlDB, mysqlMock, err := sqlmock.New()
if err != nil {
t.Fatalf("mysql sqlmock.New() error = %v", err)
}
defer mysqlDB.Close()
tdDB, tdMock, err := sqlmock.New()
if err != nil {
t.Fatalf("td sqlmock.New() error = %v", err)
}
defer tdDB.Close()
loc := time.FixedZone("Asia/Shanghai", 8*3600)
currentTS := time.Date(2026, 7, 12, 5, 54, 50, 0, loc)
previousTS := time.Date(2026, 7, 11, 23, 58, 0, 0, loc)
mysqlMock.ExpectQuery("(?s)FROM vehicle_realtime_location l.*l.protocol = \\?").
WithArgs("YUTONG_MQTT", "2026-07-12", "2026-07-12").
WillReturnRows(sqlmock.NewRows([]string{"vin", "peer", "total_mileage_event_time", "total_mileage_km"}).
AddRow("LMRKH9AC0R1004086", "mqtt://yutong/ytforward/shln/4", currentTS, 11578.0))
sourceKey := stats.SourceKeyForSource(envelope.ProtocolYutongMQTT, "", "LMRKH9AC0R1004086", "mqtt", "PLATFORM", "yutong")
mysqlMock.ExpectQuery("(?s)SELECT latest_total_mileage_km, latest_event_time.*FROM vehicle_daily_mileage_source").
WithArgs("LMRKH9AC0R1004086", "2026-07-12", "YUTONG_MQTT", sourceKey).
WillReturnRows(sqlmock.NewRows([]string{"latest_total_mileage_km", "latest_event_time"}).AddRow(11500.0, previousTS))
aggregates := map[string]*metricAgg{}
added, err := addRealtimeLocationFallbackAggregates(context.Background(), mysqlDB, tdDB, config{
TDengineDatabase: "lingniu_vehicle_ts",
DateFrom: "2026-07-12",
DateTo: "2026-07-12",
Protocols: []envelope.Protocol{envelope.ProtocolYutongMQTT},
Location: loc,
}, aggregates)
if err != nil {
t.Fatalf("addRealtimeLocationFallbackAggregates() error = %v", err)
}
if added != 1 || len(aggregates) != 1 {
t.Fatalf("added=%d aggregates=%d", added, len(aggregates))
}
for _, agg := range aggregates {
if agg.FirstKM != 11500 || agg.LatestKM != 11578 {
t.Fatalf("km range = %v -> %v", agg.FirstKM, agg.LatestKM)
}
if agg.SourceKey != sourceKey {
t.Fatalf("source key = %q", agg.SourceKey)
}
if agg.SourceCode != "yutong" || agg.PlatformName != "宇通" || agg.SourceKind != "PLATFORM" {
t.Fatalf("source metadata = code:%q platform:%q kind:%q", agg.SourceCode, agg.PlatformName, agg.SourceKind)
}
if agg.QualityReason != "realtime_location_fallback_historical_baseline" {
t.Fatalf("quality reason = %q", agg.QualityReason)
}
}
if err := mysqlMock.ExpectationsWereMet(); err != nil {
t.Fatalf("mysql sql expectations: %v", err)
}
if err := tdMock.ExpectationsWereMet(); err != nil {
t.Fatalf("td sql expectations: %v", err)
}
}
func TestAddRealtimeLocationFallbackReusesAggregateHistoryAcrossEmptyDays(t *testing.T) {
mysqlDB, mysqlMock, err := sqlmock.New()
if err != nil {
t.Fatalf("mysql sqlmock.New() error = %v", err)
}
defer mysqlDB.Close()
tdDB, tdMock, err := sqlmock.New()
if err != nil {
t.Fatalf("td sqlmock.New() error = %v", err)
}
defer tdDB.Close()
loc := time.FixedZone("Asia/Shanghai", 8*3600)
vin := "LMRKH9AC2R1004087"
endpoint := "mqtt://yutong/ytforward/shln/3"
sourceKey := stats.SourceKeyForSource(envelope.ProtocolYutongMQTT, "", vin, "mqtt", "PLATFORM", "yutong")
dayOneTS := time.Date(2026, 7, 9, 23, 50, 0, 0, loc)
dayThreeTS := time.Date(2026, 7, 11, 17, 30, 0, 0, loc)
for _, date := range []string{"2026-07-09", "2026-07-10"} {
mysqlMock.ExpectQuery("(?s)FROM vehicle_realtime_location l.*l.protocol = \\?").
WithArgs("YUTONG_MQTT", date, date).
WillReturnRows(sqlmock.NewRows([]string{"vin", "peer", "total_mileage_event_time", "total_mileage_km"}))
}
mysqlMock.ExpectQuery("(?s)FROM vehicle_realtime_location l.*l.protocol = \\?").
WithArgs("YUTONG_MQTT", "2026-07-11", "2026-07-11").
WillReturnRows(sqlmock.NewRows([]string{"vin", "peer", "total_mileage_event_time", "total_mileage_km"}).
AddRow(vin, endpoint, dayThreeTS, 120.0))
aggregates := map[string]*metricAgg{
vin + "|2026-07-09|YUTONG_MQTT|" + sourceKey: {
VIN: vin,
Date: "2026-07-09",
Protocol: envelope.ProtocolYutongMQTT,
LatestKM: 100,
Count: 4,
SourceKey: sourceKey,
DeviceID: vin,
SourceEndpoint: endpoint,
SourceCode: "yutong",
PlatformName: "宇通",
SourceKind: "PLATFORM",
LatestEventTime: dayOneTS,
},
}
added, err := addRealtimeLocationFallbackAggregates(context.Background(), mysqlDB, tdDB, config{
TDengineDatabase: "lingniu_vehicle_ts",
DateFrom: "2026-07-09",
DateTo: "2026-07-11",
Protocols: []envelope.Protocol{envelope.ProtocolYutongMQTT},
Location: loc,
}, aggregates)
if err != nil {
t.Fatalf("addRealtimeLocationFallbackAggregates() error = %v", err)
}
if added != 1 || len(aggregates) != 2 {
t.Fatalf("added=%d aggregates=%d, want 1 and 2", added, len(aggregates))
}
agg := aggregates[vin+"|2026-07-11|YUTONG_MQTT|"+sourceKey]
if agg == nil {
t.Fatal("missing realtime-location fallback aggregate")
}
if agg.FirstKM != 100 || agg.LatestKM != 120 || !agg.FirstEventTime.Equal(dayOneTS) {
t.Fatalf("fallback range = %v@%v -> %v, want 100@day-one -> 120", agg.FirstKM, agg.FirstEventTime, agg.LatestKM)
}
if agg.QualityReason != "realtime_location_fallback_historical_baseline" {
t.Fatalf("quality reason = %q", agg.QualityReason)
}
if err := mysqlMock.ExpectationsWereMet(); err != nil {
t.Fatalf("mysql sql expectations: %v", err)
}
if err := tdMock.ExpectationsWereMet(); err != nil {
t.Fatalf("td sql expectations: %v", err)
}
}
func TestQueryDailyLastSourceRowsFiltersYutongToMileageFrames(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
loc := time.FixedZone("Asia/Shanghai", 8*3600)
firstTS := time.Date(2026, 7, 8, 8, 0, 0, 0, loc)
lastTS := time.Date(2026, 7, 8, 18, 0, 0, 0, loc)
mock.ExpectQuery("(?s)FROM lingniu_vehicle_ts\\.raw_frames.*protocol = 'YUTONG_MQTT'.*parsed_json LIKE '%yutong_mqtt\\.data\\.total_mileage%'.*parsed_json LIKE '%TOTAL_MILEAGE%'").
WillReturnRows(sqlmock.NewRows([]string{
"vin", "phone", "device_id", "source_endpoint", "FIRST(event_time)", "FIRST(parsed_json)", "FIRST(raw_text)", "LAST(event_time)", "LAST(parsed_json)", "LAST(raw_text)", "COUNT(*)",
}).AddRow(
"LMRKH9AC6R1004108",
"",
"LMRKH9AC6R1004108",
"mqtt://yutong/ytforward/shln/3",
firstTS,
`{"yutong_mqtt.data.latitude":"30.0"}`,
`{"data":{"TOTAL_MILEAGE":65422000}}`,
lastTS,
`{"yutong_mqtt.data.latitude":"30.1"}`,
`{"data":{"TOTAL_MILEAGE":65423000}}`,
int64(42),
))
rows, err := queryDailyLastSourceRows(context.Background(), db, config{
TDengineDatabase: "lingniu_vehicle_ts",
Location: loc,
}, envelope.ProtocolYutongMQTT, "2026-07-08")
if err != nil {
t.Fatalf("queryDailyLastSourceRows() error = %v", err)
}
sourceRows := rows["LMRKH9AC6R1004108"]
if len(sourceRows) != 1 {
t.Fatalf("rows = %d, want 1", len(sourceRows))
}
if sourceRows[0].FirstTotalKM != 65422 || sourceRows[0].TotalKM != 65423 {
t.Fatalf("km range = %v -> %v", sourceRows[0].FirstTotalKM, sourceRows[0].TotalKM)
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}
func TestQueryPreviousLastSourceRowsFiltersYutongToMileageFrames(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
loc := time.FixedZone("Asia/Shanghai", 8*3600)
lastTS := time.Date(2026, 7, 7, 23, 58, 0, 0, loc)
mock.ExpectQuery("(?s)FROM lingniu_vehicle_ts\\.raw_frames.*event_time < '2026-07-08 00:00:00'.*protocol = 'YUTONG_MQTT'.*parsed_json LIKE '%yutong_mqtt\\.data\\.total_mileage%'.*parsed_json LIKE '%TOTAL_MILEAGE%'").
WillReturnRows(sqlmock.NewRows([]string{
"vin", "phone", "device_id", "source_endpoint", "LAST(event_time)", "LAST(parsed_json)", "LAST(raw_text)", "COUNT(*)",
}).AddRow(
"LMRKH9AC6R1004108",
"",
"LMRKH9AC6R1004108",
"mqtt://yutong/ytforward/shln/3",
lastTS,
`{"yutong_mqtt.data.latitude":"30.0"}`,
`{"data":{"TOTAL_MILEAGE":65377000}}`,
int64(31),
))
rows, err := queryPreviousLastSourceRows(context.Background(), db, config{
TDengineDatabase: "lingniu_vehicle_ts",
Location: loc,
}, envelope.ProtocolYutongMQTT, "2026-07-08")
if err != nil {
t.Fatalf("queryPreviousLastSourceRows() error = %v", err)
}
sourceRows := rows["LMRKH9AC6R1004108"]
if len(sourceRows) != 1 {
t.Fatalf("rows = %d, want 1", len(sourceRows))
}
if sourceRows[0].TotalKM != 65377 {
t.Fatalf("previous total = %v, want 65377", sourceRows[0].TotalKM)
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}
func TestBackfillBeforePredicatesBoundsHistoryScan(t *testing.T) {
where := strings.Join(backfillBeforePredicates(config{BaselineLookback: 7}, "2026-07-08"), " AND ")
if !strings.Contains(where, "event_time < '2026-07-08 00:00:00'") {
t.Fatalf("pre-window predicate missing exclusive upper bound: %s", where)
}
for _, predicate := range []string{
"event_time >= '2026-07-01 00:00:00'",
"ts >= '2026-06-30 00:00:00'",
"ts < '2026-07-09 00:00:00'",
} {
if !strings.Contains(where, predicate) {
t.Fatalf("pre-window predicate missing %q: %s", predicate, where)
}
}
}
func TestQueryPreviousLastSourceRowsNormalizesScannedTimestampToBusinessTimezone(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
loc := time.FixedZone("Asia/Shanghai", 8*3600)
utcInstant := time.Date(2026, 7, 12, 15, 59, 59, 0, time.UTC)
mock.ExpectQuery("(?s)FROM lingniu_vehicle_ts\\.raw_frames.*protocol = 'YUTONG_MQTT'").
WillReturnRows(sqlmock.NewRows([]string{
"vin", "phone", "device_id", "source_endpoint", "LAST(event_time)", "LAST(parsed_json)", "LAST(raw_text)", "COUNT(*)",
}).AddRow(
"LMRKH9AC7R1004098",
"",
"LMRKH9AC7R1004098",
"mqtt://yutong/ytforward/shln/4",
utcInstant,
`{"yutong_mqtt.data.total_mileage":"41249000"}`,
"",
int64(1),
))
rows, err := queryPreviousLastSourceRows(context.Background(), db, config{
TDengineDatabase: "lingniu_vehicle_ts",
Location: loc,
}, envelope.ProtocolYutongMQTT, "2026-07-13")
if err != nil {
t.Fatalf("queryPreviousLastSourceRows() error = %v", err)
}
sourceRows := rows["LMRKH9AC7R1004098"]
if len(sourceRows) != 1 {
t.Fatalf("rows = %d, want 1", len(sourceRows))
}
want := time.Date(2026, 7, 12, 23, 59, 59, 0, loc)
if !sourceRows[0].TS.Equal(want) || sourceRows[0].TS.Location().String() != loc.String() {
t.Fatalf("previous event time = %s (%s), want %s (%s)", sourceRows[0].TS, sourceRows[0].TS.Location(), want, want.Location())
}
}
func TestFieldsForStatsExtractsRawYutongTotalMileage(t *testing.T) {
fields := fieldsForStats(envelope.ProtocolYutongMQTT, "LMRKH9AC6R1004108", `{
"data": {
"TOTAL_MILEAGE": 65423000,
"METER_SPEED": 12.3
},
"root": {
"device": "LMRKH9AC6R1004108"
}
}`)
if got := fields["yutong_mqtt.data.total_mileage"]; got == nil {
t.Fatalf("fields missing raw yutong total mileage: %#v", fields)
}
env := envelope.FrameEnvelope{
Protocol: envelope.ProtocolYutongMQTT,
VIN: "LMRKH9AC6R1004108",
EventTimeMS: time.Date(2026, 7, 8, 8, 0, 0, 0, time.FixedZone("Asia/Shanghai", 8*3600)).UnixMilli(),
ReceivedAtMS: time.Date(2026, 7, 8, 8, 0, 0, 0, time.FixedZone("Asia/Shanghai", 8*3600)).UnixMilli(),
Fields: fields,
}
samples, err := stats.SamplesFromEnvelope(env, time.FixedZone("Asia/Shanghai", 8*3600))
if err != nil {
t.Fatalf("SamplesFromEnvelope() error = %v", err)
}
if len(samples) != 1 {
t.Fatalf("samples = %d, want 1", len(samples))
}
if samples[0].TotalMileageKM != 65423 {
t.Fatalf("total mileage km = %v, want 65423", samples[0].TotalMileageKM)
}
}
func TestClearBackfillTargetMileageClearsExactKey(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
mock.ExpectExec(`DELETE FROM vehicle_daily_mileage_source WHERE vin = \? AND stat_date = \? AND protocol = \?`).
WithArgs("LA9GG64L7PBAF4001", "2026-07-08", "JT808").
WillReturnResult(sqlmock.NewResult(0, 1))
mock.ExpectExec(`DELETE FROM vehicle_daily_mileage WHERE vin = \? AND stat_date = \? AND protocol = \?`).
WithArgs("LA9GG64L7PBAF4001", "2026-07-08", "JT808").
WillReturnResult(sqlmock.NewResult(0, 1))
if err := clearBackfillTargetMileage(context.Background(), db, "LA9GG64L7PBAF4001", "2026-07-08", envelope.ProtocolJT808); err != nil {
t.Fatalf("clearBackfillTargetMileage() error = %v", err)
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}
func TestWriteAggregatesClearsTargetRowsBeforeStaleCandidateUpsert(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
mock.MatchExpectationsInOrder(true)
projNow := time.Date(2026, 7, 8, 12, 0, 0, 0, time.FixedZone("Asia/Shanghai", 8*3600))
aggregates := map[string]*metricAgg{
"LA9GG64L7PBAF4001|2026-07-08|JT808|JT808:13307765812@115.231.168.135": {
VIN: "LA9GG64L7PBAF4001",
Date: "2026-07-08",
Protocol: envelope.ProtocolJT808,
FirstKM: 4100,
LatestKM: 4101,
Count: 1,
SourceKey: "JT808:13307765812@115.231.168.135",
Phone: "13307765812",
SourceEndpoint: "115.231.168.135:20215",
FirstEventTime: projNow,
LatestEventTime: projNow,
QualityStatus: stats.QualityInvalidDelta,
QualityReason: "outside_daily_range",
},
}
mock.ExpectExec(`DELETE FROM vehicle_daily_mileage_source WHERE vin = \? AND stat_date = \? AND protocol = \?`).
WithArgs("LA9GG64L7PBAF4001", "2026-07-08", "JT808").
WillReturnResult(sqlmock.NewResult(0, 1))
mock.ExpectExec(`DELETE FROM vehicle_daily_mileage WHERE vin = \? AND stat_date = \? AND protocol = \?`).
WithArgs("LA9GG64L7PBAF4001", "2026-07-08", "JT808").
WillReturnResult(sqlmock.NewResult(0, 1))
mock.ExpectExec(`INSERT INTO vehicle_data_source`).
WithArgs("JT808", "115.231.168.135", "115.231.168.135:20215", sqlmock.AnyArg(), sqlmock.AnyArg(), sqlmock.AnyArg(), sqlmock.AnyArg(), sqlmock.AnyArg()).
WillReturnResult(sqlmock.NewResult(0, 1))
mock.ExpectExec(`INSERT INTO vehicle_daily_mileage_source`).
WithArgs(
"LA9GG64L7PBAF4001",
"2026-07-08",
"JT808",
"JT808:13307765812@115.231.168.135",
"115.231.168.135",
"115.231.168.135:20215",
"13307765812",
"",
"",
float64(4100),
float64(4101),
float64(1),
int64(1),
projNow,
projNow,
stats.QualityInvalidDelta,
"outside_daily_range",
).
WillReturnResult(sqlmock.NewResult(0, 1))
mock.ExpectBegin()
mock.ExpectExec(`INSERT INTO vehicle_daily_mileage`).
WithArgs("LA9GG64L7PBAF4001", "2026-07-08", "JT808", int64(2500)).
WillReturnResult(sqlmock.NewResult(1, 0))
mock.ExpectExec(`UPDATE vehicle_daily_mileage_source s`).
WithArgs(
"LA9GG64L7PBAF4001",
"2026-07-08",
"JT808",
int64(2500),
"LA9GG64L7PBAF4001",
"2026-07-08",
"JT808",
).
WillReturnResult(sqlmock.NewResult(0, 0))
mock.ExpectExec(`DELETE FROM vehicle_daily_mileage`).
WithArgs(
"LA9GG64L7PBAF4001",
"2026-07-08",
"JT808",
"LA9GG64L7PBAF4001",
"2026-07-08",
"JT808",
).
WillReturnResult(sqlmock.NewResult(0, 1))
mock.ExpectCommit()
written, err := writeAggregates(context.Background(), db, aggregates, 500)
if err != nil {
t.Fatalf("writeAggregates() error = %v", err)
}
if written != 1 {
t.Fatalf("written = %d, want 1", written)
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}
func TestWriteAggregatesSkipsBlankSourceBeforeClearingTarget(t *testing.T) {
db, mock, err := sqlmock.New()
if err != nil {
t.Fatalf("sqlmock.New() error = %v", err)
}
defer db.Close()
written, err := writeAggregates(context.Background(), db, map[string]*metricAgg{
"LA9GG64L7PBAF4001|2026-07-08|JT808|JT808:13307765812@": {
VIN: "LA9GG64L7PBAF4001",
Date: "2026-07-08",
Protocol: envelope.ProtocolJT808,
FirstKM: 4100,
LatestKM: 4101,
Count: 1,
SourceKey: "JT808:13307765812@",
Phone: "13307765812",
SourceEndpoint: "",
FirstEventTime: time.Date(2026, 7, 8, 12, 0, 0, 0, time.FixedZone("Asia/Shanghai", 8*3600)),
LatestEventTime: time.Date(2026, 7, 8, 12, 0, 0, 0, time.FixedZone("Asia/Shanghai", 8*3600)),
QualityStatus: stats.QualityNoPreviousBaseline,
QualityReason: "missing_previous_source",
},
}, 500)
if err != nil {
t.Fatalf("writeAggregates() error = %v", err)
}
if written != 0 {
t.Fatalf("written = %d, want 0", written)
}
if err := mock.ExpectationsWereMet(); err != nil {
t.Fatalf("sql expectations: %v", err)
}
}
func TestAddSamplesUsesCurrentDayFirstSampleBaseline(t *testing.T) {
loc := time.FixedZone("Asia/Shanghai", 8*3600)
firstSamples, err := stats.SamplesFromEnvelope(envelope.FrameEnvelope{
Protocol: envelope.ProtocolJT808,
VIN: "LA9GG64L7PBAF4001",
Phone: "13307765812",
SourceEndpoint: "115.231.168.135:20215",
EventTimeMS: time.Date(2026, 7, 8, 12, 0, 0, 0, loc).UnixMilli(),
Fields: map[string]any{
"jt808.location.total_mileage_km": 4123.9,
},
}, loc)
if err != nil {
t.Fatalf("SamplesFromEnvelope() error = %v", err)
}
secondSamples, err := stats.SamplesFromEnvelope(envelope.FrameEnvelope{
Protocol: envelope.ProtocolJT808,
VIN: "LA9GG64L7PBAF4001",
Phone: "13307765812",
SourceEndpoint: "115.231.168.135:20215",
EventTimeMS: time.Date(2026, 7, 8, 13, 0, 0, 0, loc).UnixMilli(),
Fields: map[string]any{
"jt808.location.total_mileage_km": 4129.9,
},
}, loc)
if err != nil {
t.Fatalf("SamplesFromEnvelope() error = %v", err)
}
aggregates := map[string]*metricAgg{}
addSamples(aggregates, firstSamples)
addSamples(aggregates, secondSamples)
if len(aggregates) != 1 {
t.Fatalf("aggregate count = %d, want 1", len(aggregates))
}
for _, agg := range aggregates {
if agg.QualityStatus != stats.QualityOK {
t.Fatalf("quality = %q, want %q", agg.QualityStatus, stats.QualityOK)
}
if agg.QualityReason != "current_day_first_sample" {
t.Fatalf("quality reason = %q", agg.QualityReason)
}
if agg.FirstKM != 4123.9 || agg.LatestKM != 4129.9 {
t.Fatalf("km range = %v -> %v", agg.FirstKM, agg.LatestKM)
}
if agg.FirstEventTime != time.Date(2026, 7, 8, 12, 0, 0, 0, loc) {
t.Fatalf("first event time = %v", agg.FirstEventTime)
}
}
}
func TestLoadConfigDefaultsBackfillMethodToLastDiff(t *testing.T) {
t.Setenv("BACKFILL_METHOD", "")
t.Setenv("BACKFILL_DAYS_BACK", "")
t.Setenv("BACKFILL_WINDOW_DAYS", "")
t.Setenv("BACKFILL_DATE_FROM", "2026-07-08")
t.Setenv("BACKFILL_DATE_TO", "2026-07-08")
t.Setenv("BACKFILL_PROTOCOLS", "JT808")
t.Setenv("LOCAL_TZ", "Asia/Shanghai")
cfg, err := loadConfig()
if err != nil {
t.Fatalf("loadConfig() error = %v", err)
}
if cfg.Method != "last_diff" {
t.Fatalf("method = %q, want last_diff", cfg.Method)
}
if cfg.EventTimeFullScan {
t.Fatal("event-time full scan should be opt-in")
}
}
func TestBackfillTimePredicatesUsePrimaryTimeForCoarseScanAndEventTimeForBusinessDay(t *testing.T) {
where := strings.Join(backfillTimePredicates(config{}, "2026-07-13", "2026-07-14"), " AND ")
for _, want := range []string{
"ts >= '2026-07-12 00:00:00'",
"ts < '2026-07-15 00:00:00'",
"event_time >= '2026-07-13 00:00:00'",
"event_time < '2026-07-14 00:00:00'",
} {
if !strings.Contains(where, want) {
t.Fatalf("time predicates missing %q: %s", want, where)
}
}
}
func TestBackfillTimePredicatesAllowExplicitDeepEventTimeScan(t *testing.T) {
where := strings.Join(backfillTimePredicates(config{EventTimeFullScan: true}, "2026-07-13", "2026-07-14"), " AND ")
if strings.Contains(where, "ts >=") || strings.Contains(where, "ts <") {
t.Fatalf("deep event-time scan must not apply storage-time bounds: %s", where)
}
for _, want := range []string{
"event_time >= '2026-07-13 00:00:00'",
"event_time < '2026-07-14 00:00:00'",
} {
if !strings.Contains(where, want) {
t.Fatalf("deep event-time scan missing %q: %s", want, where)
}
}
}
func TestResolveBackfillDateRangeDefaultsToToday(t *testing.T) {
t.Setenv("BACKFILL_DATE_FROM", "")
t.Setenv("BACKFILL_DATE_TO", "")
t.Setenv("BACKFILL_DAYS_BACK", "")
t.Setenv("BACKFILL_WINDOW_DAYS", "")
loc := time.FixedZone("Asia/Shanghai", 8*3600)
dateFrom, dateTo := resolveBackfillDateRange(time.Date(2026, 7, 12, 10, 0, 0, 0, loc), loc)
if dateFrom != "2026-07-12" || dateTo != "2026-07-12" {
t.Fatalf("date range = %s -> %s", dateFrom, dateTo)
}
}
func TestResolveBackfillDateRangeUsesRelativeWindow(t *testing.T) {
t.Setenv("BACKFILL_DATE_FROM", "")
t.Setenv("BACKFILL_DATE_TO", "")
t.Setenv("BACKFILL_DAYS_BACK", "1")
t.Setenv("BACKFILL_WINDOW_DAYS", "3")
loc := time.FixedZone("Asia/Shanghai", 8*3600)
dateFrom, dateTo := resolveBackfillDateRange(time.Date(2026, 7, 12, 10, 0, 0, 0, loc), loc)
if dateFrom != "2026-07-09" || dateTo != "2026-07-11" {
t.Fatalf("date range = %s -> %s", dateFrom, dateTo)
}
}
func TestResolveBackfillDateRangePrefersExplicitDates(t *testing.T) {
t.Setenv("BACKFILL_DATE_FROM", "2026-07-01")
t.Setenv("BACKFILL_DATE_TO", "2026-07-03")
t.Setenv("BACKFILL_DAYS_BACK", "1")
t.Setenv("BACKFILL_WINDOW_DAYS", "3")
loc := time.FixedZone("Asia/Shanghai", 8*3600)
dateFrom, dateTo := resolveBackfillDateRange(time.Date(2026, 7, 12, 10, 0, 0, 0, loc), loc)
if dateFrom != "2026-07-01" || dateTo != "2026-07-03" {
t.Fatalf("date range = %s -> %s", dateFrom, dateTo)
}
}
func TestBackfillEnvFilesPrefersExplicitList(t *testing.T) {
got := backfillEnvFiles("/tmp/a.env,/tmp/b.env", "/tmp/legacy.env", []string{"/tmp/default.env"})
if got != "/tmp/a.env,/tmp/b.env" {
t.Fatalf("env files = %q", got)
}
}
func TestBackfillEnvFilesUsesExistingDefaults(t *testing.T) {
dir := t.TempDir()
missing := filepath.Join(dir, "missing.env")
base := filepath.Join(dir, "base.env")
stat := filepath.Join(dir, "stat-writer.env")
if err := os.WriteFile(base, []byte("A=1\n"), 0600); err != nil {
t.Fatalf("write base env: %v", err)
}
if err := os.WriteFile(stat, []byte("B=2\n"), 0600); err != nil {
t.Fatalf("write stat env: %v", err)
}
got := backfillEnvFiles("", "", []string{missing, base, stat})
want := base + "," + stat
if got != want {
t.Fatalf("env files = %q, want %q", got, want)
}
}

View File

@@ -7,10 +7,12 @@ require (
github.com/alicebob/miniredis/v2 v2.35.0
github.com/eclipse/paho.mqtt.golang v1.5.1
github.com/go-sql-driver/mysql v1.9.3
github.com/nats-io/nats.go v1.52.0
github.com/redis/go-redis/v9 v9.17.2
github.com/segmentio/kafka-go v0.4.49
github.com/taosdata/driver-go/v3 v3.8.1
golang.org/x/text v0.29.0
github.com/xuri/excelize/v2 v2.11.0
golang.org/x/text v0.38.0
)
require (
@@ -20,11 +22,20 @@ require (
github.com/google/uuid v1.6.0 // indirect
github.com/gorilla/websocket v1.5.3 // indirect
github.com/json-iterator/go v1.1.12 // indirect
github.com/klauspost/compress v1.15.9 // indirect
github.com/klauspost/compress v1.18.5 // indirect
github.com/modern-go/concurrent v0.0.0-20180228061459-e0a39a4cb421 // indirect
github.com/modern-go/reflect2 v1.0.2 // indirect
github.com/nats-io/nkeys v0.4.15 // indirect
github.com/nats-io/nuid v1.0.1 // indirect
github.com/pierrec/lz4/v4 v4.1.15 // indirect
github.com/richardlehane/mscfb v1.0.7 // indirect
github.com/richardlehane/msoleps v1.0.6 // indirect
github.com/tiendc/go-deepcopy v1.7.2 // indirect
github.com/xuri/efp v0.0.1 // indirect
github.com/xuri/nfp v0.0.2-0.20250530014748-2ddeb826f9a9 // indirect
github.com/yuin/gopher-lua v1.1.1 // indirect
golang.org/x/net v0.44.0 // indirect
golang.org/x/sync v0.17.0 // indirect
golang.org/x/crypto v0.53.0 // indirect
golang.org/x/net v0.56.0 // indirect
golang.org/x/sync v0.21.0 // indirect
golang.org/x/sys v0.46.0 // indirect
)

View File

@@ -28,18 +28,28 @@ github.com/gorilla/websocket v1.5.3/go.mod h1:YR8l580nyteQvAITg2hZ9XVh4b55+EU/ad
github.com/json-iterator/go v1.1.12 h1:PV8peI4a0ysnczrg+LtxykD8LfKY9ML6u2jnxaEnrnM=
github.com/json-iterator/go v1.1.12/go.mod h1:e30LSqwooZae/UwlEbR2852Gd8hjQvJoHmT4TnhNGBo=
github.com/kisielk/sqlstruct v0.0.0-20201105191214-5f3e10d3ab46/go.mod h1:yyMNCyc/Ib3bDTKd379tNMpB/7/H5TjM2Y9QJ5THLbE=
github.com/klauspost/compress v1.15.9 h1:wKRjX6JRtDdrE9qwa4b/Cip7ACOshUI4smpCQanqjSY=
github.com/klauspost/compress v1.15.9/go.mod h1:PhcZ0MbTNciWF3rruxRgKxI5NkcHHrHUDtV4Yw2GlzU=
github.com/klauspost/compress v1.18.5 h1:/h1gH5Ce+VWNLSWqPzOVn6XBO+vJbCNGvjoaGBFW2IE=
github.com/klauspost/compress v1.18.5/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ=
github.com/modern-go/concurrent v0.0.0-20180228061459-e0a39a4cb421 h1:ZqeYNhU3OHLH3mGKHDcjJRFFRrJa6eAM5H+CtDdOsPc=
github.com/modern-go/concurrent v0.0.0-20180228061459-e0a39a4cb421/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q=
github.com/modern-go/reflect2 v1.0.2 h1:xBagoLtFs94CBntxluKeaWgTMpvLxC4ur3nMaC9Gz0M=
github.com/modern-go/reflect2 v1.0.2/go.mod h1:yWuevngMOJpCy52FWWMvUC8ws7m/LJsjYzDa0/r8luk=
github.com/nats-io/nats.go v1.52.0 h1:n3avV4VBsCgsdwh71TppsTwtv+QdPs7ntSKM8qJLGsc=
github.com/nats-io/nats.go v1.52.0/go.mod h1:26HypzazeOkyO3/mqd1zZd53STJN0EjCYF9Uy2ZOBno=
github.com/nats-io/nkeys v0.4.15 h1:JACV5jRVO9V856KOapQ7x+EY8Jo3qw1vJt/9Jpwzkk4=
github.com/nats-io/nkeys v0.4.15/go.mod h1:CpMchTXC9fxA5zrMo4KpySxNjiDVvr8ANOSZdiNfUrs=
github.com/nats-io/nuid v1.0.1 h1:5iA8DT8V7q8WK2EScv2padNa/rTESc1KdnPw4TC2paw=
github.com/nats-io/nuid v1.0.1/go.mod h1:19wcPz3Ph3q0Jbyiqsd0kePYG7A95tJPxeL+1OSON2c=
github.com/pierrec/lz4/v4 v4.1.15 h1:MO0/ucJhngq7299dKLwIMtgTfbkoSPF6AoMYDd8Q4q0=
github.com/pierrec/lz4/v4 v4.1.15/go.mod h1:gZWDp/Ze/IJXGXf23ltt2EXimqmTUXEy0GFuRQyBid4=
github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM=
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
github.com/redis/go-redis/v9 v9.17.2 h1:P2EGsA4qVIM3Pp+aPocCJ7DguDHhqrXNhVcEp4ViluI=
github.com/redis/go-redis/v9 v9.17.2/go.mod h1:u410H11HMLoB+TP67dz8rL9s6QW2j76l0//kSOd3370=
github.com/richardlehane/mscfb v1.0.7 h1:oeoiM0WE79vHwE8RpIYYvIAc8ajTH2mb6UZm55/+EB0=
github.com/richardlehane/mscfb v1.0.7/go.mod h1:pe0+IUIc0AHh0+teNzBlJCtSyZdFOGgV4ZK9bsoV+Jo=
github.com/richardlehane/msoleps v1.0.6 h1:9BvkpjvD+iUBalUY4esMwv6uBkfOip/Lzvd93jvR9gg=
github.com/richardlehane/msoleps v1.0.6/go.mod h1:BWev5JBpU9Ko2WAgmZEuiz4/u3ZYTKbjLycmwiWUfWg=
github.com/segmentio/kafka-go v0.4.49 h1:GJiNX1d/g+kG6ljyJEoi9++PUMdXGAxb7JGPiDCuNmk=
github.com/segmentio/kafka-go v0.4.49/go.mod h1:Y1gn60kzLEEaW28YshXyk2+VCUKbJ3Qr6DrnT3i4+9E=
github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME=
@@ -49,24 +59,39 @@ github.com/stretchr/objx v0.5.0/go.mod h1:Yh+to48EsGEfYuaHDzXPcE3xhTkx73EhmCGUpE
github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI=
github.com/stretchr/testify v1.7.1/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg=
github.com/stretchr/testify v1.8.0/go.mod h1:yNjHg4UonilssWZ8iaSj1OCr/vHnekPRkoO+kdMU+MU=
github.com/stretchr/testify v1.8.2 h1:+h33VjcLVPDHtOdpUCuF+7gSuG3yGIftsP1YvFihtJ8=
github.com/stretchr/testify v1.8.2/go.mod h1:w2LPCIKwWwSfY2zedu0+kehJoqGctiVI29o6fzry7u4=
github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U=
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
github.com/taosdata/driver-go/v3 v3.8.1 h1:kkd4ABsGiU+oXDbsw/sic985LKAvnpF9Gb/TEunTnLE=
github.com/taosdata/driver-go/v3 v3.8.1/go.mod h1:S6OGOinfR0xxxaMGsvBi9cLkYxEIW1p6qqr8QJATTlg=
github.com/tiendc/go-deepcopy v1.7.2 h1:Ut2yYR7W9tWjTQitganoIue4UGxZwCcJy3orjrrIj44=
github.com/tiendc/go-deepcopy v1.7.2/go.mod h1:4bKjNC2r7boYOkD2IOuZpYjmlDdzjbpTRyCx+goBCJQ=
github.com/xdg-go/pbkdf2 v1.0.0 h1:Su7DPu48wXMwC3bs7MCNG+z4FhcyEuz5dlvchbq0B0c=
github.com/xdg-go/pbkdf2 v1.0.0/go.mod h1:jrpuAogTd400dnrH08LKmI/xc1MbPOebTwRqcT5RDeI=
github.com/xdg-go/scram v1.1.2 h1:FHX5I5B4i4hKRVRBCFRxq1iQRej7WO3hhBuJf+UUySY=
github.com/xdg-go/scram v1.1.2/go.mod h1:RT/sEzTbU5y00aCK8UOx6R7YryM0iF1N2MOmC3kKLN4=
github.com/xdg-go/stringprep v1.0.4 h1:XLI/Ng3O1Atzq0oBs3TWm+5ZVgkq2aqdlvP9JtoZ6c8=
github.com/xdg-go/stringprep v1.0.4/go.mod h1:mPGuuIYwz7CmR2bT9j4GbQqutWS1zV24gijq1dTyGkM=
github.com/xuri/efp v0.0.1 h1:fws5Rv3myXyYni8uwj2qKjVaRP30PdjeYe2Y6FDsCL8=
github.com/xuri/efp v0.0.1/go.mod h1:ybY/Jr0T0GTCnYjKqmdwxyxn2BQf2RcQIIvex5QldPI=
github.com/xuri/excelize/v2 v2.11.0 h1:HxaEFl6sRN2+8J5a8HaKq+0M4FsjBGMnWWtjOCPSG88=
github.com/xuri/excelize/v2 v2.11.0/go.mod h1:jxFLbzaIwGQ5ufFNvYfUOHqXhfPaNmP14KWfmNz2Uak=
github.com/xuri/nfp v0.0.2-0.20250530014748-2ddeb826f9a9 h1:+C0TIdyyYmzadGaL/HBLbf3WdLgC29pgyhTjAT/0nuE=
github.com/xuri/nfp v0.0.2-0.20250530014748-2ddeb826f9a9/go.mod h1:WwHg+CVyzlv/TX9xqBFXEZAuxOPxn2k1GNHwG41IIUQ=
github.com/yuin/gopher-lua v1.1.1 h1:kYKnWBjvbNP4XLT3+bPEwAXJx262OhaHDWDVOPjL46M=
github.com/yuin/gopher-lua v1.1.1/go.mod h1:GBR0iDaNXjAgGg9zfCvksxSRnQx76gclCIb7kdAd1Pw=
golang.org/x/net v0.44.0 h1:evd8IRDyfNBMBTTY5XRF1vaZlD+EmWx6x8PkhR04H/I=
golang.org/x/net v0.44.0/go.mod h1:ECOoLqd5U3Lhyeyo/QDCEVQ4sNgYsqvCZ722XogGieY=
golang.org/x/sync v0.17.0 h1:l60nONMj9l5drqw6jlhIELNv9I0A4OFgRsG9k2oT9Ug=
golang.org/x/sync v0.17.0/go.mod h1:9KTHXmSnoGruLpwFjVSX0lNNA75CykiMECbovNTZqGI=
golang.org/x/text v0.29.0 h1:1neNs90w9YzJ9BocxfsQNHKuAT4pkghyXc4nhZ6sJvk=
golang.org/x/text v0.29.0/go.mod h1:7MhJOA9CD2qZyOKYazxdYMF85OwPdEr9jTtBpO7ydH4=
golang.org/x/crypto v0.53.0 h1:QZ4Muo8THX6CizN2vPPd5fBGHyogrdK9fG4wLPFUsto=
golang.org/x/crypto v0.53.0/go.mod h1:DNLU434OwVakk9PzuwV8w62mAJpRJL3vsgcfp4Qnsio=
golang.org/x/image v0.38.0 h1:5l+q+Y9JDC7mBOMjo4/aPhMDcxEptsX+Tt3GgRQRPuE=
golang.org/x/image v0.38.0/go.mod h1:/3f6vaXC+6CEanU4KJxbcUZyEePbyKbaLoDOe4ehFYY=
golang.org/x/net v0.56.0 h1:Rw8j/hFzGvJUZwNBXnAtf5sVDVt+65SK2C7IxCxZt5o=
golang.org/x/net v0.56.0/go.mod h1:D3Ku6r+V6JROoZK144D2XfMHFcMq/0zSfLelVTCFKec=
golang.org/x/sync v0.21.0 h1:HLII4xRRTtCRkxYp4HNFF0Js/Og6q2i++KXbg0gHCwM=
golang.org/x/sync v0.21.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
golang.org/x/sys v0.46.0 h1:noSf2Fq6F8DBgS+LysIkx7rIExoNHJsxOAtPp4rthXw=
golang.org/x/sys v0.46.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
golang.org/x/text v0.38.0 h1:sXmwo9DwP3OK9EZ7PqAdaooSGozfl/3a6/xJcbzPRhE=
golang.org/x/text v0.38.0/go.mod h1:YXZt3QhHUKYT53r2lLKFIVi6Ao1jdzrTR/KQ09qyxF4=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=

View File

@@ -0,0 +1,215 @@
package authentication
import (
"crypto/subtle"
"fmt"
"strings"
"lingniu-vehicle-ingest/go/vehicle-gateway/internal/envelope"
)
type Mode string
const (
ModeDisabled Mode = "disabled"
ModeObserve Mode = "observe"
ModeEnforce Mode = "enforce"
)
const (
StatusAccepted = "accepted"
StatusRejected = "rejected"
StatusUnknownAccount = "unknown_account"
StatusMissingCredential = "missing_credential"
StatusUnconfigured = "unconfigured"
)
type Result struct {
Applicable bool
Allowed bool
Mode Mode
Status string
Source string
}
type Authenticator interface {
Authenticate(envelope.FrameEnvelope) Result
}
func ParseMode(value string, fallback Mode) (Mode, error) {
value = strings.ToLower(strings.TrimSpace(value))
if value == "" {
value = string(fallback)
}
switch Mode(value) {
case ModeDisabled, ModeObserve, ModeEnforce:
return Mode(value), nil
default:
return "", fmt.Errorf("unsupported authentication mode %q", value)
}
}
type GB32960PlatformAuthenticator struct {
mode Mode
credentials map[string][]string
}
func NewGB32960PlatformAuthenticator(mode Mode, credentials map[string][]string) *GB32960PlatformAuthenticator {
normalized := make(map[string][]string, len(credentials))
for username, passwords := range credentials {
username = strings.TrimSpace(username)
if username == "" {
continue
}
for _, password := range passwords {
if password == "" {
continue
}
normalized[username] = append(normalized[username], password)
}
}
return &GB32960PlatformAuthenticator{mode: mode, credentials: normalized}
}
func (a *GB32960PlatformAuthenticator) Authenticate(env envelope.FrameEnvelope) Result {
if env.Protocol != envelope.ProtocolGB32960 || env.MessageID != "0x05" || a == nil || a.mode == ModeDisabled {
return Result{}
}
login := nestedMap(env.Parsed, "platform_login")
username := strings.TrimSpace(textValue(login, "username"))
password := textValue(login, "password")
status := StatusRejected
switch {
case len(a.credentials) == 0:
status = StatusUnconfigured
case username == "" || password == "":
status = StatusMissingCredential
default:
expected, ok := a.credentials[username]
if !ok || len(expected) == 0 {
status = StatusUnknownAccount
} else if anyConstantTimeEqual(expected, password) {
status = StatusAccepted
}
}
source := "none"
if status == StatusAccepted {
source = "configured"
}
return resultForMode(a.mode, status, source)
}
type JT808Authenticator struct {
mode Mode
authCode string
deviceTokens JT808DeviceTokenProvider
}
type JT808DeviceTokenProvider interface {
JT808AuthToken(phone string) (string, bool)
}
func NewJT808Authenticator(mode Mode, authCode string, deviceTokens JT808DeviceTokenProvider) *JT808Authenticator {
return &JT808Authenticator{mode: mode, authCode: authCode, deviceTokens: deviceTokens}
}
func (a *JT808Authenticator) Authenticate(env envelope.FrameEnvelope) Result {
if env.Protocol != envelope.ProtocolJT808 || env.MessageID != "0x0102" || a == nil || a.mode == ModeDisabled {
return Result{}
}
token := textValue(nestedMap(env.Parsed, "authentication"), "token")
status := StatusRejected
source := "none"
switch {
case token == "":
status = StatusMissingCredential
default:
if a.authCode != "" && constantTimeEqual(a.authCode, token) {
status = StatusAccepted
source = "configured"
break
}
deviceToken, knownDevice := "", false
if a.deviceTokens != nil {
deviceToken, knownDevice = a.deviceTokens.JT808AuthToken(env.Phone)
}
if knownDevice && deviceToken != "" && constantTimeEqual(deviceToken, token) {
status = StatusAccepted
source = "device"
} else if a.authCode == "" && !knownDevice {
status = StatusUnconfigured
}
}
return resultForMode(a.mode, status, source)
}
func Apply(env *envelope.FrameEnvelope, result Result) {
if env == nil || !result.Applicable {
return
}
env.AuthenticationMode = string(result.Mode)
env.AuthenticationStatus = result.Status
env.AuthenticationEnforced = result.Mode == ModeEnforce
}
// RedactParsedCredentials removes convenience copies of secrets before parsed
// fields are flattened and published. The original protocol frame remains in
// raw_hex for restricted forensic access.
func RedactParsedCredentials(env *envelope.FrameEnvelope) {
if env == nil || env.Protocol != envelope.ProtocolGB32960 {
return
}
login := nestedMap(env.Parsed, "platform_login")
if login == nil {
return
}
if password := textValue(login, "password"); password != "" {
login["password_present"] = true
}
delete(login, "password")
}
func resultForMode(mode Mode, status string, source ...string) Result {
allowed := status == StatusAccepted || mode != ModeEnforce
credentialSource := "none"
if len(source) > 0 && strings.TrimSpace(source[0]) != "" {
credentialSource = strings.TrimSpace(source[0])
}
return Result{Applicable: true, Allowed: allowed, Mode: mode, Status: status, Source: credentialSource}
}
func constantTimeEqual(expected string, actual string) bool {
if len(expected) != len(actual) {
return false
}
return subtle.ConstantTimeCompare([]byte(expected), []byte(actual)) == 1
}
func anyConstantTimeEqual(expected []string, actual string) bool {
matched := 0
for _, candidate := range expected {
if len(candidate) == len(actual) {
matched |= subtle.ConstantTimeCompare([]byte(candidate), []byte(actual))
}
}
return matched == 1
}
func nestedMap(parent map[string]any, key string) map[string]any {
if parent == nil {
return nil
}
value, _ := parent[key].(map[string]any)
return value
}
func textValue(values map[string]any, key string) string {
if values == nil {
return ""
}
value, ok := values[key]
if !ok || value == nil {
return ""
}
return strings.TrimSpace(fmt.Sprint(value))
}

Some files were not shown because too many files have changed in this diff Show More