fix(pipeline): 修复取报-处理-回填链路的超时、吞吐、回填与 FIFO 问题
外部调用(ACM2-33) - 共享 MySQL 与自有 PG 全部补有界超时:驱动 connect/socket 超时 + Hikari 池超时 (注意 Micronaut 的 Hikari 项是毫秒数,不是 Duration 字面量) - 删除主泵/毒丸路径的 inline 回填:跨库写不再占用 FIFO 关键路径,回填统一由扫描驱动 - 游标自愈:save() 改为「先 UPDATE、缺行 INSERT」,缺行只记一次 ERROR - Pump/Dispatcher 外层 catch 补 error 日志与失败计数 投递吞吐(ACM2-34) - Dispatcher 改批量领取;队头退避未到期或首条发送失败即停止本轮(保持目标内保序) - 有活不再 sleep;删除不可达的 state==SENT 死条件;markSent 移入分支; superseded 清理加轮数上限与空 eventId 保护 回填闭环(ACM2-36,V4) - MISSING(运行时确认行不存在)立即放弃自动重试并告警;暂时性故障达上限后停止自动重试 - 新增 reopen 人工恢复入口;放弃 ≠ 标记已确认(BACKFILL_AT 仍为空,清除前提不成立) - 扫描改 (BACKFILL_ATTEMPTS, MSG_ID) 公平轮转并排除放弃行,消除全局回填饥饿 - V4 只加字段与必要索引,不按年龄做任何存量推断 切流播种(ACM2-35,V5) - cutover-watermark 四模式(min/zero/max/显式 ID),默认不播种、代码不做默认选择 - 升级实例拒绝重新播种(SEEDED_AT 为 NULL ≠ 从未消费);播种与水位同语句落库 - 非法取值由启动自检挡下 错误分类与入口契约(ACM2-37) - 未知 SCHD 子类型改为 UNSUPPORTED,不再静默当全量日计划合并 - ADFT 运营日冲突改走 ProtocolViolation → DEAD(PROTOCOL) - 兼容入口 receivedAt 缺失回退到注入 Clock;MessageLifecycleGate 强制注入 + 装配断言 - FIFO:主泵只领取 msgId ≤ W,兼容入口登记的行在水位追平前不被领取 时间源与可观测(ACM2-38 / ACM2-41 阶段 0) - 仓储/处理器/作业全部经注入 Clock;移除 markTerminal/markBackfilled 的 Instant.now() 默认值 - 退避表档位与 max-attempts 对齐并加启动自检 - Micrometer 7 个 gauge(@Context 急切注册)+ /health 与 /metrics 共用积压快照缓存 - 只读迟到检测:监视被放行的空洞 ID 是否后来真的出现,只计数告警、不补入队 Plane: ACM2-33 ACM2-34 ACM2-35 ACM2-36 ACM2-37 ACM2-38 ACM2-41 Tests: 88 → 120(1 skipped 需真实 PG)
This commit is contained in:
@@ -47,6 +47,8 @@ dependencies {
|
|||||||
implementation(libs.micronaut.redis.lettuce)
|
implementation(libs.micronaut.redis.lettuce)
|
||||||
implementation(libs.micronaut.kafka)
|
implementation(libs.micronaut.kafka)
|
||||||
implementation(libs.micronaut.discovery.eureka)
|
implementation(libs.micronaut.discovery.eureka)
|
||||||
|
// 指标:把管道积压/回填/水位等暴露为 MeterRegistry 指标(版本由 platform BOM 管)
|
||||||
|
implementation(libs.micronaut.micrometer.core)
|
||||||
implementation(libs.jackson.dataformat.xml)
|
implementation(libs.jackson.dataformat.xml)
|
||||||
implementation(libs.jackson.module.kotlin)
|
implementation(libs.jackson.module.kotlin)
|
||||||
implementation(libs.logstash.logback.encoder)
|
implementation(libs.logstash.logback.encoder)
|
||||||
|
|||||||
@@ -33,6 +33,8 @@ micronaut-redis-lettuce = { module = "io.micronaut.redis:micronaut-redis-lettuce
|
|||||||
micronaut-kafka = { module = "io.micronaut.kafka:micronaut-kafka" }
|
micronaut-kafka = { module = "io.micronaut.kafka:micronaut-kafka" }
|
||||||
micronaut-discovery-eureka = { module = "io.micronaut.discovery:micronaut-discovery-client" }
|
micronaut-discovery-eureka = { module = "io.micronaut.discovery:micronaut-discovery-client" }
|
||||||
micronaut-management = { module = "io.micronaut:micronaut-management" }
|
micronaut-management = { module = "io.micronaut:micronaut-management" }
|
||||||
|
# 指标接入(Micrometer):版本由 micronaut-platform BOM 统一管理,勿在此钉版本。
|
||||||
|
micronaut-micrometer-core = { module = "io.micronaut.micrometer:micronaut-micrometer-core" }
|
||||||
# Kotlin 注解处理(KSP)处理器——与 kotlin-ksp 插件配套(U01)
|
# Kotlin 注解处理(KSP)处理器——与 kotlin-ksp 插件配套(U01)
|
||||||
micronaut-inject-kotlin = { module = "io.micronaut:micronaut-inject-kotlin", version.ref = "micronaut" }
|
micronaut-inject-kotlin = { module = "io.micronaut:micronaut-inject-kotlin", version.ref = "micronaut" }
|
||||||
jackson-dataformat-xml = { module = "com.fasterxml.jackson.dataformat:jackson-dataformat-xml", version = "2.18.2" }
|
jackson-dataformat-xml = { module = "com.fasterxml.jackson.dataformat:jackson-dataformat-xml", version = "2.18.2" }
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
package com.gzzn.omms.msgexchange
|
package com.gzzn.omms.msgexchange
|
||||||
|
|
||||||
|
import com.gzzn.omms.msgexchange.config.PipelineProps
|
||||||
import com.gzzn.omms.msgexchange.delivery.Dispatcher
|
import com.gzzn.omms.msgexchange.delivery.Dispatcher
|
||||||
import com.gzzn.omms.msgexchange.ingress.InboxPoller
|
import com.gzzn.omms.msgexchange.ingress.InboxPoller
|
||||||
import com.gzzn.omms.msgexchange.processing.Pump
|
import com.gzzn.omms.msgexchange.processing.Pump
|
||||||
@@ -24,6 +25,7 @@ class PipelineLifecycle(
|
|||||||
private val pump: Pump,
|
private val pump: Pump,
|
||||||
private val dispatcher: Dispatcher,
|
private val dispatcher: Dispatcher,
|
||||||
private val jobRunner: JobRunner,
|
private val jobRunner: JobRunner,
|
||||||
|
private val props: PipelineProps,
|
||||||
) {
|
) {
|
||||||
private val log = org.slf4j.LoggerFactory.getLogger(PipelineLifecycle::class.java)
|
private val log = org.slf4j.LoggerFactory.getLogger(PipelineLifecycle::class.java)
|
||||||
private val threads = mutableListOf<Thread>()
|
private val threads = mutableListOf<Thread>()
|
||||||
@@ -38,6 +40,9 @@ class PipelineLifecycle(
|
|||||||
|
|
||||||
private fun startIfNeeded() {
|
private fun startIfNeeded() {
|
||||||
if (started) return
|
if (started) return
|
||||||
|
// 自检放在起线程之前:配置错位(例如退避表档位数与 max-attempts 不匹配、
|
||||||
|
// 有一段退避永远走不到)必须让启动失败,而不是静默按错误参数长期运行。
|
||||||
|
props.pipeline.validate()
|
||||||
started = true
|
started = true
|
||||||
threads += spawn("msgx-inbox-poller", poller::loop)
|
threads += spawn("msgx-inbox-poller", poller::loop)
|
||||||
threads += spawn("msgx-pump", pump::loop)
|
threads += spawn("msgx-pump", pump::loop)
|
||||||
|
|||||||
@@ -73,14 +73,18 @@ class JacksonXmlCodec : XmlCodec {
|
|||||||
override fun encodeRqrd(kind: String, rangeJson: String): String =
|
override fun encodeRqrd(kind: String, rangeJson: String): String =
|
||||||
"""<?xml version="1.0" encoding="UTF-8"?><MSG><META><TYPE>RQFD</TYPE></META></MSG>"""
|
"""<?xml version="1.0" encoding="UTF-8"?><MSG><META><TYPE>RQFD</TYPE></META></MSG>"""
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 子类型分派必须是**白名单**:未知的 SCHD 子类型不能静默当成全量日计划(DNLD)——
|
||||||
|
* 那会让一条本该"暂不支持、等人工处置"的报文走最重的整包合并写入路径。
|
||||||
|
* 空 STYP 按 legacy 约定视为 DNLD(该约定待与 SIS 逐类对拍确认,见 Q8)。
|
||||||
|
*/
|
||||||
private fun kindOf(type: String, styp: String): MsgKind = when (type) {
|
private fun kindOf(type: String, styp: String): MsgKind = when (type) {
|
||||||
"SCHD" -> MsgKind.Schd(
|
"SCHD" -> when (styp) {
|
||||||
when (styp) {
|
"RESP" -> MsgKind.Schd(MsgKind.SchdSubtype.RESP)
|
||||||
"RESP" -> MsgKind.SchdSubtype.RESP
|
"ADFT" -> MsgKind.Schd(MsgKind.SchdSubtype.ADFT)
|
||||||
"ADFT" -> MsgKind.SchdSubtype.ADFT
|
"DNLD", "" -> MsgKind.Schd(MsgKind.SchdSubtype.DNLD)
|
||||||
else -> MsgKind.SchdSubtype.DNLD
|
else -> MsgKind.Unsupported("SCHD-$styp")
|
||||||
},
|
}
|
||||||
)
|
|
||||||
"FLOP" -> if (styp == "FDEL") MsgKind.Fdel else MsgKind.Flop(styp)
|
"FLOP" -> if (styp == "FDEL") MsgKind.Fdel else MsgKind.Flop(styp)
|
||||||
else -> MsgKind.Unsupported("$type-$styp")
|
else -> MsgKind.Unsupported("$type-$styp")
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -20,5 +20,23 @@ class MailboxProps {
|
|||||||
var username: String = ""
|
var username: String = ""
|
||||||
var password: String = ""
|
var password: String = ""
|
||||||
var driverClassName: String = "com.mysql.cj.jdbc.Driver"
|
var driverClassName: String = "com.mysql.cj.jdbc.Driver"
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 共享 MySQL 的驱动级超时(毫秒)。这些值必须存在且有限:Connector/J 默认
|
||||||
|
* `socketTimeout=0` 表示**无限等待**,一次网络黑洞就能把调用线程永久挂住。
|
||||||
|
*
|
||||||
|
* 收报轮询、主泵读原文、回填写信箱都会经过这条连接,因此超时是"外部调用有界"
|
||||||
|
* 这条约束的落点,不是可选调优项。
|
||||||
|
*/
|
||||||
|
var connectTimeoutMs: Int = 3_000
|
||||||
|
|
||||||
|
/** 单次 socket 读等待上限;超时抛异常,交由各自的失败/退避路径处理。 */
|
||||||
|
var socketTimeoutMs: Int = 30_000
|
||||||
|
|
||||||
|
/** Hikari 取连接(池耗尽)等待上限。 */
|
||||||
|
var poolConnectionTimeoutMs: Long = 5_000
|
||||||
|
|
||||||
|
/** Hikari 连接有效性校验上限。 */
|
||||||
|
var poolValidationTimeoutMs: Long = 3_000
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -22,7 +22,9 @@ class PipelineProps {
|
|||||||
var pollInterval: Duration = Duration.ofSeconds(1) // KEEP 现役节奏
|
var pollInterval: Duration = Duration.ofSeconds(1) // KEEP 现役节奏
|
||||||
var claimBatch: Int = 50
|
var claimBatch: Int = 50
|
||||||
var maxAttempts: Int = 5 // 处理/投递同值
|
var maxAttempts: Int = 5 // 处理/投递同值
|
||||||
var backoffMs: List<Long> = listOf(1000, 2000, 4000, 8000, 16000)
|
// 档位数必须等于 max-attempts − 1:attempts 达到上限即转 DEAD,不再计算下次重试,
|
||||||
|
// 因此最多只用得到 max-attempts − 1 个档位。多出来的档位永远走不到(会被 validate() 拦下)。
|
||||||
|
var backoffMs: List<Long> = listOf(1000, 2000, 4000, 8000)
|
||||||
var backoffCapMs: Long = 60_000
|
var backoffCapMs: Long = 60_000
|
||||||
var headDeadline: Duration = Duration.ofMinutes(10) // 最坏 HOL 上界(毒丸升级)
|
var headDeadline: Duration = Duration.ofMinutes(10) // 最坏 HOL 上界(毒丸升级)
|
||||||
|
|
||||||
@@ -47,6 +49,45 @@ class PipelineProps {
|
|||||||
/** 每次回填扫描最多处理多少条。 */
|
/** 每次回填扫描最多处理多少条。 */
|
||||||
var backfillBatch: Int = 100
|
var backfillBatch: Int = 100
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 回填的自动重试次数上限。达到上限后**停止自动重试**(置为 abandoned 并告警),
|
||||||
|
* 但保留人工恢复能力——暂时性故障不该变成永久失去补偿,也不该永久占满扫描批次。
|
||||||
|
*/
|
||||||
|
var backfillMaxAttempts: Int = 100
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 切流水位播种(一次性、显式)。取值:
|
||||||
|
* `min` = `W:MIN(ID)−1`(读当前全部现存行)、`zero` = `W:0`(按空洞规则从 0 扫)、
|
||||||
|
* `max` = `W:MAX(ID)`(跳过当前可见存量)、或一个具体 ID。
|
||||||
|
*
|
||||||
|
* **默认 null = 不播种**,保持既有行为。是否跳过存量属于切流决策,必须由人显式配置:
|
||||||
|
* 代码不做默认选择,也不会自动退化成 `max`;升级实例(已有水位或已有处理记录)会拒绝重新播种。
|
||||||
|
*/
|
||||||
|
var cutoverWatermark: String? = null
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 迟到到达检测(只读,ACM2-41 阶段 0):复查"已判定为永久空洞的 ID"是否后来真的出现。
|
||||||
|
* 默认 60 秒一轮;设为 0 或负数即关闭。
|
||||||
|
*
|
||||||
|
* 检测只计数与告警,**不入队、不改变任何处理语义**——自动补入队(阶段 1)需要先与库方定案。
|
||||||
|
*/
|
||||||
|
var lateDetectPeriod: Duration = Duration.ofSeconds(60)
|
||||||
|
|
||||||
|
/** 每轮最多复查多少个被放行的空洞 ID。 */
|
||||||
|
var lateDetectBatch: Int = 200
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 普通事件(`KAFKA:msg`)每轮向一个目标领取的条数上限。
|
||||||
|
* 逐条领取会让投递吞吐被"每条一次 DB 往返 + 一轮一次 sleep"压到每秒 1 条。
|
||||||
|
*/
|
||||||
|
var deliveryBatch: Int = 200
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 每轮最多连续领取多少批,之后让出一次循环去跑 `schd` flush,
|
||||||
|
* 避免长积压把状态通知饿死。
|
||||||
|
*/
|
||||||
|
var deliveryDrainRounds: Int = 10
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 服务启动后是否自动拉起收报、主泵、投递三个循环。
|
* 服务启动后是否自动拉起收报、主泵、投递三个循环。
|
||||||
* 默认关闭:只有接了真实仓储、或者明确用内存 stub 跑的时候才安全。
|
* 默认关闭:只有接了真实仓储、或者明确用内存 stub 跑的时候才安全。
|
||||||
@@ -58,6 +99,31 @@ class PipelineProps {
|
|||||||
val index = (attempt - 1).coerceAtLeast(0)
|
val index = (attempt - 1).coerceAtLeast(0)
|
||||||
return backoffMs.getOrNull(index)?.coerceAtMost(backoffCapMs) ?: backoffCapMs
|
return backoffMs.getOrNull(index)?.coerceAtMost(backoffCapMs) ?: backoffCapMs
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 配置自检:退避表档位数必须等于 `max-attempts − 1`。
|
||||||
|
*
|
||||||
|
* 因为 attempts 一达到 `max-attempts` 就转 `DEAD`、不再计算下次重试,能真正用到的档位
|
||||||
|
* 只有 `max-attempts − 1` 个。表更长会有一段**永远走不到**(默认配置里的 16 秒档就是这样),
|
||||||
|
* 表更短则会提前封顶。两种错位都应该在启动时暴露,而不是静默生效。
|
||||||
|
*/
|
||||||
|
fun validate() {
|
||||||
|
require(backoffMs.size == maxAttempts - 1) {
|
||||||
|
"msgx.pipeline.backoff-ms has ${backoffMs.size} slots, " +
|
||||||
|
"but max-attempts=$maxAttempts implies exactly ${maxAttempts - 1}"
|
||||||
|
}
|
||||||
|
// 切流播种只接受四种取值;非法值必须在启动时挡掉,而不是每轮轮询刷错误日志。
|
||||||
|
val cutover = cutoverWatermark
|
||||||
|
require(
|
||||||
|
cutover == null ||
|
||||||
|
cutover.equals("min", ignoreCase = true) ||
|
||||||
|
cutover.equals("zero", ignoreCase = true) ||
|
||||||
|
cutover.equals("max", ignoreCase = true) ||
|
||||||
|
cutover.toLongOrNull() != null,
|
||||||
|
) {
|
||||||
|
"msgx.pipeline.cutover-watermark must be one of min|zero|max|<id>, got '$cutover'"
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@ConfigurationProperties("schd")
|
@ConfigurationProperties("schd")
|
||||||
|
|||||||
@@ -11,6 +11,7 @@ import com.gzzn.omms.msgexchange.infra.retry.FailureScheduler
|
|||||||
import jakarta.inject.Singleton
|
import jakarta.inject.Singleton
|
||||||
import java.time.Duration
|
import java.time.Duration
|
||||||
import java.time.Instant
|
import java.time.Instant
|
||||||
|
import java.util.concurrent.atomic.AtomicLong
|
||||||
|
|
||||||
/** 对外投递的出口:同步等 Kafka 确认,语义是至少发一次(可能重复,但不会丢)。以后 ES 投影也从这里加。 */
|
/** 对外投递的出口:同步等 Kafka 确认,语义是至少发一次(可能重复,但不会丢)。以后 ES 投影也从这里加。 */
|
||||||
interface DeliveryPort {
|
interface DeliveryPort {
|
||||||
@@ -46,6 +47,9 @@ class Dispatcher(
|
|||||||
) {
|
) {
|
||||||
private val log = org.slf4j.LoggerFactory.getLogger(Dispatcher::class.java)
|
private val log = org.slf4j.LoggerFactory.getLogger(Dispatcher::class.java)
|
||||||
|
|
||||||
|
/** 连续失败计数:仅用于日志/排障。 */
|
||||||
|
private val tickFailures = AtomicLong(0)
|
||||||
|
|
||||||
@Volatile
|
@Volatile
|
||||||
private var running = true
|
private var running = true
|
||||||
|
|
||||||
@@ -60,35 +64,77 @@ class Dispatcher(
|
|||||||
while (running) {
|
while (running) {
|
||||||
try {
|
try {
|
||||||
tick()
|
tick()
|
||||||
|
tickFailures.set(0)
|
||||||
} catch (e: InterruptedException) {
|
} catch (e: InterruptedException) {
|
||||||
Thread.currentThread().interrupt()
|
Thread.currentThread().interrupt()
|
||||||
return
|
return
|
||||||
} catch (e: Exception) {
|
} catch (e: Exception) {
|
||||||
|
// tick 整体失败(DB 断连等)必须在别处留痕,否则只剩"投递不动"没有原因。
|
||||||
|
val failures = tickFailures.incrementAndGet()
|
||||||
|
log.error("dispatcher tick failed (failure #{})", failures, e)
|
||||||
sleepQuietly(props.pipeline.pollInterval)
|
sleepQuietly(props.pipeline.pollInterval)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
internal fun tick() {
|
internal fun tick() {
|
||||||
val head = msgEvents.headUnsent(Targets.KAFKA_MSG)
|
val sent = drainKafkaMsg()
|
||||||
if (head != null && (head.state == EventStatus.SENT || head.nextAttemptAt == null || head.nextAttemptAt <= scheduler.now())) {
|
|
||||||
deliver(Targets.KAFKA_MSG, head)
|
|
||||||
}
|
|
||||||
if (flushDue()) flushSchd()
|
if (flushDue()) flushSchd()
|
||||||
if (running) sleepQuietly(props.pipeline.pollInterval)
|
// 有活就不睡:pollInterval 只用于"本轮无事可做或队头在退避",不能当投递节流用。
|
||||||
|
if (sent == 0 && running) sleepQuietly(props.pipeline.pollInterval)
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 按 `EVENT_ID` 顺序批量投递 `KAFKA:msg`。
|
||||||
|
*
|
||||||
|
* 保序规则不变:**同一目标内不越序**——队头失败或退避未到期时立即停止本轮,
|
||||||
|
* 不跳过它去投后面的。批量只用来省掉"每条一次 DB 往返 + 一轮一次 sleep"。
|
||||||
|
*
|
||||||
|
* @return 本轮成功发出的条数
|
||||||
|
*/
|
||||||
|
private fun drainKafkaMsg(): Int {
|
||||||
|
val batchSize = props.pipeline.deliveryBatch.coerceAtLeast(1)
|
||||||
|
val maxRounds = props.pipeline.deliveryDrainRounds.coerceAtLeast(1)
|
||||||
|
var sent = 0
|
||||||
|
var round = 0
|
||||||
|
while (round < maxRounds) {
|
||||||
|
round++
|
||||||
|
val batch = try {
|
||||||
|
msgEvents.claimBatch(Targets.KAFKA_MSG, batchSize)
|
||||||
|
} catch (e: Exception) {
|
||||||
|
log.error("claimBatch(KAFKA_MSG) failed", e)
|
||||||
|
return sent
|
||||||
|
}
|
||||||
|
if (batch.isEmpty()) return sent
|
||||||
|
for (e in batch) {
|
||||||
|
val next = e.nextAttemptAt
|
||||||
|
if (next != null && next > scheduler.now()) return sent // 队头退避未到期:停止推进
|
||||||
|
if (!deliver(Targets.KAFKA_MSG, e)) return sent // 失败已记账:停止以保序
|
||||||
|
sent++
|
||||||
|
}
|
||||||
|
if (batch.size < batchSize) return sent
|
||||||
|
}
|
||||||
|
log.warn("kafka msg drain hit round cap ({} rounds, sent={}); yielding to schd flush", maxRounds, sent)
|
||||||
|
return sent
|
||||||
}
|
}
|
||||||
|
|
||||||
private fun flushDue(): Boolean =
|
private fun flushDue(): Boolean =
|
||||||
lastFlush?.let { Duration.between(it, scheduler.now()) >= props.schd.flushPeriod } ?: true
|
lastFlush?.let { Duration.between(it, scheduler.now()) >= props.schd.flushPeriod } ?: true
|
||||||
|
|
||||||
private fun deliver(target: String, e: MsgEvent) {
|
/** @return 是否已确认发出(false 表示已按退避/死信记账,调用方应停止本轮以保序) */
|
||||||
try {
|
private fun deliver(target: String, e: MsgEvent): Boolean {
|
||||||
|
val eventId = e.eventId ?: return false
|
||||||
|
return try {
|
||||||
when (target) {
|
when (target) {
|
||||||
Targets.KAFKA_MSG -> port.sendKafka("msg", e.partitionKey, e.payloadJson)
|
Targets.KAFKA_MSG -> port.sendKafka("msg", e.partitionKey, e.payloadJson)
|
||||||
|
// 未知目标不能在"未发送"的情况下被标记为已发;记账后停止本轮。
|
||||||
|
else -> error("unknown delivery target: $target")
|
||||||
}
|
}
|
||||||
msgEvents.markSent(e.eventId ?: return)
|
msgEvents.markSent(eventId)
|
||||||
|
true
|
||||||
} catch (ex: Exception) {
|
} catch (ex: Exception) {
|
||||||
retryOrDead(e, ex.message ?: ex.javaClass.simpleName)
|
retryOrDead(e, ex.message ?: ex.javaClass.simpleName)
|
||||||
|
false
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -117,14 +163,26 @@ class Dispatcher(
|
|||||||
}
|
}
|
||||||
val sentIds = batch.mapNotNull { it.eventId }.toSet() - failures.mapNotNull { it.eventId }.toSet()
|
val sentIds = batch.mapNotNull { it.eventId }.toSet() - failures.mapNotNull { it.eventId }.toSet()
|
||||||
if (sentIds.isNotEmpty()) msgEvents.markAllSent(sentIds.toList())
|
if (sentIds.isNotEmpty()) msgEvents.markAllSent(sentIds.toList())
|
||||||
// 被更新版本压掉的旧事件也要标成已发,否则它们会一直留在队里:同一个 FLID 只按最新版本输出一次
|
|
||||||
val sentVersions = batch.associate { it.partitionKey to it.stateVersion }
|
val sentVersions = batch.associate { it.partitionKey to it.stateVersion }
|
||||||
|
// 被更新版本压掉的旧事件也要标成已发,否则它们会一直留在队里:同一个 FLID 只按最新版本输出一次。
|
||||||
|
// 这里必须**有界**:若 markAllSent 因任何原因没有生效(例如行缺 eventId),
|
||||||
|
// 旧的无界 while(true) 会原地空转。
|
||||||
runCatching {
|
runCatching {
|
||||||
while (true) {
|
var rounds = 0
|
||||||
|
while (rounds < SUPERSEDED_CLEANUP_MAX_ROUNDS) {
|
||||||
|
rounds++
|
||||||
val superseded = msgEvents.mergePendingSchd(scheduler.now(), props.schd.flushLimit)
|
val superseded = msgEvents.mergePendingSchd(scheduler.now(), props.schd.flushLimit)
|
||||||
.filter { sentVersions[it.partitionKey]?.let { v -> it.stateVersion < v } == true }
|
.filter { sentVersions[it.partitionKey]?.let { v -> it.stateVersion < v } == true }
|
||||||
if (superseded.isEmpty()) break
|
if (superseded.isEmpty()) break
|
||||||
msgEvents.markAllSent(superseded.mapNotNull { it.eventId })
|
val ids = superseded.mapNotNull { it.eventId }
|
||||||
|
if (ids.isEmpty()) {
|
||||||
|
log.warn("superseded cleanup: {} rows without event_id, stop to avoid a spin", superseded.size)
|
||||||
|
break
|
||||||
|
}
|
||||||
|
msgEvents.markAllSent(ids)
|
||||||
|
if (rounds == SUPERSEDED_CLEANUP_MAX_ROUNDS) {
|
||||||
|
log.warn("superseded cleanup hit round cap ({}); remaining rows will be handled next flush", rounds)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}.onFailure { log.warn("superseded cleanup failed: {}", it.message) }
|
}.onFailure { log.warn("superseded cleanup failed: {}", it.message) }
|
||||||
failures.forEach { retryOrDead(it, it.lastError ?: "send-failed") }
|
failures.forEach { retryOrDead(it, it.lastError ?: "send-failed") }
|
||||||
@@ -145,4 +203,9 @@ class Dispatcher(
|
|||||||
private fun sleepQuietly(d: Duration) {
|
private fun sleepQuietly(d: Duration) {
|
||||||
if (!d.isNegative && !d.isZero) Thread.sleep(d.toMillis().coerceAtLeast(1))
|
if (!d.isNegative && !d.isZero) Thread.sleep(d.toMillis().coerceAtLeast(1))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private companion object {
|
||||||
|
/** superseded 清理的轮数上限:只用于防止"标不掉又不报错"时的原地空转。 */
|
||||||
|
const val SUPERSEDED_CLEANUP_MAX_ROUNDS = 100
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -72,6 +72,15 @@ data class ProcState(
|
|||||||
val backfillNextAt: Instant? = null,
|
val backfillNextAt: Instant? = null,
|
||||||
val backfillAttempts: Int = 0,
|
val backfillAttempts: Int = 0,
|
||||||
val backfillError: String? = null,
|
val backfillError: String? = null,
|
||||||
|
/**
|
||||||
|
* 非空表示已判定"不必再回填"(信箱行不存在,或达到尝试上限)。
|
||||||
|
*
|
||||||
|
* **它不等于标记已确认**:`backfillAt` 仍为空,所以不满足"边界内全部行已打标"的清除条件。
|
||||||
|
* 停止重试与"已满足清除前提"是两件事,不能互相替代。
|
||||||
|
*/
|
||||||
|
val backfillAbandonedAt: Instant? = null,
|
||||||
|
/** 放弃原因(`MISSING_ROW` / `MAX_ATTEMPTS`),供人工对账与恢复判断。 */
|
||||||
|
val backfillAbandonedReason: String? = null,
|
||||||
/** 首次被主泵取得的时刻;重试不刷新,用作 HOL deadline 的稳定起点。 */
|
/** 首次被主泵取得的时刻;重试不刷新,用作 HOL deadline 的稳定起点。 */
|
||||||
val processingStartedAt: Instant? = null,
|
val processingStartedAt: Instant? = null,
|
||||||
val updatedAt: Instant = Instant.now(),
|
val updatedAt: Instant = Instant.now(),
|
||||||
|
|||||||
@@ -0,0 +1,48 @@
|
|||||||
|
package com.gzzn.omms.msgexchange.infra.health
|
||||||
|
|
||||||
|
import com.gzzn.omms.msgexchange.infra.persistence.Backlog
|
||||||
|
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
||||||
|
import io.micronaut.context.BeanProvider
|
||||||
|
import io.micronaut.context.annotation.Value
|
||||||
|
import jakarta.inject.Singleton
|
||||||
|
import java.time.Clock
|
||||||
|
import java.time.Duration
|
||||||
|
import java.time.Instant
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 处理侧积压快照的唯一取数点,带 TTL 缓存。
|
||||||
|
*
|
||||||
|
* `backlog()` 是 `PROC_STATE` 的**全表聚合**;健康检查(`/health`)与指标抓取(`/metrics`)
|
||||||
|
* 都可能被高频调用,因此两者共用同一份缓存,避免把数据库拖慢。代价是数字最多滞后一个
|
||||||
|
* 缓存窗口——对积压/信龄这类告警信号足够。
|
||||||
|
*
|
||||||
|
* 缓存窗口由 `msgx.health.backlog-cache-ttl-ms` 配置(默认 30 秒;设为 0 表示不缓存)。
|
||||||
|
* 需要实时精确计数时应改为维护计数器,而不是把窗口调到 0。
|
||||||
|
*/
|
||||||
|
@Singleton
|
||||||
|
class BacklogSnapshotProvider(
|
||||||
|
private val procState: BeanProvider<ProcStateRepository>,
|
||||||
|
private val clock: Clock,
|
||||||
|
@Value("\${msgx.health.backlog-cache-ttl-ms:30000}") private val ttlMs: Long = 30_000,
|
||||||
|
) {
|
||||||
|
@Volatile
|
||||||
|
private var cached: Backlog? = null
|
||||||
|
|
||||||
|
@Volatile
|
||||||
|
private var cachedAt: Instant? = null
|
||||||
|
|
||||||
|
/** @return 积压快照;仓储未绑定(例如 stub 关闭)时返回 null,由调用方按"未绑定"处理。 */
|
||||||
|
fun snapshot(): Backlog? {
|
||||||
|
if (!procState.isPresent) return null
|
||||||
|
val now = clock.instant()
|
||||||
|
val at = cachedAt
|
||||||
|
val value = cached
|
||||||
|
if (ttlMs > 0 && at != null && value != null && Duration.between(at, now).toMillis() < ttlMs) {
|
||||||
|
return value
|
||||||
|
}
|
||||||
|
val fresh = procState.get().backlog()
|
||||||
|
cached = fresh
|
||||||
|
cachedAt = now
|
||||||
|
return fresh
|
||||||
|
}
|
||||||
|
}
|
||||||
+20
-5
@@ -1,5 +1,6 @@
|
|||||||
package com.gzzn.omms.msgexchange.infra.health
|
package com.gzzn.omms.msgexchange.infra.health
|
||||||
|
|
||||||
|
import com.gzzn.omms.msgexchange.infra.persistence.Backlog
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.CminmsgInboxRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.CminmsgInboxRepository
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.InboxCursorRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.InboxCursorRepository
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
||||||
@@ -10,6 +11,7 @@ import io.micronaut.management.health.indicator.HealthIndicator
|
|||||||
import io.micronaut.management.health.indicator.HealthResult
|
import io.micronaut.management.health.indicator.HealthResult
|
||||||
import jakarta.inject.Singleton
|
import jakarta.inject.Singleton
|
||||||
import org.reactivestreams.Publisher
|
import org.reactivestreams.Publisher
|
||||||
|
import java.time.Clock
|
||||||
import java.time.Duration
|
import java.time.Duration
|
||||||
import java.time.Instant
|
import java.time.Instant
|
||||||
|
|
||||||
@@ -28,6 +30,8 @@ class InboxLifecycleHealthIndicator(
|
|||||||
private val procState: BeanProvider<ProcStateRepository>,
|
private val procState: BeanProvider<ProcStateRepository>,
|
||||||
private val cursor: BeanProvider<InboxCursorRepository>,
|
private val cursor: BeanProvider<InboxCursorRepository>,
|
||||||
private val mailbox: BeanProvider<CminmsgInboxRepository>,
|
private val mailbox: BeanProvider<CminmsgInboxRepository>,
|
||||||
|
private val clock: Clock,
|
||||||
|
private val backlogs: BacklogSnapshotProvider,
|
||||||
) : HealthIndicator {
|
) : HealthIndicator {
|
||||||
|
|
||||||
override fun getResult(): Publisher<HealthResult> =
|
override fun getResult(): Publisher<HealthResult> =
|
||||||
@@ -37,6 +41,9 @@ class InboxLifecycleHealthIndicator(
|
|||||||
procState = if (procState.isPresent) procState.get() else null,
|
procState = if (procState.isPresent) procState.get() else null,
|
||||||
cursor = if (cursor.isPresent) cursor.get() else null,
|
cursor = if (cursor.isPresent) cursor.get() else null,
|
||||||
mailbox = if (mailbox.isPresent) mailbox.get() else null,
|
mailbox = if (mailbox.isPresent) mailbox.get() else null,
|
||||||
|
now = clock.instant(),
|
||||||
|
// 与 /metrics 共用同一份 30 秒缓存:两者都不该把全表聚合打成高频查询。
|
||||||
|
backlog = backlogs.snapshot(),
|
||||||
)
|
)
|
||||||
}.getOrElse { down(it) },
|
}.getOrElse { down(it) },
|
||||||
)
|
)
|
||||||
@@ -47,23 +54,31 @@ internal fun lifecycleHealth(
|
|||||||
procState: ProcStateRepository?,
|
procState: ProcStateRepository?,
|
||||||
cursor: InboxCursorRepository?,
|
cursor: InboxCursorRepository?,
|
||||||
mailbox: CminmsgInboxRepository?,
|
mailbox: CminmsgInboxRepository?,
|
||||||
now: Instant = Instant.now(),
|
now: Instant,
|
||||||
|
/** 允许调用方传入缓存/已算好的积压快照;为空时现查(`backlog()` 是全表聚合)。 */
|
||||||
|
backlog: Backlog? = null,
|
||||||
): HealthResult {
|
): HealthResult {
|
||||||
if (procState == null) {
|
if (procState == null) {
|
||||||
return HealthResult.builder(NAME).status(HealthStatus.UP)
|
return HealthResult.builder(NAME).status(HealthStatus.UP)
|
||||||
.details(mapOf("message" to "proc_state repository not bound (stub off, impl pending)"))
|
.details(mapOf("message" to "proc_state repository not bound (stub off, impl pending)"))
|
||||||
.build()
|
.build()
|
||||||
}
|
}
|
||||||
val backlog = procState.backlog()
|
val snapshot = backlog ?: procState.backlog()
|
||||||
val watermark = cursor?.load()?.committedUpTo
|
val watermark = cursor?.load()?.committedUpTo
|
||||||
val maxId = runCatching { mailbox?.maxId() }.getOrNull()
|
val maxId = runCatching { mailbox?.maxId() }.getOrNull()
|
||||||
return HealthResult.builder(NAME).status(HealthStatus.UP).details(
|
return HealthResult.builder(NAME).status(HealthStatus.UP).details(
|
||||||
linkedMapOf<String, Any>(
|
linkedMapOf<String, Any>(
|
||||||
"backlog" to backlog.unfinished,
|
"backlog" to snapshot.unfinished,
|
||||||
"oldestUnprocessedSeconds" to (
|
"oldestUnprocessedSeconds" to (
|
||||||
backlog.oldestReceivedAt?.let { Duration.between(it, now).seconds } ?: -1L
|
snapshot.oldestReceivedAt?.let { Duration.between(it, now).seconds } ?: -1L
|
||||||
|
),
|
||||||
|
"unmarkedTerminal" to snapshot.unmarkedTerminal,
|
||||||
|
// 已放弃自动回填的条数:**不等于**标记已完成,需要人工对账;非 0 应告警。
|
||||||
|
"backfillAbandoned" to snapshot.abandonedBackfill,
|
||||||
|
// 最老一条"仍待自动回填"记录的年龄(秒):回填延迟的真实观测值。
|
||||||
|
"oldestUnmarkedBackfillSeconds" to (
|
||||||
|
snapshot.oldestUnmarkedAt?.let { Duration.between(it, now).seconds } ?: -1L
|
||||||
),
|
),
|
||||||
"unmarkedTerminal" to backlog.unmarkedTerminal,
|
|
||||||
"watermark" to (watermark ?: -1L),
|
"watermark" to (watermark ?: -1L),
|
||||||
"watermarkLag" to if (watermark != null && maxId != null) maxId - watermark else -1L,
|
"watermarkLag" to if (watermark != null && maxId != null) maxId - watermark else -1L,
|
||||||
),
|
),
|
||||||
|
|||||||
@@ -0,0 +1,37 @@
|
|||||||
|
package com.gzzn.omms.msgexchange.infra.metrics
|
||||||
|
|
||||||
|
import jakarta.inject.Singleton
|
||||||
|
import java.util.concurrent.atomic.AtomicLong
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 管道运行期的**进程内**计数。
|
||||||
|
*
|
||||||
|
* 存在的意义是让业务模块(如收报)能零依赖地上报事实:模块只依赖本类,
|
||||||
|
* 不直接依赖 Micrometer;指标如何在 `/metrics` 上暴露由 [PipelineMetrics] 决定。
|
||||||
|
*
|
||||||
|
* 注意:计数在重启后归零。需要跨重启的累计值应由指标后端聚合,不在这里做持久化。
|
||||||
|
*/
|
||||||
|
@Singleton
|
||||||
|
class PipelineCounters {
|
||||||
|
private val holeAgedOut = AtomicLong(0)
|
||||||
|
private val lateArrivalDetected = AtomicLong(0)
|
||||||
|
|
||||||
|
/** 水位因空洞超过老化阈值而放行(判定为永久空洞)的次数。 */
|
||||||
|
fun holeAgedOutIncrement() {
|
||||||
|
holeAgedOut.incrementAndGet()
|
||||||
|
}
|
||||||
|
|
||||||
|
fun holeAgedOutCount(): Long = holeAgedOut.get()
|
||||||
|
|
||||||
|
/**
|
||||||
|
* "迟到到达"检测命中的**不同**消息 ID 数(ACM2-41 阶段 0)。
|
||||||
|
*
|
||||||
|
* 含义:该 ID 曾被判定为永久空洞并放行,之后却真的出现在信箱里——即上游提交晚于水位推进。
|
||||||
|
* 阶段 0 只计数与告警,**不会**补入队;因此这个值 > 0 表示"确有迟到发生,需要与库方对契约"。
|
||||||
|
*/
|
||||||
|
fun lateArrivalDetectedIncrement() {
|
||||||
|
lateArrivalDetected.incrementAndGet()
|
||||||
|
}
|
||||||
|
|
||||||
|
fun lateArrivalDetectedCount(): Long = lateArrivalDetected.get()
|
||||||
|
}
|
||||||
@@ -0,0 +1,84 @@
|
|||||||
|
package com.gzzn.omms.msgexchange.infra.metrics
|
||||||
|
|
||||||
|
import com.gzzn.omms.msgexchange.infra.health.BacklogSnapshotProvider
|
||||||
|
import com.gzzn.omms.msgexchange.infra.persistence.CminmsgInboxRepository
|
||||||
|
import com.gzzn.omms.msgexchange.infra.persistence.InboxCursorRepository
|
||||||
|
import io.micronaut.context.BeanProvider
|
||||||
|
import io.micronaut.context.annotation.Context
|
||||||
|
import io.micronaut.context.annotation.Requires
|
||||||
|
import io.micrometer.core.instrument.Gauge
|
||||||
|
import io.micrometer.core.instrument.MeterRegistry
|
||||||
|
import jakarta.annotation.PostConstruct
|
||||||
|
import java.time.Clock
|
||||||
|
import java.time.Duration
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 把管道的可观测事实暴露成 Micrometer 指标(默认经 `/metrics` 抓取)。
|
||||||
|
*
|
||||||
|
* 只暴露规范里要求观测的量:
|
||||||
|
* - `msgx.pipeline.backlog.unfinished`:还没处理完的消息条数
|
||||||
|
* - `msgx.pipeline.backlog.oldest_unprocessed_seconds`:最老未处理消息的信龄
|
||||||
|
* - `msgx.pipeline.backfill.unmarked_terminal`:已终态但未打标的条数
|
||||||
|
* - `msgx.pipeline.backfill.abandoned`:已放弃自动回填的条数(**非 0 需人工对账**)
|
||||||
|
* - `msgx.pipeline.backfill.oldest_unmarked_seconds`:最老一条仍待自动回填的年龄
|
||||||
|
* - `msgx.pipeline.watermark.lag`:水位落后信箱最新 ID 的距离
|
||||||
|
* - `msgx.pipeline.hole.aged_out.total`:永久空洞放行次数
|
||||||
|
*
|
||||||
|
* 取数统一走 [BacklogSnapshotProvider](30 秒 TTL),因此指标抓取不会打穿数据库。
|
||||||
|
* 无法取数时以 `NaN` 上报(Micrometer 的惯例表示"本次无值"),而不是伪造 0。
|
||||||
|
*
|
||||||
|
* `Gauge` 默认对目标对象持**弱引用**,这里显式 `strongReference(true)`:
|
||||||
|
* 单例被容器持有本不会被回收,但显式声明可避免将来重构踩坑。
|
||||||
|
*
|
||||||
|
* 用 `@Context` 而不是 `@Singleton`:没有任何 bean 依赖它,惰性单例永远不会被创建,
|
||||||
|
* 指标也就注册不上。指标装配必须在启动时急切完成。
|
||||||
|
*/
|
||||||
|
@Context
|
||||||
|
@Requires(beans = [MeterRegistry::class])
|
||||||
|
class PipelineMetrics(
|
||||||
|
private val registry: MeterRegistry,
|
||||||
|
private val backlogs: BacklogSnapshotProvider,
|
||||||
|
private val cursor: BeanProvider<InboxCursorRepository>,
|
||||||
|
private val mailbox: BeanProvider<CminmsgInboxRepository>,
|
||||||
|
private val counters: PipelineCounters,
|
||||||
|
private val clock: Clock,
|
||||||
|
) {
|
||||||
|
|
||||||
|
@PostConstruct
|
||||||
|
fun bind() {
|
||||||
|
backlogGauge("msgx.pipeline.backlog.unfinished") { it.unfinished.toDouble() }
|
||||||
|
backlogGauge("msgx.pipeline.backlog.oldest_unprocessed_seconds") { snapshot ->
|
||||||
|
snapshot.oldestReceivedAt
|
||||||
|
?.let { Duration.between(it, clock.instant()).seconds.toDouble() }
|
||||||
|
?: -1.0
|
||||||
|
}
|
||||||
|
backlogGauge("msgx.pipeline.backfill.unmarked_terminal") { it.unmarkedTerminal.toDouble() }
|
||||||
|
backlogGauge("msgx.pipeline.backfill.abandoned") { it.abandonedBackfill.toDouble() }
|
||||||
|
backlogGauge("msgx.pipeline.backfill.oldest_unmarked_seconds") { snapshot ->
|
||||||
|
snapshot.oldestUnmarkedAt
|
||||||
|
?.let { Duration.between(it, clock.instant()).seconds.toDouble() }
|
||||||
|
?: -1.0
|
||||||
|
}
|
||||||
|
|
||||||
|
Gauge.builder("msgx.pipeline.watermark.lag", backlogs) { _ ->
|
||||||
|
val watermark = if (cursor.isPresent) cursor.get().load().committedUpTo else null
|
||||||
|
val maxId = if (mailbox.isPresent) runCatching { mailbox.get().maxId() }.getOrNull() else null
|
||||||
|
if (watermark != null && maxId != null) (maxId - watermark).toDouble() else -1.0
|
||||||
|
}.strongReference(true).register(registry)
|
||||||
|
|
||||||
|
Gauge.builder("msgx.pipeline.hole.aged_out.total", counters) { it.holeAgedOutCount().toDouble() }
|
||||||
|
.strongReference(true)
|
||||||
|
.register(registry)
|
||||||
|
|
||||||
|
// 迟到到达检测命中数(阶段 0 只观测):> 0 表示上游提交确实晚于水位推进,需要与库方对契约。
|
||||||
|
Gauge.builder("msgx.pipeline.late_arrival.detected.total", counters) { it.lateArrivalDetectedCount().toDouble() }
|
||||||
|
.strongReference(true)
|
||||||
|
.register(registry)
|
||||||
|
}
|
||||||
|
|
||||||
|
private fun backlogGauge(name: String, value: (com.gzzn.omms.msgexchange.infra.persistence.Backlog) -> Double) {
|
||||||
|
Gauge.builder(name, backlogs) { provider ->
|
||||||
|
provider.snapshot()?.let(value) ?: Double.NaN
|
||||||
|
}.strongReference(true).register(registry)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -89,15 +89,31 @@ interface ProcStateRepository {
|
|||||||
errorClass: ErrorClass? = null,
|
errorClass: ErrorClass? = null,
|
||||||
lastError: String? = null,
|
lastError: String? = null,
|
||||||
attempts: Int? = null,
|
attempts: Int? = null,
|
||||||
now: Instant = Instant.now(),
|
now: Instant,
|
||||||
)
|
)
|
||||||
|
|
||||||
/** 回填成功:记下完成时间,清掉待办。 */
|
/** 回填成功:记下完成时间,清掉待办。 */
|
||||||
fun markBackfilled(msgId: Long, now: Instant = Instant.now())
|
fun markBackfilled(msgId: Long, now: Instant)
|
||||||
|
|
||||||
/** 回填失败:次数 +1、按退避推后、记下原因。处理终态不受影响,不会被改回去。 */
|
/** 回填失败:次数 +1、按退避推后、记下原因。处理终态不受影响,不会被改回去。 */
|
||||||
fun recordBackfillFailure(msgId: Long, error: String?, attempts: Int, nextAttemptAt: Instant, now: Instant)
|
fun recordBackfillFailure(msgId: Long, error: String?, attempts: Int, nextAttemptAt: Instant, now: Instant)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 放弃回填:判定该行不必再自动尝试(信箱行不存在,或达到尝试上限)。
|
||||||
|
*
|
||||||
|
* **abandoned ≠ 标记已确认**:`BACKFILL_AT` 仍为空,因此**不**满足"边界内全部行已打标"的
|
||||||
|
* 清除前提;放弃只是停止自动重试并把事实留痕,供人工对账。
|
||||||
|
*
|
||||||
|
* @return false 表示该行不存在
|
||||||
|
*/
|
||||||
|
fun markBackfillAbandoned(msgId: Long, reason: String, now: Instant): Boolean
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 人工恢复入口:清除放弃标记并重新排队一次回填(暂时性故障恢复后、或人工对账确认后使用)。
|
||||||
|
* @return false 表示该行不存在或本来就未被放弃
|
||||||
|
*/
|
||||||
|
fun reopenBackfill(msgId: Long, now: Instant): Boolean
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 找出现在该回填的记录:已经到终态、还没确认回填,并且退避时间已到。
|
* 找出现在该回填的记录:已经到终态、还没确认回填,并且退避时间已到。
|
||||||
*
|
*
|
||||||
@@ -111,6 +127,9 @@ interface ProcStateRepository {
|
|||||||
|
|
||||||
/** 积压观测:还没处理完的条数、最老一条的接收时间、处理完但还没回填的条数。 */
|
/** 积压观测:还没处理完的条数、最老一条的接收时间、处理完但还没回填的条数。 */
|
||||||
fun backlog(): Backlog
|
fun backlog(): Backlog
|
||||||
|
|
||||||
|
/** 是否已有任何处理记录(含终态)。切流播种用它判断"这个实例是否已经消费过"。 */
|
||||||
|
fun hasAny(): Boolean
|
||||||
}
|
}
|
||||||
|
|
||||||
/** 扫描到的待回填记录。 */
|
/** 扫描到的待回填记录。 */
|
||||||
@@ -120,9 +139,17 @@ data class BackfillDue(val msgId: Long, val attempts: Int)
|
|||||||
* 处理侧积压快照。
|
* 处理侧积压快照。
|
||||||
* @param unfinished 还没处理完的消息条数
|
* @param unfinished 还没处理完的消息条数
|
||||||
* @param oldestReceivedAt 其中最早一条的接收时间(据此算信龄)
|
* @param oldestReceivedAt 其中最早一条的接收时间(据此算信龄)
|
||||||
* @param unmarkedTerminal 已经处理完、但还没把标记写回信箱的条数
|
* @param unmarkedTerminal 已经处理完、但还没把标记写回信箱的条数(含已放弃的)
|
||||||
|
* @param abandonedBackfill 其中已放弃自动回填的条数(需要人工对账的信号)
|
||||||
|
* @param oldestUnmarkedAt 最老一条待回填记录的接收时间(回填延迟观测)
|
||||||
*/
|
*/
|
||||||
data class Backlog(val unfinished: Int, val oldestReceivedAt: Instant?, val unmarkedTerminal: Int)
|
data class Backlog(
|
||||||
|
val unfinished: Int,
|
||||||
|
val oldestReceivedAt: Instant?,
|
||||||
|
val unmarkedTerminal: Int,
|
||||||
|
val abandonedBackfill: Int = 0,
|
||||||
|
val oldestUnmarkedAt: Instant? = null,
|
||||||
|
)
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 待发事件(outbox)表:业务提交时把要发的事件一起写进来,投递线程再从这张表往外发,
|
* 待发事件(outbox)表:业务提交时把要发的事件一起写进来,投递线程再从这张表往外发,
|
||||||
@@ -263,12 +290,25 @@ interface InboxCursorRepository {
|
|||||||
/**
|
/**
|
||||||
* @param committedUpTo 水位 W
|
* @param committedUpTo 水位 W
|
||||||
* @param holeSince W 后面那个缺口最早被发现的时刻;当前没有缺口时为 null
|
* @param holeSince W 后面那个缺口最早被发现的时刻;当前没有缺口时为 null
|
||||||
|
* @param seededAt 非空 = 已按 `msgx.pipeline.cutover-watermark` 播种过。
|
||||||
|
* **它为空不等于"从未消费"**:已有库新增该列后同样是 null,判断必须叠加"水位为 0 且无处理记录"。
|
||||||
*/
|
*/
|
||||||
data class Cursor(val committedUpTo: Long = 0L, val holeSince: Instant? = null)
|
data class Cursor(
|
||||||
|
val committedUpTo: Long = 0L,
|
||||||
|
val holeSince: Instant? = null,
|
||||||
|
val seededAt: Instant? = null,
|
||||||
|
)
|
||||||
|
|
||||||
fun load(): Cursor
|
fun load(): Cursor
|
||||||
|
|
||||||
|
/** 只推进水位与空洞计时,不改动 [Cursor.seededAt]。 */
|
||||||
fun save(cursor: Cursor)
|
fun save(cursor: Cursor)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 切流播种:一次性写入水位并把空洞计时清空,同时记录播种事实。
|
||||||
|
* 与 [save] 分开,避免"普通轮次推进水位"把播种标记抹掉。
|
||||||
|
*/
|
||||||
|
fun markSeeded(committedUpTo: Long, now: Instant)
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -299,6 +339,15 @@ interface CminmsgInboxRepository {
|
|||||||
/** 信箱当前最大 ID,空表返回 null;只用来观测收报落后了多少。 */
|
/** 信箱当前最大 ID,空表返回 null;只用来观测收报落后了多少。 */
|
||||||
fun maxId(): Long?
|
fun maxId(): Long?
|
||||||
|
|
||||||
|
/** 信箱当前最小 ID,空表返回 null;切流播种用它推算 `W = MIN(ID) − 1`。 */
|
||||||
|
fun minId(): Long?
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 批量查询这些 ID 里哪些**当前存在于信箱**。
|
||||||
|
* 只用于"迟到到达"检测(被放行的空洞 ID 后来是否真的出现),不读大字段。
|
||||||
|
*/
|
||||||
|
fun existingIds(msgIds: Collection<Long>): Set<Long>
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 把处理标记写回信箱,并且**只写还是空标记的行**:库里已有值时不覆盖、不回退,
|
* 把处理标记写回信箱,并且**只写还是空标记的行**:库里已有值时不覆盖、不回退,
|
||||||
* 重复调用没有副作用。结果明确区分本次写入、已有标记和信箱行缺失。
|
* 重复调用没有副作用。结果明确区分本次写入、已有标记和信箱行缺失。
|
||||||
|
|||||||
+19
@@ -71,6 +71,25 @@ class JdbcCminmsgInboxRepository(
|
|||||||
rs.getLong("max_id").takeIf { !rs.wasNull() }
|
rs.getLong("max_id").takeIf { !rs.wasNull() }
|
||||||
}
|
}
|
||||||
|
|
||||||
|
override fun minId(): Long? =
|
||||||
|
ds.queryOne("SELECT MIN(CMINMSGS_ID) AS min_id FROM cminmsgs", {}) { rs ->
|
||||||
|
rs.getLong("min_id").takeIf { !rs.wasNull() }
|
||||||
|
}
|
||||||
|
|
||||||
|
override fun existingIds(msgIds: Collection<Long>): Set<Long> {
|
||||||
|
if (msgIds.isEmpty()) return emptySet()
|
||||||
|
val out = linkedSetOf<Long>()
|
||||||
|
// 分块避免 IN 列表过长(迟到检测一次最多查 late-detect-batch 个)。
|
||||||
|
msgIds.chunked(200).forEach { chunk ->
|
||||||
|
val placeholders = chunk.joinToString(",") { "?" }
|
||||||
|
ds.query(
|
||||||
|
"SELECT CMINMSGS_ID FROM cminmsgs WHERE CMINMSGS_ID IN ($placeholders)",
|
||||||
|
{ ps -> chunk.forEachIndexed { i, id -> ps.setLong(i + 1, id) } },
|
||||||
|
) { rs -> rs.getLong("CMINMSGS_ID") }.forEach { out += it }
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 只更新还是空标记的行,所以重复调用不会覆盖库里已有的值;
|
* 只更新还是空标记的行,所以重复调用不会覆盖库里已有的值;
|
||||||
* 影响 0 行时再查一次主键,区分“已有标记”和“信箱行缺失”;后者不能记为回填成功。
|
* 影响 0 行时再查一次主键,区分“已有标记”和“信箱行缺失”;后者不能记为回填成功。
|
||||||
|
|||||||
+133
-18
@@ -27,11 +27,13 @@ import com.gzzn.omms.msgexchange.infra.persistence.SnapshotLogRepository
|
|||||||
import io.micronaut.context.annotation.Requires
|
import io.micronaut.context.annotation.Requires
|
||||||
import jakarta.inject.Singleton
|
import jakarta.inject.Singleton
|
||||||
import java.sql.ResultSet
|
import java.sql.ResultSet
|
||||||
|
import java.time.Clock
|
||||||
import java.time.Duration
|
import java.time.Duration
|
||||||
import java.time.Instant
|
import java.time.Instant
|
||||||
import java.time.LocalDate
|
import java.time.LocalDate
|
||||||
import java.time.LocalDateTime
|
import java.time.LocalDateTime
|
||||||
import java.time.ZoneId
|
import java.time.ZoneId
|
||||||
|
import java.util.concurrent.atomic.AtomicBoolean
|
||||||
import com.gzzn.omms.msgexchange.domain.OperationDayCalculator
|
import com.gzzn.omms.msgexchange.domain.OperationDayCalculator
|
||||||
import java.time.format.DateTimeFormatter
|
import java.time.format.DateTimeFormatter
|
||||||
import java.util.Locale
|
import java.util.Locale
|
||||||
@@ -70,6 +72,7 @@ class JdbcPipelineLockRepository(
|
|||||||
@Requires(missingProperty = "msgx.stubs")
|
@Requires(missingProperty = "msgx.stubs")
|
||||||
class JdbcProcStateRepository(
|
class JdbcProcStateRepository(
|
||||||
private val ds: DataSource,
|
private val ds: DataSource,
|
||||||
|
private val clock: Clock,
|
||||||
) : ProcStateRepository {
|
) : ProcStateRepository {
|
||||||
/** 入队(幂等):主键冲突时什么都不做,所以重复扫描和兼容入口并发调用都安全。 */
|
/** 入队(幂等):主键冲突时什么都不做,所以重复扫描和兼容入口并发调用都安全。 */
|
||||||
override fun insertIfAbsent(msgId: Long, receivedAt: Instant?): Boolean =
|
override fun insertIfAbsent(msgId: Long, receivedAt: Instant?): Boolean =
|
||||||
@@ -79,7 +82,7 @@ class JdbcProcStateRepository(
|
|||||||
{ ps ->
|
{ ps ->
|
||||||
ps.setLong(1, msgId)
|
ps.setLong(1, msgId)
|
||||||
ps.setTimestamp(2, receivedAt?.toSqlTimestamp())
|
ps.setTimestamp(2, receivedAt?.toSqlTimestamp())
|
||||||
ps.setTimestamp(3, Instant.now().toSqlTimestamp())
|
ps.setTimestamp(3, clock.instant().toSqlTimestamp())
|
||||||
},
|
},
|
||||||
) == 1
|
) == 1
|
||||||
|
|
||||||
@@ -117,7 +120,7 @@ class JdbcProcStateRepository(
|
|||||||
"UPDATE proc_state SET identity_key = ?, updated_at = ? WHERE msg_id = ? AND identity_key IS NULL",
|
"UPDATE proc_state SET identity_key = ?, updated_at = ? WHERE msg_id = ? AND identity_key IS NULL",
|
||||||
{ ps ->
|
{ ps ->
|
||||||
ps.setString(1, identityKey)
|
ps.setString(1, identityKey)
|
||||||
ps.setTimestamp(2, Instant.now().toSqlTimestamp())
|
ps.setTimestamp(2, clock.instant().toSqlTimestamp())
|
||||||
ps.setLong(3, msgId)
|
ps.setLong(3, msgId)
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
@@ -149,7 +152,7 @@ class JdbcProcStateRepository(
|
|||||||
ps.setInt(3, attempts ?: 0)
|
ps.setInt(3, attempts ?: 0)
|
||||||
ps.setString(4, errorClass?.name)
|
ps.setString(4, errorClass?.name)
|
||||||
ps.setString(5, lastError)
|
ps.setString(5, lastError)
|
||||||
ps.setTimestamp(6, Instant.now().toSqlTimestamp())
|
ps.setTimestamp(6, clock.instant().toSqlTimestamp())
|
||||||
ps.setLong(7, msgId)
|
ps.setLong(7, msgId)
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
@@ -170,6 +173,7 @@ class JdbcProcStateRepository(
|
|||||||
SET state = ?, error_class = ?, last_error = ?, attempts = COALESCE(?, attempts),
|
SET state = ?, error_class = ?, last_error = ?, attempts = COALESCE(?, attempts),
|
||||||
next_attempt_at = NULL,
|
next_attempt_at = NULL,
|
||||||
backfill_at = NULL, backfill_next_at = ?, backfill_attempts = 0, backfill_error = NULL,
|
backfill_at = NULL, backfill_next_at = ?, backfill_attempts = 0, backfill_error = NULL,
|
||||||
|
backfill_abandoned_at = NULL, backfill_abandoned_reason = NULL,
|
||||||
updated_at = ?
|
updated_at = ?
|
||||||
WHERE msg_id = ?
|
WHERE msg_id = ?
|
||||||
""".trimIndent(),
|
""".trimIndent(),
|
||||||
@@ -209,18 +213,55 @@ class JdbcProcStateRepository(
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 放弃自动回填。注意 **不写 `BACKFILL_AT`**:放弃 ≠ 标记已确认,
|
||||||
|
* 因此"边界内全部行已打标"的清除前提仍然不成立,库方不应据此清除。
|
||||||
|
*/
|
||||||
|
override fun markBackfillAbandoned(msgId: Long, reason: String, now: Instant): Boolean =
|
||||||
|
ds.update(
|
||||||
|
"UPDATE proc_state SET backfill_abandoned_at = ?, backfill_abandoned_reason = ?, " +
|
||||||
|
"backfill_next_at = NULL, updated_at = ? " +
|
||||||
|
"WHERE msg_id = ? AND backfill_at IS NULL AND backfill_abandoned_at IS NULL",
|
||||||
|
{ ps ->
|
||||||
|
ps.setTimestamp(1, now.toSqlTimestamp())
|
||||||
|
ps.setString(2, reason.take(64))
|
||||||
|
ps.setTimestamp(3, now.toSqlTimestamp())
|
||||||
|
ps.setLong(4, msgId)
|
||||||
|
},
|
||||||
|
) == 1
|
||||||
|
|
||||||
|
/** 人工恢复:清掉放弃标记并立刻重排一次回填(暂时性故障恢复后或对账后使用)。 */
|
||||||
|
override fun reopenBackfill(msgId: Long, now: Instant): Boolean =
|
||||||
|
ds.update(
|
||||||
|
"UPDATE proc_state SET backfill_abandoned_at = NULL, backfill_abandoned_reason = NULL, " +
|
||||||
|
"backfill_next_at = ?, updated_at = ? " +
|
||||||
|
"WHERE msg_id = ? AND backfill_at IS NULL AND backfill_abandoned_at IS NOT NULL",
|
||||||
|
{ ps ->
|
||||||
|
ps.setTimestamp(1, now.toSqlTimestamp())
|
||||||
|
ps.setTimestamp(2, now.toSqlTimestamp())
|
||||||
|
ps.setLong(3, msgId)
|
||||||
|
},
|
||||||
|
) == 1
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 到了该回填的时候:终态 + 还没有标记 + (退避到期 或 收信时间已经很久)。
|
* 到了该回填的时候:终态 + 还没有标记 + (退避到期 或 收信时间已经很久)。
|
||||||
* 后面这个"很久"是兜底,保证标记最终一定会补上,库方才能按标记清理信箱。
|
* 后面这个"很久"是兜底,保证标记最终一定会补上,库方才能按标记清理信箱。
|
||||||
*/
|
*/
|
||||||
|
/**
|
||||||
|
* 到了该回填的时候:终态 + 还没有标记 + **未放弃** + (退避到期 或 收信时间已经很久)。
|
||||||
|
*
|
||||||
|
* 排序用**公平轮转**:先按已尝试次数升序,再按 msg_id。若只按 msg_id 升序,
|
||||||
|
* 最旧的一批永久失败行会持续占满批次,后面的记录永远轮不到(全局回填饥饿)。
|
||||||
|
*/
|
||||||
override fun findBackfillDue(now: Instant, overdueBefore: Instant, limit: Int): List<BackfillDue> =
|
override fun findBackfillDue(now: Instant, overdueBefore: Instant, limit: Int): List<BackfillDue> =
|
||||||
ds.query(
|
ds.query(
|
||||||
"""
|
"""
|
||||||
SELECT msg_id, backfill_attempts FROM proc_state
|
SELECT msg_id, backfill_attempts FROM proc_state
|
||||||
WHERE backfill_at IS NULL
|
WHERE backfill_at IS NULL
|
||||||
|
AND backfill_abandoned_at IS NULL
|
||||||
AND state IN ('SUCCEEDED', 'SKIPPED', 'DEAD')
|
AND state IN ('SUCCEEDED', 'SKIPPED', 'DEAD')
|
||||||
AND (backfill_next_at IS NULL OR backfill_next_at <= ? OR (received_at IS NOT NULL AND received_at < ?))
|
AND (backfill_next_at IS NULL OR backfill_next_at <= ? OR (received_at IS NOT NULL AND received_at < ?))
|
||||||
ORDER BY msg_id ASC LIMIT ?
|
ORDER BY backfill_attempts ASC, msg_id ASC LIMIT ?
|
||||||
""".trimIndent(),
|
""".trimIndent(),
|
||||||
{ ps ->
|
{ ps ->
|
||||||
ps.setTimestamp(1, now.toSqlTimestamp())
|
ps.setTimestamp(1, now.toSqlTimestamp())
|
||||||
@@ -236,20 +277,27 @@ class JdbcProcStateRepository(
|
|||||||
"UPDATE proc_state SET state = 'PENDING', attempts = 0, next_attempt_at = NULL, processing_started_at = NULL, updated_at = ? " +
|
"UPDATE proc_state SET state = 'PENDING', attempts = 0, next_attempt_at = NULL, processing_started_at = NULL, updated_at = ? " +
|
||||||
"WHERE state IN ('FAILED', 'DEAD') AND error_class IN ($placeholders)",
|
"WHERE state IN ('FAILED', 'DEAD') AND error_class IN ($placeholders)",
|
||||||
{ ps ->
|
{ ps ->
|
||||||
ps.setTimestamp(1, Instant.now().toSqlTimestamp())
|
ps.setTimestamp(1, clock.instant().toSqlTimestamp())
|
||||||
errorClasses.forEachIndexed { i, ec -> ps.setString(i + 2, ec.name) }
|
errorClasses.forEachIndexed { i, ec -> ps.setString(i + 2, ec.name) }
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
/** 积压观测:未处理条数、最老一条的接收时间、处理完但未回填的条数。 */
|
/** 积压观测:未处理条数、最老一条的接收时间、处理完但未回填的条数。 */
|
||||||
|
override fun hasAny(): Boolean =
|
||||||
|
ds.queryOne("SELECT 1 FROM proc_state LIMIT 1", {}) { 1 } != null
|
||||||
|
|
||||||
override fun backlog(): Backlog =
|
override fun backlog(): Backlog =
|
||||||
ds.queryOne(
|
ds.queryOne(
|
||||||
"""
|
"""
|
||||||
SELECT
|
SELECT
|
||||||
count(*) FILTER (WHERE state IN ('PENDING', 'FAILED')) AS unfinished,
|
count(*) FILTER (WHERE state IN ('PENDING', 'FAILED')) AS unfinished,
|
||||||
min(received_at) FILTER (WHERE state IN ('PENDING', 'FAILED')) AS oldest_received_at,
|
min(received_at) FILTER (WHERE state IN ('PENDING', 'FAILED')) AS oldest_received_at,
|
||||||
count(*) FILTER (WHERE state IN ('SUCCEEDED', 'SKIPPED', 'DEAD') AND backfill_at IS NULL) AS unmarked_terminal
|
count(*) FILTER (WHERE state IN ('SUCCEEDED', 'SKIPPED', 'DEAD') AND backfill_at IS NULL) AS unmarked_terminal,
|
||||||
|
count(*) FILTER (WHERE state IN ('SUCCEEDED', 'SKIPPED', 'DEAD') AND backfill_at IS NULL
|
||||||
|
AND backfill_abandoned_at IS NOT NULL) AS abandoned_backfill,
|
||||||
|
min(received_at) FILTER (WHERE state IN ('SUCCEEDED', 'SKIPPED', 'DEAD') AND backfill_at IS NULL
|
||||||
|
AND backfill_abandoned_at IS NULL) AS oldest_unmarked_at
|
||||||
FROM proc_state
|
FROM proc_state
|
||||||
""".trimIndent(),
|
""".trimIndent(),
|
||||||
{},
|
{},
|
||||||
@@ -258,6 +306,8 @@ class JdbcProcStateRepository(
|
|||||||
unfinished = rs.getInt("unfinished"),
|
unfinished = rs.getInt("unfinished"),
|
||||||
oldestReceivedAt = rs.getInstant("oldest_received_at"),
|
oldestReceivedAt = rs.getInstant("oldest_received_at"),
|
||||||
unmarkedTerminal = rs.getInt("unmarked_terminal"),
|
unmarkedTerminal = rs.getInt("unmarked_terminal"),
|
||||||
|
abandonedBackfill = rs.getInt("abandoned_backfill"),
|
||||||
|
oldestUnmarkedAt = rs.getInstant("oldest_unmarked_at"),
|
||||||
)
|
)
|
||||||
} ?: Backlog(0, null, 0)
|
} ?: Backlog(0, null, 0)
|
||||||
|
|
||||||
@@ -274,14 +324,17 @@ class JdbcProcStateRepository(
|
|||||||
backfillNextAt = rs.getInstant("backfill_next_at"),
|
backfillNextAt = rs.getInstant("backfill_next_at"),
|
||||||
backfillAttempts = rs.getInt("backfill_attempts"),
|
backfillAttempts = rs.getInt("backfill_attempts"),
|
||||||
backfillError = rs.getString("backfill_error"),
|
backfillError = rs.getString("backfill_error"),
|
||||||
|
backfillAbandonedAt = rs.getInstant("backfill_abandoned_at"),
|
||||||
|
backfillAbandonedReason = rs.getString("backfill_abandoned_reason"),
|
||||||
processingStartedAt = rs.getInstant("processing_started_at"),
|
processingStartedAt = rs.getInstant("processing_started_at"),
|
||||||
updatedAt = rs.getInstant("updated_at") ?: Instant.now(),
|
updatedAt = rs.getInstant("updated_at") ?: clock.instant(),
|
||||||
)
|
)
|
||||||
|
|
||||||
private companion object {
|
private companion object {
|
||||||
const val SELECT_PROC =
|
const val SELECT_PROC =
|
||||||
"SELECT msg_id, state, identity_key, attempts, next_attempt_at, error_class, last_error, " +
|
"SELECT msg_id, state, identity_key, attempts, next_attempt_at, error_class, last_error, " +
|
||||||
"received_at, backfill_at, backfill_next_at, backfill_attempts, backfill_error, processing_started_at, updated_at FROM proc_state"
|
"received_at, backfill_at, backfill_next_at, backfill_attempts, backfill_error, " +
|
||||||
|
"backfill_abandoned_at, backfill_abandoned_reason, processing_started_at, updated_at FROM proc_state"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -291,24 +344,82 @@ class JdbcProcStateRepository(
|
|||||||
@Requires(missingProperty = "msgx.stubs")
|
@Requires(missingProperty = "msgx.stubs")
|
||||||
class JdbcInboxCursorRepository(
|
class JdbcInboxCursorRepository(
|
||||||
private val ds: DataSource,
|
private val ds: DataSource,
|
||||||
|
private val clock: Clock,
|
||||||
) : InboxCursorRepository {
|
) : InboxCursorRepository {
|
||||||
override fun load(): InboxCursorRepository.Cursor =
|
override fun load(): InboxCursorRepository.Cursor =
|
||||||
ds.queryOne(
|
ds.queryOne(
|
||||||
"SELECT committed_up_to, hole_since FROM inbox_cursor WHERE cursor_id = 1",
|
"SELECT committed_up_to, hole_since, seeded_at FROM inbox_cursor WHERE cursor_id = 1",
|
||||||
{},
|
{},
|
||||||
) { rs -> InboxCursorRepository.Cursor(rs.getLong("committed_up_to"), rs.getInstant("hole_since")) }
|
) { rs ->
|
||||||
?: InboxCursorRepository.Cursor()
|
InboxCursorRepository.Cursor(
|
||||||
|
rs.getLong("committed_up_to"),
|
||||||
|
rs.getInstant("hole_since"),
|
||||||
|
rs.getInstant("seeded_at"),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
?: InboxCursorRepository.Cursor().also {
|
||||||
|
// 缺行不是"水位为 0"的同义词:这里只报一次,随后 save() 会 upsert 自愈。
|
||||||
|
if (missingCursorWarned.compareAndSet(false, true)) {
|
||||||
|
log.error("INBOX_CURSOR row (cursor_id=1) is missing; watermark will be re-created on the next save()")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 水位推进必须是 upsert,不能是裸 UPDATE:
|
||||||
|
* 裸 UPDATE 在游标行缺失时影响 0 行且不报错,会让水位永远停在初值、每轮重扫同一批,
|
||||||
|
* 收报静默死锁在第一批。upsert 让缺行自愈,且与入队同事务提交。
|
||||||
|
*/
|
||||||
override fun save(cursor: InboxCursorRepository.Cursor) {
|
override fun save(cursor: InboxCursorRepository.Cursor) {
|
||||||
ds.update(
|
// 先 UPDATE、缺行再 INSERT:比 ON CONFLICT 更可移植(H2 的 PostgreSQL 兼容模式不支持 ON CONFLICT),
|
||||||
|
// 且 InboxPoller 始终把它放在同一个 PG 事务里提交,因此两条语句对外仍是原子的。
|
||||||
|
// 关键点是**缺行必须能自愈**:裸 UPDATE 影响 0 行却不报错,会让水位永远停在初值、每轮重扫同一批。
|
||||||
|
val updated = ds.update(
|
||||||
"UPDATE inbox_cursor SET committed_up_to = ?, hole_since = ?, updated_at = ? WHERE cursor_id = 1",
|
"UPDATE inbox_cursor SET committed_up_to = ?, hole_since = ?, updated_at = ? WHERE cursor_id = 1",
|
||||||
{ ps ->
|
{ ps ->
|
||||||
ps.setLong(1, cursor.committedUpTo)
|
ps.setLong(1, cursor.committedUpTo)
|
||||||
ps.setTimestamp(2, cursor.holeSince?.toSqlTimestamp())
|
ps.setTimestamp(2, cursor.holeSince?.toSqlTimestamp())
|
||||||
ps.setTimestamp(3, Instant.now().toSqlTimestamp())
|
ps.setTimestamp(3, clock.instant().toSqlTimestamp())
|
||||||
|
},
|
||||||
|
)
|
||||||
|
if (updated > 0) return
|
||||||
|
ds.update(
|
||||||
|
"INSERT INTO inbox_cursor (cursor_id, committed_up_to, hole_since, updated_at) VALUES (1, ?, ?, ?)",
|
||||||
|
{ ps ->
|
||||||
|
ps.setLong(1, cursor.committedUpTo)
|
||||||
|
ps.setTimestamp(2, cursor.holeSince?.toSqlTimestamp())
|
||||||
|
ps.setTimestamp(3, clock.instant().toSqlTimestamp())
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 切流播种:一次性写入水位 + 清空空洞计时 + 记录播种事实(同一条语句)。
|
||||||
|
* 与 [save] 分开,普通轮次推进水位不会把播种标记抹掉。
|
||||||
|
*/
|
||||||
|
override fun markSeeded(committedUpTo: Long, now: Instant) {
|
||||||
|
val updated = ds.update(
|
||||||
|
"UPDATE inbox_cursor SET committed_up_to = ?, hole_since = NULL, seeded_at = ?, updated_at = ? " +
|
||||||
|
"WHERE cursor_id = 1",
|
||||||
|
{ ps ->
|
||||||
|
ps.setLong(1, committedUpTo)
|
||||||
|
ps.setTimestamp(2, now.toSqlTimestamp())
|
||||||
|
ps.setTimestamp(3, now.toSqlTimestamp())
|
||||||
|
},
|
||||||
|
)
|
||||||
|
if (updated > 0) return
|
||||||
|
ds.update(
|
||||||
|
"INSERT INTO inbox_cursor (cursor_id, committed_up_to, hole_since, seeded_at, updated_at) " +
|
||||||
|
"VALUES (1, ?, NULL, ?, ?)",
|
||||||
|
{ ps ->
|
||||||
|
ps.setLong(1, committedUpTo)
|
||||||
|
ps.setTimestamp(2, now.toSqlTimestamp())
|
||||||
|
ps.setTimestamp(3, now.toSqlTimestamp())
|
||||||
|
},
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
private val missingCursorWarned = AtomicBoolean(false)
|
||||||
|
private val log = org.slf4j.LoggerFactory.getLogger(JdbcInboxCursorRepository::class.java)
|
||||||
}
|
}
|
||||||
|
|
||||||
@Singleton
|
@Singleton
|
||||||
@@ -316,6 +427,7 @@ class JdbcInboxCursorRepository(
|
|||||||
@Requires(missingProperty = "msgx.stubs")
|
@Requires(missingProperty = "msgx.stubs")
|
||||||
class JdbcMsgEventRepository(
|
class JdbcMsgEventRepository(
|
||||||
private val ds: DataSource,
|
private val ds: DataSource,
|
||||||
|
private val clock: Clock,
|
||||||
) : MsgEventRepository {
|
) : MsgEventRepository {
|
||||||
override fun insertAll(events: List<MsgEvent>): List<Long> =
|
override fun insertAll(events: List<MsgEvent>): List<Long> =
|
||||||
events.map { e ->
|
events.map { e ->
|
||||||
@@ -414,7 +526,7 @@ class JdbcMsgEventRepository(
|
|||||||
nextAttemptAt = rs.getInstant("next_attempt_at"),
|
nextAttemptAt = rs.getInstant("next_attempt_at"),
|
||||||
errorClass = rs.getString("error_class")?.let(ErrorClass::valueOf),
|
errorClass = rs.getString("error_class")?.let(ErrorClass::valueOf),
|
||||||
lastError = rs.getString("last_error"),
|
lastError = rs.getString("last_error"),
|
||||||
createdAt = rs.getInstant("created_at") ?: Instant.now(),
|
createdAt = rs.getInstant("created_at") ?: clock.instant(),
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -423,6 +535,7 @@ class JdbcMsgEventRepository(
|
|||||||
@Requires(missingProperty = "msgx.stubs")
|
@Requires(missingProperty = "msgx.stubs")
|
||||||
class JdbcFlightStateRepository(
|
class JdbcFlightStateRepository(
|
||||||
private val ds: DataSource,
|
private val ds: DataSource,
|
||||||
|
private val clock: Clock,
|
||||||
) : FlightStateRepository {
|
) : FlightStateRepository {
|
||||||
|
|
||||||
private fun mapMainRow(rs: ResultSet) = FlightMainRow(
|
private fun mapMainRow(rs: ResultSet) = FlightMainRow(
|
||||||
@@ -431,7 +544,7 @@ class JdbcFlightStateRepository(
|
|||||||
state = FlightState.valueOf(rs.getString("state")),
|
state = FlightState.valueOf(rs.getString("state")),
|
||||||
stateVersion = rs.getLong("state_version"),
|
stateVersion = rs.getLong("state_version"),
|
||||||
lastMsgId = rs.getLong("last_msg_id").takeIf { !rs.wasNull() },
|
lastMsgId = rs.getLong("last_msg_id").takeIf { !rs.wasNull() },
|
||||||
updatedAt = rs.getInstant("updated_at") ?: Instant.now(),
|
updatedAt = rs.getInstant("updated_at") ?: clock.instant(),
|
||||||
)
|
)
|
||||||
|
|
||||||
override fun findMainRow(flid: String): FlightMainRow? =
|
override fun findMainRow(flid: String): FlightMainRow? =
|
||||||
@@ -711,6 +824,7 @@ class JdbcFlightStateRepository(
|
|||||||
@Requires(missingProperty = "msgx.stubs")
|
@Requires(missingProperty = "msgx.stubs")
|
||||||
class JdbcSnapshotLogRepository(
|
class JdbcSnapshotLogRepository(
|
||||||
private val ds: DataSource,
|
private val ds: DataSource,
|
||||||
|
private val clock: Clock,
|
||||||
) : SnapshotLogRepository {
|
) : SnapshotLogRepository {
|
||||||
/** 只追加写;这里不抛异常,写失败由调用方捕获并记为指标。 */
|
/** 只追加写;这里不抛异常,写失败由调用方捕获并记为指标。 */
|
||||||
override fun append(entry: SnapshotLogEntry) {
|
override fun append(entry: SnapshotLogEntry) {
|
||||||
@@ -730,7 +844,7 @@ class JdbcSnapshotLogRepository(
|
|||||||
ps.setString(8, entry.result.name)
|
ps.setString(8, entry.result.name)
|
||||||
ps.setString(9, entry.flags.joinToString(",") { it.name })
|
ps.setString(9, entry.flags.joinToString(",") { it.name })
|
||||||
ps.setString(10, entry.archiveKey)
|
ps.setString(10, entry.archiveKey)
|
||||||
ps.setTimestamp(11, Instant.now().toSqlTimestamp())
|
ps.setTimestamp(11, clock.instant().toSqlTimestamp())
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
@@ -741,6 +855,7 @@ class JdbcSnapshotLogRepository(
|
|||||||
@Requires(missingProperty = "msgx.stubs")
|
@Requires(missingProperty = "msgx.stubs")
|
||||||
class JdbcReqTrackRepository(
|
class JdbcReqTrackRepository(
|
||||||
private val ds: DataSource,
|
private val ds: DataSource,
|
||||||
|
private val clock: Clock,
|
||||||
) : ReqTrackRepository {
|
) : ReqTrackRepository {
|
||||||
override fun insert(reqType: String, operationDay: LocalDate, sender: String): Long =
|
override fun insert(reqType: String, operationDay: LocalDate, sender: String): Long =
|
||||||
ds.updateReturningLong(
|
ds.updateReturningLong(
|
||||||
@@ -749,7 +864,7 @@ class JdbcReqTrackRepository(
|
|||||||
ps.setString(1, reqType)
|
ps.setString(1, reqType)
|
||||||
ps.setDate(2, java.sql.Date.valueOf(operationDay))
|
ps.setDate(2, java.sql.Date.valueOf(operationDay))
|
||||||
ps.setString(3, sender)
|
ps.setString(3, sender)
|
||||||
ps.setTimestamp(4, Instant.now().toSqlTimestamp())
|
ps.setTimestamp(4, clock.instant().toSqlTimestamp())
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -789,7 +904,7 @@ class JdbcReqTrackRepository(
|
|||||||
)
|
)
|
||||||
""".trimIndent(),
|
""".trimIndent(),
|
||||||
{ ps ->
|
{ ps ->
|
||||||
ps.setTimestamp(1, Instant.now().toSqlTimestamp())
|
ps.setTimestamp(1, clock.instant().toSqlTimestamp())
|
||||||
ps.setString(2, reqType)
|
ps.setString(2, reqType)
|
||||||
ps.setDate(3, java.sql.Date.valueOf(operationDay))
|
ps.setDate(3, java.sql.Date.valueOf(operationDay))
|
||||||
ps.setString(4, sender)
|
ps.setString(4, sender)
|
||||||
|
|||||||
+6
@@ -24,6 +24,12 @@ class MailboxDataSourceFactory {
|
|||||||
driverClassName = cfg.driverClassName
|
driverClassName = cfg.driverClassName
|
||||||
maximumPoolSize = 5
|
maximumPoolSize = 5
|
||||||
poolName = "mailbox"
|
poolName = "mailbox"
|
||||||
|
// 有界外部调用:连接、取连接、校验都必须有上限,否则一次网络黑洞会永久挂住调用线程。
|
||||||
|
connectionTimeout = cfg.poolConnectionTimeoutMs
|
||||||
|
validationTimeout = cfg.poolValidationTimeoutMs
|
||||||
|
// 驱动级超时通过 dataSourceProperties 透传给 Connector/J(socketTimeout 默认 0 = 无限等待)。
|
||||||
|
addDataSourceProperty("connectTimeout", cfg.connectTimeoutMs.toString())
|
||||||
|
addDataSourceProperty("socketTimeout", cfg.socketTimeoutMs.toString())
|
||||||
}
|
}
|
||||||
return HikariDataSource(hikari)
|
return HikariDataSource(hikari)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -29,6 +29,7 @@ class ProcFailure(
|
|||||||
attempts = attempts,
|
attempts = attempts,
|
||||||
errorClass = ErrorClass.EXHAUSTED,
|
errorClass = ErrorClass.EXHAUSTED,
|
||||||
lastError = "$reason; attempts=$attempts",
|
lastError = "$reason; attempts=$attempts",
|
||||||
|
now = scheduler.now(),
|
||||||
)
|
)
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -10,11 +10,16 @@ import jakarta.inject.Singleton
|
|||||||
*
|
*
|
||||||
* 只有"再试一次有可能成功"的错误类才放行。报文本身不合法的(MALFORMED)重放多少次都一样,
|
* 只有"再试一次有可能成功"的错误类才放行。报文本身不合法的(MALFORMED)重放多少次都一样,
|
||||||
* 所以不在白名单里;没列出的错误类也一律不放行。
|
* 所以不在白名单里;没列出的错误类也一律不放行。
|
||||||
|
*
|
||||||
|
* 与回填共用同一个 [MessageLifecycleGate],**必须由容器注入同一个单例**:若两者各持一把锁,
|
||||||
|
* 互斥失效,"旧回填给已重新入队的消息写标记"的窗口就会重新打开。
|
||||||
|
* 装配正确性由 `PipelineSmokeTest` 的同例断言守(不依赖 Kotlin 默认参数值)。
|
||||||
*/
|
*/
|
||||||
@Singleton
|
@Singleton
|
||||||
class ReplayService(
|
class ReplayService(
|
||||||
private val procState: ProcStateRepository,
|
private val procState: ProcStateRepository,
|
||||||
private val lifecycleGate: MessageLifecycleGate = MessageLifecycleGate(),
|
/** 与回填共用的互斥门;`internal` 以便装配测试断言两者拿到同一实例。 */
|
||||||
|
internal val lifecycleGate: MessageLifecycleGate,
|
||||||
) {
|
) {
|
||||||
private val log = org.slf4j.LoggerFactory.getLogger(ReplayService::class.java)
|
private val log = org.slf4j.LoggerFactory.getLogger(ReplayService::class.java)
|
||||||
/** 可以重放的错误类:解码逻辑修好后能过、处理器补齐后能过、基础设施抖动已恢复、以及重试耗尽但人工复核认为还能再试的。 */
|
/** 可以重放的错误类:解码逻辑修好后能过、处理器补齐后能过、基础设施抖动已恢复、以及重试耗尽但人工复核认为还能再试的。 */
|
||||||
|
|||||||
@@ -130,6 +130,8 @@ class StubProcState : ProcStateRepository {
|
|||||||
backfillNextAt = now,
|
backfillNextAt = now,
|
||||||
backfillAttempts = 0,
|
backfillAttempts = 0,
|
||||||
backfillError = null,
|
backfillError = null,
|
||||||
|
backfillAbandonedAt = null,
|
||||||
|
backfillAbandonedReason = null,
|
||||||
updatedAt = now,
|
updatedAt = now,
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
@@ -151,23 +153,54 @@ class StubProcState : ProcStateRepository {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
override fun markBackfillAbandoned(msgId: Long, reason: String, now: Instant): Boolean {
|
||||||
|
val old = rows[msgId] ?: return false
|
||||||
|
if (old.backfillAt != null || old.backfillAbandonedAt != null) return false
|
||||||
|
// 放弃 ≠ 标记已确认:这里**不**设 backfillAt,清除前提因此仍不成立。
|
||||||
|
rows[msgId] = old.copy(
|
||||||
|
backfillAbandonedAt = now,
|
||||||
|
backfillAbandonedReason = reason,
|
||||||
|
backfillNextAt = null,
|
||||||
|
updatedAt = now,
|
||||||
|
)
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
override fun reopenBackfill(msgId: Long, now: Instant): Boolean {
|
||||||
|
val old = rows[msgId] ?: return false
|
||||||
|
if (old.backfillAt != null || old.backfillAbandonedAt == null) return false
|
||||||
|
rows[msgId] = old.copy(
|
||||||
|
backfillAbandonedAt = null,
|
||||||
|
backfillAbandonedReason = null,
|
||||||
|
backfillNextAt = now,
|
||||||
|
updatedAt = now,
|
||||||
|
)
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
override fun findBackfillDue(now: Instant, overdueBefore: Instant, limit: Int): List<BackfillDue> =
|
override fun findBackfillDue(now: Instant, overdueBefore: Instant, limit: Int): List<BackfillDue> =
|
||||||
rows.values
|
rows.values
|
||||||
.filter { it.state.isTerminal() && it.backfillAt == null }
|
.filter { it.state.isTerminal() && it.backfillAt == null && it.backfillAbandonedAt == null }
|
||||||
.filter {
|
.filter {
|
||||||
it.backfillNextAt == null || it.backfillNextAt <= now ||
|
it.backfillNextAt == null || it.backfillNextAt <= now ||
|
||||||
(it.receivedAt != null && it.receivedAt < overdueBefore)
|
(it.receivedAt != null && it.receivedAt < overdueBefore)
|
||||||
}
|
}
|
||||||
.sortedBy { it.msgId }
|
// 公平轮转:先按已尝试次数,再按 msg_id。只按 msg_id 会让最旧的一批永久失败行占满批次。
|
||||||
|
.sortedWith(compareBy({ it.backfillAttempts }, { it.msgId }))
|
||||||
.take(limit)
|
.take(limit)
|
||||||
.map { BackfillDue(it.msgId, it.backfillAttempts) }
|
.map { BackfillDue(it.msgId, it.backfillAttempts) }
|
||||||
|
|
||||||
|
override fun hasAny(): Boolean = rows.isNotEmpty()
|
||||||
|
|
||||||
override fun backlog(): Backlog {
|
override fun backlog(): Backlog {
|
||||||
val unfinished = rows.values.filter { it.state == ProcStatus.PENDING || it.state == ProcStatus.FAILED }
|
val unfinished = rows.values.filter { it.state == ProcStatus.PENDING || it.state == ProcStatus.FAILED }
|
||||||
|
val unmarked = rows.values.filter { it.state.isTerminal() && it.backfillAt == null }
|
||||||
return Backlog(
|
return Backlog(
|
||||||
unfinished = unfinished.size,
|
unfinished = unfinished.size,
|
||||||
oldestReceivedAt = unfinished.mapNotNull { it.receivedAt }.minOrNull(),
|
oldestReceivedAt = unfinished.mapNotNull { it.receivedAt }.minOrNull(),
|
||||||
unmarkedTerminal = rows.values.count { it.state.isTerminal() && it.backfillAt == null },
|
unmarkedTerminal = unmarked.size,
|
||||||
|
abandonedBackfill = unmarked.count { it.backfillAbandonedAt != null },
|
||||||
|
oldestUnmarkedAt = unmarked.filter { it.backfillAbandonedAt == null }.mapNotNull { it.receivedAt }.minOrNull(),
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -404,6 +437,10 @@ class StubInbox : CminmsgInboxRepository {
|
|||||||
|
|
||||||
override fun maxId(): Long? = raws.keys.maxOrNull()
|
override fun maxId(): Long? = raws.keys.maxOrNull()
|
||||||
|
|
||||||
|
override fun minId(): Long? = raws.keys.minOrNull()
|
||||||
|
|
||||||
|
override fun existingIds(msgIds: Collection<Long>): Set<Long> = msgIds.filter { raws.containsKey(it) }.toSet()
|
||||||
|
|
||||||
/** 只写还没有标记的行,并区分已有标记与行缺失。 */
|
/** 只写还没有标记的行,并区分已有标记与行缺失。 */
|
||||||
override fun markProcessedIfUnmarked(msgId: Long, value: String): MailboxMarkResult {
|
override fun markProcessedIfUnmarked(msgId: Long, value: String): MailboxMarkResult {
|
||||||
if (!raws.containsKey(msgId)) return MailboxMarkResult.MISSING
|
if (!raws.containsKey(msgId)) return MailboxMarkResult.MISSING
|
||||||
@@ -444,7 +481,12 @@ class StubInboxCursor : InboxCursorRepository {
|
|||||||
override fun load(): InboxCursorRepository.Cursor = cursor
|
override fun load(): InboxCursorRepository.Cursor = cursor
|
||||||
|
|
||||||
override fun save(cursor: InboxCursorRepository.Cursor) {
|
override fun save(cursor: InboxCursorRepository.Cursor) {
|
||||||
this.cursor = cursor
|
// 普通轮次只推进水位与空洞计时;播种标记由 markSeeded 负责,不能被这里抹掉。
|
||||||
|
this.cursor = this.cursor.copy(committedUpTo = cursor.committedUpTo, holeSince = cursor.holeSince)
|
||||||
|
}
|
||||||
|
|
||||||
|
override fun markSeeded(committedUpTo: Long, now: Instant) {
|
||||||
|
cursor = InboxCursorRepository.Cursor(committedUpTo = committedUpTo, holeSince = null, seededAt = now)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -6,9 +6,12 @@ import com.gzzn.omms.msgexchange.infra.persistence.InboxCursorRepository
|
|||||||
import com.gzzn.omms.msgexchange.infra.persistence.MailboxRow
|
import com.gzzn.omms.msgexchange.infra.persistence.MailboxRow
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.PipelineTransactionManager
|
import com.gzzn.omms.msgexchange.infra.persistence.PipelineTransactionManager
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
||||||
|
import com.gzzn.omms.msgexchange.infra.metrics.PipelineCounters
|
||||||
import jakarta.inject.Singleton
|
import jakarta.inject.Singleton
|
||||||
|
import java.time.Clock
|
||||||
import java.time.Duration
|
import java.time.Duration
|
||||||
import java.time.Instant
|
import java.time.Instant
|
||||||
|
import java.util.concurrent.atomic.AtomicBoolean
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 收报:轮询共享信箱,把新消息登记到自有 PG 的 PROC_STATE,等着主泵处理。
|
* 收报:轮询共享信箱,把新消息登记到自有 PG 的 PROC_STATE,等着主泵处理。
|
||||||
@@ -36,6 +39,8 @@ class InboxPoller(
|
|||||||
private val cursor: InboxCursorRepository,
|
private val cursor: InboxCursorRepository,
|
||||||
private val txManager: PipelineTransactionManager,
|
private val txManager: PipelineTransactionManager,
|
||||||
private val props: PipelineProps,
|
private val props: PipelineProps,
|
||||||
|
private val clock: Clock,
|
||||||
|
private val counters: PipelineCounters,
|
||||||
) {
|
) {
|
||||||
private val log = org.slf4j.LoggerFactory.getLogger(InboxPoller::class.java)
|
private val log = org.slf4j.LoggerFactory.getLogger(InboxPoller::class.java)
|
||||||
|
|
||||||
@@ -44,6 +49,8 @@ class InboxPoller(
|
|||||||
|
|
||||||
/** @return 本轮新登记的消息条数(已登记过的行不计入,也不影响水位推进)。 */
|
/** @return 本轮新登记的消息条数(已登记过的行不计入,也不影响水位推进)。 */
|
||||||
fun pollOnce(now: Instant = Instant.now()): Int {
|
fun pollOnce(now: Instant = Instant.now()): Int {
|
||||||
|
seedCutoverWatermarkIfConfigured(now)
|
||||||
|
detectLateArrivalsIfDue(now)
|
||||||
val batch = props.pipeline.claimBatch.coerceAtLeast(1)
|
val batch = props.pipeline.claimBatch.coerceAtLeast(1)
|
||||||
val watermark = cursor.load()
|
val watermark = cursor.load()
|
||||||
val rows = mailbox.readRange(watermark.committedUpTo, batch)
|
val rows = mailbox.readRange(watermark.committedUpTo, batch)
|
||||||
@@ -61,6 +68,8 @@ class InboxPoller(
|
|||||||
} else {
|
} else {
|
||||||
// 缺口等太久了:当成永久缺失跳过,让水位继续往前走
|
// 缺口等太久了:当成永久缺失跳过,让水位继续往前走
|
||||||
committedTo = rows.first { it.msgId > contiguous }.msgId - 1
|
committedTo = rows.first { it.msgId > contiguous }.msgId - 1
|
||||||
|
counters.holeAgedOutIncrement()
|
||||||
|
rememberAgedHoles(contiguous + 1, committedTo)
|
||||||
log.warn("hole after W={} aged out, watermark advanced to {}", contiguous, committedTo)
|
log.warn("hole after W={} aged out, watermark advanced to {}", contiguous, committedTo)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -79,12 +88,114 @@ class InboxPoller(
|
|||||||
return enqueued
|
return enqueued
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 切流水位播种:**只在显式配置 `msgx.pipeline.cutover-watermark` 时动作,且至多一次**。
|
||||||
|
*
|
||||||
|
* 安全约束(评审要求,逐条对应):
|
||||||
|
* - 默认不配置即不动作,代码不做默认选择、也不会自动退化成 `max`;
|
||||||
|
* - `SEEDED_AT` 为 NULL **不等于**"从未消费"(已有库新增列后同样是 NULL),
|
||||||
|
* 因此必须叠加"水位为 0 且没有任何处理记录"这条判据;
|
||||||
|
* - 升级实例(已消费过)**拒绝重新播种**;重新切流必须是显式操作;
|
||||||
|
* - 播种事实与水位在同一条语句里落库,失败后下一轮重试不会留下部分状态。
|
||||||
|
* - 信箱为空时无从界定边界,保持未播种(等有行时再判断)。
|
||||||
|
*/
|
||||||
|
private fun seedCutoverWatermarkIfConfigured(now: Instant) {
|
||||||
|
val mode = props.pipeline.cutoverWatermark ?: return
|
||||||
|
val current = cursor.load()
|
||||||
|
if (current.seededAt != null) return
|
||||||
|
|
||||||
|
if (current.committedUpTo > 0L || procState.hasAny()) {
|
||||||
|
if (refusedReseed.compareAndSet(false, true)) {
|
||||||
|
log.error(
|
||||||
|
"cutover-watermark={} is configured but this instance has already consumed messages " +
|
||||||
|
"(W={}); refusing to re-seed. Re-cutover must be an explicit operation.",
|
||||||
|
mode, current.committedUpTo,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
val min = mailbox.minId() ?: return
|
||||||
|
val target = when {
|
||||||
|
mode.equals("min", ignoreCase = true) -> min - 1
|
||||||
|
mode.equals("zero", ignoreCase = true) -> 0L
|
||||||
|
mode.equals("max", ignoreCase = true) -> mailbox.maxId() ?: return
|
||||||
|
else -> mode.toLongOrNull() ?: return // 非法值已在启动 validate() 挡掉
|
||||||
|
}
|
||||||
|
|
||||||
|
txManager.inTransaction { cursor.markSeeded(target, now) }
|
||||||
|
log.warn("cutover watermark seeded: mode={} W={} (mailbox min={})", mode, target, min)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** "拒绝重新播种"的告警只打一次,避免每轮刷屏。 */
|
||||||
|
private val refusedReseed = AtomicBoolean(false)
|
||||||
|
|
||||||
|
/**
|
||||||
|
* "已被判定为永久空洞"的 ID 监视队列(ACM2-41 阶段 0,只读检测)。
|
||||||
|
*
|
||||||
|
* 为什么只监视这些 ID 就够:水位只会越过两类 ID——连续存在的(已入队)与被判定为永久空洞的。
|
||||||
|
* 因此凡是"水位越过之后才出现在信箱里"的 ID,必然曾经被当作空洞放行过。
|
||||||
|
*
|
||||||
|
* 队列有界([HOLE_WATCH_LIMIT]),单轮复查量受 `late-detect-batch` 约束;
|
||||||
|
* 它是**进程内**状态,重启后丢失,因此检测是尽力而为的观测(阶段 0 不做补入队)。
|
||||||
|
*/
|
||||||
|
private val holeWatch = ArrayDeque<Long>()
|
||||||
|
|
||||||
|
private var lastLateDetectAt: Instant? = null
|
||||||
|
|
||||||
|
/** 记下被放行的空洞 ID(超大空洞只记前一段,避免内存被吃掉)。 */
|
||||||
|
private fun rememberAgedHoles(from: Long, to: Long) {
|
||||||
|
val cap = props.pipeline.lateDetectBatch.coerceAtLeast(1) * 2
|
||||||
|
var id = from
|
||||||
|
var added = 0
|
||||||
|
while (id <= to && added < cap && holeWatch.size < HOLE_WATCH_LIMIT) {
|
||||||
|
holeWatch.addLast(id)
|
||||||
|
id++
|
||||||
|
added++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 只读的迟到检测:复查监视队列里的 ID 是否**真的出现在信箱里**。
|
||||||
|
* 命中即计数 + 告警,**不入队**(补入队属阶段 1,需先与库方就提交契约定案)。
|
||||||
|
*/
|
||||||
|
private fun detectLateArrivalsIfDue(now: Instant) {
|
||||||
|
val period = props.pipeline.lateDetectPeriod
|
||||||
|
if (period.isZero || period.isNegative) return
|
||||||
|
val last = lastLateDetectAt
|
||||||
|
if (last != null && Duration.between(last, now) < period) return
|
||||||
|
lastLateDetectAt = now
|
||||||
|
if (holeWatch.isEmpty()) return
|
||||||
|
|
||||||
|
val batch = props.pipeline.lateDetectBatch.coerceAtLeast(1)
|
||||||
|
val probe = ArrayList<Long>(minOf(batch, holeWatch.size))
|
||||||
|
repeat(minOf(batch, holeWatch.size)) { probe += holeWatch.removeFirst() }
|
||||||
|
|
||||||
|
val present = try {
|
||||||
|
mailbox.existingIds(probe)
|
||||||
|
} catch (e: Exception) {
|
||||||
|
log.warn("late-arrival detection skipped: {}", e.message)
|
||||||
|
probe.forEach { holeWatch.addLast(it) } // 查不动就把监视放回去,别丢
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if (present.isNotEmpty()) {
|
||||||
|
present.forEach { counters.lateArrivalDetectedIncrement() }
|
||||||
|
log.error(
|
||||||
|
"LATE ARRIVAL: {} message(s) appeared below the watermark after being aged out: {}",
|
||||||
|
present.size, present.sorted().take(20),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
// 已命中的不再监视(计数表示"不同 ID");仍未出现的轮转回队尾继续看。
|
||||||
|
probe.filter { it !in present }.forEach { holeWatch.addLast(it) }
|
||||||
|
}
|
||||||
|
|
||||||
fun loop() {
|
fun loop() {
|
||||||
running = true
|
running = true
|
||||||
log.info("inbox poller loop started")
|
log.info("inbox poller loop started")
|
||||||
while (running) {
|
while (running) {
|
||||||
try {
|
try {
|
||||||
pollOnce()
|
// 时间一律走注入的 Clock:空洞老化阈值是"等多久算永久缺失"的唯一判据,不能依赖系统时钟。
|
||||||
|
pollOnce(clock.instant())
|
||||||
sleepQuietly(props.pipeline.pollInterval)
|
sleepQuietly(props.pipeline.pollInterval)
|
||||||
} catch (_: InterruptedException) {
|
} catch (_: InterruptedException) {
|
||||||
Thread.currentThread().interrupt()
|
Thread.currentThread().interrupt()
|
||||||
@@ -123,4 +234,9 @@ class InboxPoller(
|
|||||||
Thread.currentThread().interrupt()
|
Thread.currentThread().interrupt()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private companion object {
|
||||||
|
/** 监视队列上限:防止超大空洞把内存吃光(阶段 0 是尽力而为的观测,不追求全覆盖)。 */
|
||||||
|
const val HOLE_WATCH_LIMIT = 4096
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ package com.gzzn.omms.msgexchange.ingress
|
|||||||
import com.gzzn.omms.msgexchange.infra.persistence.CminmsgInboxRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.CminmsgInboxRepository
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
||||||
import jakarta.inject.Singleton
|
import jakarta.inject.Singleton
|
||||||
|
import java.time.Clock
|
||||||
import java.time.Instant
|
import java.time.Instant
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -11,11 +12,15 @@ import java.time.Instant
|
|||||||
*
|
*
|
||||||
* 两步不在同一个事务里(跨库没有事务)。如果信箱写成功、PG 入队失败,原文仍然在信箱里,
|
* 两步不在同一个事务里(跨库没有事务)。如果信箱写成功、PG 入队失败,原文仍然在信箱里,
|
||||||
* 收报轮询会按 ID 把它补进来,所以不会丢消息。
|
* 收报轮询会按 ID 把它补进来,所以不会丢消息。
|
||||||
|
*
|
||||||
|
* 【缺口】这里**不参与水位**:它直接写 PROC_STATE,因此当水位卡在低位空洞时,
|
||||||
|
* 这条高 ID 会被主泵提前领取,破坏 FIFO。修复方向见 ACM2-37 / message-lifecycle.md §5.1。
|
||||||
*/
|
*/
|
||||||
@Singleton
|
@Singleton
|
||||||
class InboxService(
|
class InboxService(
|
||||||
private val inbox: CminmsgInboxRepository,
|
private val inbox: CminmsgInboxRepository,
|
||||||
private val procState: ProcStateRepository,
|
private val procState: ProcStateRepository,
|
||||||
|
private val clock: Clock,
|
||||||
) {
|
) {
|
||||||
private val log = org.slf4j.LoggerFactory.getLogger(InboxService::class.java)
|
private val log = org.slf4j.LoggerFactory.getLogger(InboxService::class.java)
|
||||||
|
|
||||||
@@ -23,7 +28,14 @@ class InboxService(
|
|||||||
|
|
||||||
fun accept(rawXml: String): Receipt {
|
fun accept(rawXml: String): Receipt {
|
||||||
val id = inbox.insertRaw(rawXml)
|
val id = inbox.insertRaw(rawXml)
|
||||||
val receivedAt = checkNotNull(inbox.receivedAtOf(id)) { "mailbox receive time missing for msgId=$id" }
|
// 信箱接收时间缺失不能让"已经落信"的写入对外报失败(那会与"确认落信才返回 ID"的
|
||||||
|
// 契约矛盾,并让客户端重试产生额外重复行)。回退到入队时间并留痕;
|
||||||
|
// 时间源用注入的 Clock,保证与收报/回填一致、测试可确定。
|
||||||
|
val mailboxReceivedAt = inbox.receivedAtOf(id)
|
||||||
|
if (mailboxReceivedAt == null) {
|
||||||
|
log.warn("mailbox receive time missing msgId={}, falling back to enqueue time", id)
|
||||||
|
}
|
||||||
|
val receivedAt = mailboxReceivedAt ?: clock.instant()
|
||||||
procState.insertIfAbsent(id, receivedAt)
|
procState.insertIfAbsent(id, receivedAt)
|
||||||
log.info("compat-accepted msgId={}", id)
|
log.info("compat-accepted msgId={}", id)
|
||||||
return Receipt(id, receivedAt)
|
return Receipt(id, receivedAt)
|
||||||
|
|||||||
@@ -4,6 +4,7 @@ import com.gzzn.omms.msgexchange.config.HistoryProps
|
|||||||
import com.gzzn.omms.msgexchange.config.PipelineProps
|
import com.gzzn.omms.msgexchange.config.PipelineProps
|
||||||
import com.gzzn.omms.msgexchange.processing.BackfillService
|
import com.gzzn.omms.msgexchange.processing.BackfillService
|
||||||
import jakarta.inject.Singleton
|
import jakarta.inject.Singleton
|
||||||
|
import java.time.Clock
|
||||||
import java.time.Duration
|
import java.time.Duration
|
||||||
import java.time.Instant
|
import java.time.Instant
|
||||||
import java.time.LocalDate
|
import java.time.LocalDate
|
||||||
@@ -24,6 +25,7 @@ class JobRunner(
|
|||||||
private val historySweep: HistorySweepJob,
|
private val historySweep: HistorySweepJob,
|
||||||
@Suppress("unused") private val pipelineProps: PipelineProps,
|
@Suppress("unused") private val pipelineProps: PipelineProps,
|
||||||
@Suppress("unused") private val historyProps: HistoryProps,
|
@Suppress("unused") private val historyProps: HistoryProps,
|
||||||
|
private val clock: Clock,
|
||||||
) {
|
) {
|
||||||
private val log = org.slf4j.LoggerFactory.getLogger(JobRunner::class.java)
|
private val log = org.slf4j.LoggerFactory.getLogger(JobRunner::class.java)
|
||||||
private val zone: ZoneId = ZoneId.of("Asia/Shanghai")
|
private val zone: ZoneId = ZoneId.of("Asia/Shanghai")
|
||||||
@@ -65,11 +67,13 @@ class JobRunner(
|
|||||||
}
|
}
|
||||||
|
|
||||||
private fun maybeHistorySweep() {
|
private fun maybeHistorySweep() {
|
||||||
val today = LocalDate.now(zone)
|
// 时区换算后的机场时钟:切日与 03:30 门槛都以机场时区为准,且走注入 Clock(测试可确定)。
|
||||||
|
val airportClock = clock.withZone(zone)
|
||||||
|
val today = LocalDate.now(airportClock)
|
||||||
if (lastHistoryDay == today) return
|
if (lastHistoryDay == today) return
|
||||||
val now = java.time.LocalTime.now(zone)
|
val now = java.time.LocalTime.now(airportClock)
|
||||||
if ((now.hour == 3 && now.minute >= 30) || now.hour > 3) {
|
if ((now.hour == 3 && now.minute >= 30) || now.hour > 3) {
|
||||||
val outcome = historySweep.run()
|
val outcome = historySweep.run(clock.instant())
|
||||||
lastHistoryDay = today
|
lastHistoryDay = today
|
||||||
if (outcome.selected > 0 || outcome.snapLogPurged > 0) {
|
if (outcome.selected > 0 || outcome.snapLogPurged > 0) {
|
||||||
log.info("history sweep: {}", outcome)
|
log.info("history sweep: {}", outcome)
|
||||||
|
|||||||
@@ -34,7 +34,8 @@ class BackfillService(
|
|||||||
private val mailboxProps: MailboxProps,
|
private val mailboxProps: MailboxProps,
|
||||||
private val props: PipelineProps,
|
private val props: PipelineProps,
|
||||||
private val clock: Clock,
|
private val clock: Clock,
|
||||||
private val lifecycleGate: MessageLifecycleGate = MessageLifecycleGate(),
|
/** 与人工重放共用的互斥门;`internal` 以便装配测试断言两者拿到同一实例。 */
|
||||||
|
internal val lifecycleGate: MessageLifecycleGate,
|
||||||
) {
|
) {
|
||||||
private val log = org.slf4j.LoggerFactory.getLogger(BackfillService::class.java)
|
private val log = org.slf4j.LoggerFactory.getLogger(BackfillService::class.java)
|
||||||
|
|
||||||
@@ -43,6 +44,12 @@ class BackfillService(
|
|||||||
private val MAX_BACKOFF: Duration = Duration.ofMinutes(15)
|
private val MAX_BACKOFF: Duration = Duration.ofMinutes(15)
|
||||||
private val TERMINAL_STATES = setOf(ProcStatus.SUCCEEDED, ProcStatus.SKIPPED, ProcStatus.DEAD)
|
private val TERMINAL_STATES = setOf(ProcStatus.SUCCEEDED, ProcStatus.SKIPPED, ProcStatus.DEAD)
|
||||||
|
|
||||||
|
/** 放弃原因:运行时查询确认信箱行不存在(确定性结论,重试不会改变结果)。 */
|
||||||
|
const val ABANDON_MISSING_ROW = "MISSING_ROW"
|
||||||
|
|
||||||
|
/** 放弃原因:暂时性故障达到尝试上限;停止自动重试,但保留人工恢复能力。 */
|
||||||
|
const val ABANDON_MAX_ATTEMPTS = "MAX_ATTEMPTS"
|
||||||
|
|
||||||
fun backoffDelayFor(attempts: Int): Duration {
|
fun backoffDelayFor(attempts: Int): Duration {
|
||||||
val shift = (attempts - 1).coerceIn(0, 20)
|
val shift = (attempts - 1).coerceIn(0, 20)
|
||||||
return INITIAL_BACKOFF.multipliedBy(1L shl shift).coerceAtMost(MAX_BACKOFF)
|
return INITIAL_BACKOFF.multipliedBy(1L shl shift).coerceAtMost(MAX_BACKOFF)
|
||||||
@@ -58,15 +65,27 @@ class BackfillService(
|
|||||||
val row = procState.find(msgId) ?: return@exclusive
|
val row = procState.find(msgId) ?: return@exclusive
|
||||||
if (row.state !in TERMINAL_STATES) return@exclusive
|
if (row.state !in TERMINAL_STATES) return@exclusive
|
||||||
if (row.backfillAt != null) return@exclusive
|
if (row.backfillAt != null) return@exclusive
|
||||||
|
if (row.backfillAbandonedAt != null) return@exclusive
|
||||||
record(msgId, attempts = row.backfillAttempts, now = now)?.let {
|
record(msgId, attempts = row.backfillAttempts, now = now)?.let {
|
||||||
log.warn("backfill failed msgId={} error={} (sweep will retry)", msgId, it)
|
log.warn("backfill failed msgId={} error={} (sweep will retry)", msgId, it)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 人工恢复入口:清除放弃标记并重排一次回填。
|
||||||
|
* 用于暂时性故障恢复之后,或人工对账确认该行仍应打标之后。
|
||||||
|
*/
|
||||||
|
fun reopen(msgId: Long, now: Instant = clock.instant()): Boolean =
|
||||||
|
lifecycleGate.exclusive { procState.reopenBackfill(msgId, now) }
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 批量补写,由 JobRunner 每 30 秒调用一次。待办状态都在数据库里,
|
* 批量补写,由 JobRunner 每 30 秒调用一次。待办状态都在数据库里,
|
||||||
* 进程重启后接着跑,不需要额外恢复步骤。
|
* 进程重启后接着跑,不需要额外恢复步骤。
|
||||||
|
*
|
||||||
|
* 注意:**调度周期不是完成时限**——扫描与历史作业串行,批次积压、单行调用超时与
|
||||||
|
* 历史作业耗时都会延长实际回填延迟。
|
||||||
|
*
|
||||||
* @return 本批处理的条数
|
* @return 本批处理的条数
|
||||||
*/
|
*/
|
||||||
fun sweep(now: Instant = clock.instant()): Int {
|
fun sweep(now: Instant = clock.instant()): Int {
|
||||||
@@ -75,19 +94,37 @@ class BackfillService(
|
|||||||
return due.size
|
return due.size
|
||||||
}
|
}
|
||||||
|
|
||||||
/** 回填一条。@return 失败原因;返回 null 表示标记已确认(包括"别的路径已经标过了")。 */
|
/** 回填一条。@return 失败原因;返回 null 表示已处理完(写成功/早已标记/已放弃)。 */
|
||||||
private fun record(msgId: Long, attempts: Int, now: Instant): String? =
|
private fun record(msgId: Long, attempts: Int, now: Instant): String? =
|
||||||
try {
|
try {
|
||||||
// 没有真正写进去说明库里已经有标记了,同样算成功(不覆盖已有值)
|
when (mailbox.markProcessedIfUnmarked(msgId, mailboxProps.processedValue)) {
|
||||||
val result = mailbox.markProcessedIfUnmarked(msgId, mailboxProps.processedValue)
|
MailboxMarkResult.MARKED, MailboxMarkResult.ALREADY_MARKED -> {
|
||||||
check(result != MailboxMarkResult.MISSING) { "mailbox-row-missing" }
|
// 没有真正写进去说明库里已经有标记了,同样算成功(不覆盖已有值)
|
||||||
procState.markBackfilled(msgId, now)
|
procState.markBackfilled(msgId, now)
|
||||||
null
|
null
|
||||||
|
}
|
||||||
|
MailboxMarkResult.MISSING -> {
|
||||||
|
// 确定性结论:行不存在,重试不会改变结果。停止自动补偿并把事实留痕。
|
||||||
|
// 这**不等于**标记已确认(BACKFILL_AT 仍为空),清除前提因此仍然不成立。
|
||||||
|
procState.markBackfillAbandoned(msgId, ABANDON_MISSING_ROW, now)
|
||||||
|
log.error("backfill abandoned: mailbox row missing msgId={}", msgId)
|
||||||
|
null
|
||||||
|
}
|
||||||
|
}
|
||||||
} catch (e: Exception) {
|
} catch (e: Exception) {
|
||||||
|
// 超时/连接失败是暂时性的,**不能**当作缺行证据:按退避重试。
|
||||||
|
// 达到上限后停止自动重试(保留人工恢复能力),避免永久占满扫描批次造成饥饿。
|
||||||
val reason = e.message ?: e.javaClass.simpleName
|
val reason = e.message ?: e.javaClass.simpleName
|
||||||
runCatching {
|
val nextAttempts = attempts + 1
|
||||||
procState.recordBackfillFailure(msgId, reason, attempts + 1, now.plus(backoffDelayFor(attempts + 1)), now)
|
if (nextAttempts >= props.pipeline.backfillMaxAttempts) {
|
||||||
}.onFailure { log.error("record backfill failure failed msgId={}", msgId, it) }
|
runCatching { procState.markBackfillAbandoned(msgId, ABANDON_MAX_ATTEMPTS, now) }
|
||||||
|
.onFailure { log.error("abandon backfill failed msgId={}", msgId, it) }
|
||||||
|
log.error("backfill abandoned after {} attempts msgId={} error={}", nextAttempts, msgId, reason)
|
||||||
|
} else {
|
||||||
|
runCatching {
|
||||||
|
procState.recordBackfillFailure(msgId, reason, nextAttempts, now.plus(backoffDelayFor(nextAttempts)), now)
|
||||||
|
}.onFailure { log.error("record backfill failure failed msgId={}", msgId, it) }
|
||||||
|
}
|
||||||
reason
|
reason
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -17,10 +17,12 @@ import com.gzzn.omms.msgexchange.domain.flight.MergeChange
|
|||||||
import com.gzzn.omms.msgexchange.domain.flight.ScheduleRecord
|
import com.gzzn.omms.msgexchange.domain.flight.ScheduleRecord
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.FlightStateRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.FlightStateRepository
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.MsgEventRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.MsgEventRepository
|
||||||
|
import com.gzzn.omms.msgexchange.infra.persistence.PersistOutcome
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.PipelineLockRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.PipelineLockRepository
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.PipelineTransactionManager
|
import com.gzzn.omms.msgexchange.infra.persistence.PipelineTransactionManager
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
||||||
import jakarta.inject.Singleton
|
import jakarta.inject.Singleton
|
||||||
|
import java.time.Clock
|
||||||
import java.time.Instant
|
import java.time.Instant
|
||||||
import java.time.ZoneId
|
import java.time.ZoneId
|
||||||
|
|
||||||
@@ -39,21 +41,22 @@ class FlopProcessor(
|
|||||||
private val msgEvents: MsgEventRepository,
|
private val msgEvents: MsgEventRepository,
|
||||||
private val procState: ProcStateRepository,
|
private val procState: ProcStateRepository,
|
||||||
private val mapper: ObjectMapper,
|
private val mapper: ObjectMapper,
|
||||||
|
private val clock: Clock,
|
||||||
) {
|
) {
|
||||||
fun apply(head: ProcState, msg: DecodedMessage, payload: FlopPayload): ApplyResult = txManager.inTransaction {
|
fun apply(head: ProcState, msg: DecodedMessage, payload: FlopPayload): ApplyResult = txManager.inTransaction {
|
||||||
lock.lock()
|
lock.lock()
|
||||||
val current = flightState.loadFullSnapshot(payload.flid)
|
val current = flightState.loadFullSnapshot(payload.flid)
|
||||||
if (current == null) {
|
if (current == null) {
|
||||||
// 迟到/未知航班:幂等成功,不创建(创建入口只有 SCHD/ADFT);终态同事务落库
|
// 迟到/未知航班:幂等成功,不创建(创建入口只有 SCHD/ADFT);终态同事务落库
|
||||||
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED)
|
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED, now = clock.instant())
|
||||||
return@inTransaction ApplyResult.Succeeded
|
return@inTransaction ApplyResult.Succeeded
|
||||||
}
|
}
|
||||||
|
|
||||||
val change = MergeChange(flid = payload.flid, scalars = payload.scalars, collections = payload.collections)
|
val change = MergeChange(flid = payload.flid, scalars = payload.scalars, collections = payload.collections)
|
||||||
val next = FlightStateEngine.mergedState(current, change)
|
val next = FlightStateEngine.mergedState(current, change)
|
||||||
flightState.persistFullState(next, msgId = head.msgId, now = Instant.now())
|
flightState.persistFullState(next, msgId = head.msgId, now = clock.instant())
|
||||||
msgEvents.insertAll(eventsFor(next, mapper))
|
msgEvents.insertAll(eventsFor(next, mapper))
|
||||||
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED)
|
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED, now = clock.instant())
|
||||||
ApplyResult.Succeeded
|
ApplyResult.Succeeded
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -72,10 +75,11 @@ class FdelProcessor(
|
|||||||
private val msgEvents: MsgEventRepository,
|
private val msgEvents: MsgEventRepository,
|
||||||
private val procState: ProcStateRepository,
|
private val procState: ProcStateRepository,
|
||||||
private val mapper: ObjectMapper,
|
private val mapper: ObjectMapper,
|
||||||
|
private val clock: Clock,
|
||||||
) {
|
) {
|
||||||
fun apply(head: ProcState, msg: DecodedMessage, payload: FlopPayload): ApplyResult = txManager.inTransaction {
|
fun apply(head: ProcState, msg: DecodedMessage, payload: FlopPayload): ApplyResult = txManager.inTransaction {
|
||||||
lock.lock()
|
lock.lock()
|
||||||
val deleted = flightState.markDeleted(payload.flid, msgId = head.msgId, now = Instant.now())
|
val deleted = flightState.markDeleted(payload.flid, msgId = head.msgId, now = clock.instant())
|
||||||
if (deleted) {
|
if (deleted) {
|
||||||
val current = flightState.loadFullSnapshot(payload.flid)
|
val current = flightState.loadFullSnapshot(payload.flid)
|
||||||
// 只有"在用 → 删除"这一步才发删除通知,而且和状态变更写在同一个事务里
|
// 只有"在用 → 删除"这一步才发删除通知,而且和状态变更写在同一个事务里
|
||||||
@@ -105,7 +109,7 @@ class FdelProcessor(
|
|||||||
),
|
),
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED) // 没删到东西说明是迟到或重复报文,照样算成功
|
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED, now = clock.instant()) // 没删到东西说明是迟到或重复报文,照样算成功
|
||||||
ApplyResult.Succeeded
|
ApplyResult.Succeeded
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -126,6 +130,7 @@ class AdftProcessor(
|
|||||||
private val procState: ProcStateRepository,
|
private val procState: ProcStateRepository,
|
||||||
operationDayProps: OperationDayProps,
|
operationDayProps: OperationDayProps,
|
||||||
private val mapper: ObjectMapper,
|
private val mapper: ObjectMapper,
|
||||||
|
private val clock: Clock,
|
||||||
) {
|
) {
|
||||||
private val opDay = OperationDayCalculator(
|
private val opDay = OperationDayCalculator(
|
||||||
zone = runCatching { ZoneId.of(operationDayProps.zone) }.getOrElse { ZoneId.of("Asia/Shanghai") },
|
zone = runCatching { ZoneId.of(operationDayProps.zone) }.getOrElse { ZoneId.of("Asia/Shanghai") },
|
||||||
@@ -133,47 +138,57 @@ class AdftProcessor(
|
|||||||
)
|
)
|
||||||
|
|
||||||
fun apply(head: ProcState, msg: DecodedMessage, record: ScheduleRecord): ApplyResult =
|
fun apply(head: ProcState, msg: DecodedMessage, record: ScheduleRecord): ApplyResult =
|
||||||
txManager.inTransaction {
|
try {
|
||||||
lock.lock()
|
txManager.inTransaction {
|
||||||
val main = flightState.findMainRow(record.flid)
|
lock.lock()
|
||||||
if (main != null && main.state == FlightState.DELETED) {
|
val main = flightState.findMainRow(record.flid)
|
||||||
// 已删除的航班重新激活:状态改回 ACTIVE、版本号加一,并登记状态事件
|
if (main != null && main.state == FlightState.DELETED) {
|
||||||
if (flightState.revive(record.flid, msgId = head.msgId, now = Instant.now())) {
|
// 已删除的航班重新激活:状态改回 ACTIVE、版本号加一,并登记状态事件
|
||||||
val current = flightState.loadFullSnapshot(record.flid)
|
if (flightState.revive(record.flid, msgId = head.msgId, now = clock.instant())) {
|
||||||
if (current != null) {
|
val current = flightState.loadFullSnapshot(record.flid)
|
||||||
val next = FlightStateEngine.mergedState(current, setOnly(record))
|
if (current != null) {
|
||||||
flightState.persistFullState(next, msgId = head.msgId, now = Instant.now())
|
val next = FlightStateEngine.mergedState(current, setOnly(record))
|
||||||
msgEvents.insertAll(eventsFor(next, mapper))
|
flightState.persistFullState(next, msgId = head.msgId, now = clock.instant())
|
||||||
|
msgEvents.insertAll(eventsFor(next, mapper))
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
// 并发下可能已被别的消息处理掉:不改版本、不重复发事件,但留下痕迹便于对账。
|
||||||
|
log.warn("adft revive no-op (already active or vanished) msgId={} flid={}", head.msgId, record.flid)
|
||||||
}
|
}
|
||||||
|
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED, now = clock.instant())
|
||||||
|
return@inTransaction ApplyResult.Succeeded
|
||||||
}
|
}
|
||||||
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED)
|
|
||||||
return@inTransaction ApplyResult.Succeeded
|
|
||||||
}
|
|
||||||
|
|
||||||
val current = flightState.loadFullSnapshot(record.flid)
|
val current = flightState.loadFullSnapshot(record.flid)
|
||||||
val next: FlightSnapshot = if (current == null) {
|
val next: FlightSnapshot = if (current == null) {
|
||||||
// 新建航班:带了计划时间就算出运营日,算不出来先留空,等日计划报文来收录
|
// 新建航班:带了计划时间就算出运营日,算不出来先留空,等日计划报文来收录
|
||||||
val day = opDay.compute(record.scalars["SODT"])
|
val day = opDay.compute(record.scalars["SODT"])
|
||||||
FlightSnapshot(
|
FlightSnapshot(
|
||||||
flid = record.flid,
|
flid = record.flid,
|
||||||
operationDay = day, // 算不出来就留空,不能默认拿收报当天顶上
|
operationDay = day, // 算不出来就留空,不能默认拿收报当天顶上
|
||||||
state = FlightState.ACTIVE,
|
state = FlightState.ACTIVE,
|
||||||
stateVersion = 1L,
|
stateVersion = 1L,
|
||||||
scalars = record.scalars,
|
scalars = record.scalars,
|
||||||
collections = FlightStateEngine.COLLECTION_KEYS.associateWith { key ->
|
collections = FlightStateEngine.COLLECTION_KEYS.associateWith { key ->
|
||||||
record.collections[key] ?: emptyList()
|
record.collections[key] ?: emptyList()
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
} else {
|
} else {
|
||||||
FlightStateEngine.mergedState(current, setOnly(record))
|
FlightStateEngine.mergedState(current, setOnly(record))
|
||||||
|
}
|
||||||
|
val outcome = flightState.persistFullState(next, msgId = head.msgId, now = clock.instant())
|
||||||
|
if (outcome == PersistOutcome.DAY_GUARD_VIOLATION) {
|
||||||
|
// 运营日冲突是**协议级**问题:与 SCHD 同分类 —— 整笔回滚、不重试、交人工。
|
||||||
|
// 若按 INFRA 抛出去,会被当成暂时性故障白白重试到耗尽,并给出误导的错误类别。
|
||||||
|
throw ProtocolViolation("operation-day guard violated flid=${record.flid}")
|
||||||
|
}
|
||||||
|
msgEvents.insertAll(eventsFor(next, mapper))
|
||||||
|
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED, now = clock.instant())
|
||||||
|
ApplyResult.Succeeded
|
||||||
}
|
}
|
||||||
val outcome = flightState.persistFullState(next, msgId = head.msgId, now = Instant.now())
|
} catch (e: ProtocolViolation) {
|
||||||
check(outcome != com.gzzn.omms.msgexchange.infra.persistence.PersistOutcome.DAY_GUARD_VIOLATION) {
|
log.error("adft DEAD(PROTOCOL) msgId={} reason={}", head.msgId, e.message)
|
||||||
"operation-day guard violated flid=${record.flid}"
|
ApplyResult.DeadProtocol(e.message ?: "protocol-violation")
|
||||||
}
|
|
||||||
msgEvents.insertAll(eventsFor(next, mapper))
|
|
||||||
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED)
|
|
||||||
ApplyResult.Succeeded
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -185,6 +200,10 @@ class AdftProcessor(
|
|||||||
scalars = record.scalars,
|
scalars = record.scalars,
|
||||||
collections = record.collections,
|
collections = record.collections,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
private companion object {
|
||||||
|
private val log = org.slf4j.LoggerFactory.getLogger(AdftProcessor::class.java)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// =====================================================================
|
// =====================================================================
|
||||||
|
|||||||
@@ -11,12 +11,14 @@ import com.gzzn.omms.msgexchange.domain.ProcState
|
|||||||
import com.gzzn.omms.msgexchange.domain.ProcStatus
|
import com.gzzn.omms.msgexchange.domain.ProcStatus
|
||||||
import com.gzzn.omms.msgexchange.infra.log.TraceLog
|
import com.gzzn.omms.msgexchange.infra.log.TraceLog
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.CminmsgInboxRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.CminmsgInboxRepository
|
||||||
|
import com.gzzn.omms.msgexchange.infra.persistence.InboxCursorRepository
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
||||||
import com.gzzn.omms.msgexchange.infra.retry.ProcFailure
|
import com.gzzn.omms.msgexchange.infra.retry.ProcFailure
|
||||||
import jakarta.inject.Singleton
|
import jakarta.inject.Singleton
|
||||||
import java.time.Clock
|
import java.time.Clock
|
||||||
import java.time.Duration
|
import java.time.Duration
|
||||||
import java.time.Instant
|
import java.time.Instant
|
||||||
|
import java.util.concurrent.atomic.AtomicLong
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 处理主泵:一个线程按消息 ID 从小到大一条条处理,保证先来的先处理。
|
* 处理主泵:一个线程按消息 ID 从小到大一条条处理,保证先来的先处理。
|
||||||
@@ -33,13 +35,16 @@ import java.time.Instant
|
|||||||
@Singleton
|
@Singleton
|
||||||
class Pump(
|
class Pump(
|
||||||
private val procState: ProcStateRepository,
|
private val procState: ProcStateRepository,
|
||||||
|
private val cursor: InboxCursorRepository,
|
||||||
private val processor: MessageProcessor,
|
private val processor: MessageProcessor,
|
||||||
private val backfill: BackfillService,
|
|
||||||
private val props: PipelineProps,
|
private val props: PipelineProps,
|
||||||
private val clock: Clock,
|
private val clock: Clock,
|
||||||
) {
|
) {
|
||||||
private val log = org.slf4j.LoggerFactory.getLogger(Pump::class.java)
|
private val log = org.slf4j.LoggerFactory.getLogger(Pump::class.java)
|
||||||
|
|
||||||
|
/** 连续失败计数:仅用于日志/排障,不代表业务状态。 */
|
||||||
|
private val tickFailures = AtomicLong(0)
|
||||||
|
|
||||||
@Volatile
|
@Volatile
|
||||||
private var running = true
|
private var running = true
|
||||||
|
|
||||||
@@ -52,11 +57,15 @@ class Pump(
|
|||||||
while (running) {
|
while (running) {
|
||||||
try {
|
try {
|
||||||
tick()
|
tick()
|
||||||
|
tickFailures.set(0)
|
||||||
} catch (e: InterruptedException) {
|
} catch (e: InterruptedException) {
|
||||||
Thread.currentThread().interrupt()
|
Thread.currentThread().interrupt()
|
||||||
return
|
return
|
||||||
} catch (e: Exception) {
|
} catch (e: Exception) {
|
||||||
// 兜底:单条消息的失败状态已在 processOne 内记录,这里只避免线程退出
|
// 兜底:单条消息的失败状态已在 processOne 内记录;但 tick 整体失败(DB 断连、
|
||||||
|
// 锁超时)在别处没有痕迹,必须留日志,否则排障无据可查。
|
||||||
|
val failures = tickFailures.incrementAndGet()
|
||||||
|
log.error("pump tick failed (failure #{})", failures, e)
|
||||||
sleepQuietly(props.pipeline.pollInterval)
|
sleepQuietly(props.pipeline.pollInterval)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -64,10 +73,25 @@ class Pump(
|
|||||||
|
|
||||||
internal fun tick() {
|
internal fun tick() {
|
||||||
val head = procState.headUnfinished()
|
val head = procState.headUnfinished()
|
||||||
|
if (head == null) {
|
||||||
|
sleepQuietly(props.pipeline.pollInterval)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
// 只领取"已被水位覆盖"的队头(`msgId <= W`)。
|
||||||
|
//
|
||||||
|
// 水位以内的行都是收报按 ID 顺序发现并登记的;水位之外的行只可能来自兼容入口
|
||||||
|
// 直接写 PROC_STATE(它不参与水位)。若允许领取,它就会越过那些尚未入队的较小 ID,
|
||||||
|
// 破坏 FIFO(缺口 G2)。这种行在空洞补齐、`W` 追平之后自然可领取。
|
||||||
|
val watermark = cursor.load().committedUpTo
|
||||||
|
if (head.msgId > watermark) {
|
||||||
|
warnBeyondWatermark(head.msgId, watermark)
|
||||||
|
sleepQuietly(props.pipeline.pollInterval)
|
||||||
|
return
|
||||||
|
}
|
||||||
when {
|
when {
|
||||||
head == null -> sleepQuietly(props.pipeline.pollInterval)
|
|
||||||
head.state == ProcStatus.FAILED && poisoned(head) -> {
|
head.state == ProcStatus.FAILED && poisoned(head) -> {
|
||||||
log.error("poison -> DEAD msgId={} attempts={} lastError={}", head.msgId, head.attempts, head.lastError)
|
log.error("poison -> DEAD msgId={} attempts={} lastError={}", head.msgId, head.attempts, head.lastError)
|
||||||
|
// markTerminal 在同一条 UPDATE 里登记回填意图;回填由扫描补写,不在这里做跨库写。
|
||||||
procState.markTerminal(
|
procState.markTerminal(
|
||||||
head.msgId, ProcStatus.DEAD,
|
head.msgId, ProcStatus.DEAD,
|
||||||
errorClass = ErrorClass.EXHAUSTED,
|
errorClass = ErrorClass.EXHAUSTED,
|
||||||
@@ -75,7 +99,6 @@ class Pump(
|
|||||||
attempts = head.attempts,
|
attempts = head.attempts,
|
||||||
now = clock.instant(),
|
now = clock.instant(),
|
||||||
)
|
)
|
||||||
backfill.attempt(head.msgId)
|
|
||||||
}
|
}
|
||||||
head.state == ProcStatus.FAILED && (head.nextAttemptAt ?: Instant.EPOCH) > clock.instant() ->
|
head.state == ProcStatus.FAILED && (head.nextAttemptAt ?: Instant.EPOCH) > clock.instant() ->
|
||||||
sleepQuietly(Duration.between(clock.instant(), head.nextAttemptAt))
|
sleepQuietly(Duration.between(clock.instant(), head.nextAttemptAt))
|
||||||
@@ -90,6 +113,19 @@ class Pump(
|
|||||||
private fun poisoned(head: ProcState): Boolean =
|
private fun poisoned(head: ProcState): Boolean =
|
||||||
isHeadPoisoned(head, clock.instant(), props)
|
isHeadPoisoned(head, clock.instant(), props)
|
||||||
|
|
||||||
|
/** 上一次"队头在水位之外"告警时的水位值:只在它变化时告警,避免每秒刷屏。 */
|
||||||
|
private val warnedWatermark = AtomicLong(Long.MIN_VALUE)
|
||||||
|
|
||||||
|
private fun warnBeyondWatermark(msgId: Long, watermark: Long) {
|
||||||
|
if (warnedWatermark.getAndSet(watermark) != watermark) {
|
||||||
|
log.warn(
|
||||||
|
"head msgId={} is beyond watermark W={}; waiting for discovery " +
|
||||||
|
"(row injected by the compat entry point?)",
|
||||||
|
msgId, watermark,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
private fun sleepQuietly(d: Duration) {
|
private fun sleepQuietly(d: Duration) {
|
||||||
if (!d.isNegative && !d.isZero) Thread.sleep(d.toMillis().coerceAtLeast(1))
|
if (!d.isNegative && !d.isZero) Thread.sleep(d.toMillis().coerceAtLeast(1))
|
||||||
}
|
}
|
||||||
@@ -100,10 +136,11 @@ internal fun isHeadPoisoned(head: ProcState, now: Instant, props: PipelineProps)
|
|||||||
Duration.between(head.processingStartedAt ?: head.updatedAt, now) >= props.pipeline.headDeadline
|
Duration.between(head.processingStartedAt ?: head.updatedAt, now) >= props.pipeline.headDeadline
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 处理一条消息:读原文 → 解码 → 绑定业务身份 → 分派给对应处理器 → 提交后回填标记。
|
* 处理一条消息:读原文 → 解码 → 绑定业务身份 → 分派给对应处理器。
|
||||||
*
|
*
|
||||||
* 业务数据和终态由各处理器在自己的事务里写入。终态一旦落下(成功、跳过或死信),
|
* 业务数据、终态与回填意图都由各处理器在自己的事务里写入(终态与回填意图是同一条 UPDATE)。
|
||||||
* 这里马上试一次把处理标记写回信箱;写不进去也没关系,回填扫描会按退避继续重试。
|
* **这里不做信箱回填**:主泵是 FIFO 关键路径,跨库写会把它绑在共享 MySQL 的可用性上。
|
||||||
|
* 回填由 `JobRunner` 定时的 `BackfillService.sweep` 驱动(调度周期不等于完成时限)。
|
||||||
*
|
*
|
||||||
* 任何意外异常都算在当前这条消息头上(记 FAILED(INFRA) 后重试),不会把主泵线程带崩。
|
* 任何意外异常都算在当前这条消息头上(记 FAILED(INFRA) 后重试),不会把主泵线程带崩。
|
||||||
* 报文非法和整包协议拒绝不重试,直接进死信等人工处置。
|
* 报文非法和整包协议拒绝不重试,直接进死信等人工处置。
|
||||||
@@ -118,14 +155,14 @@ class MessageProcessor(
|
|||||||
private val fdelProcessor: FdelProcessor,
|
private val fdelProcessor: FdelProcessor,
|
||||||
private val adftProcessor: AdftProcessor,
|
private val adftProcessor: AdftProcessor,
|
||||||
private val procFailure: ProcFailure,
|
private val procFailure: ProcFailure,
|
||||||
private val backfill: BackfillService,
|
|
||||||
private val props: PipelineProps,
|
private val props: PipelineProps,
|
||||||
|
private val clock: Clock,
|
||||||
) {
|
) {
|
||||||
private val log = org.slf4j.LoggerFactory.getLogger(MessageProcessor::class.java)
|
private val log = org.slf4j.LoggerFactory.getLogger(MessageProcessor::class.java)
|
||||||
|
|
||||||
fun processOne(head: ProcState) {
|
fun processOne(head: ProcState) {
|
||||||
TraceLog.withTrace(head.msgId) {
|
TraceLog.withTrace(head.msgId) {
|
||||||
val terminal = try {
|
try {
|
||||||
processInternal(head)
|
processInternal(head)
|
||||||
} catch (e: InterruptedException) {
|
} catch (e: InterruptedException) {
|
||||||
Thread.currentThread().interrupt()
|
Thread.currentThread().interrupt()
|
||||||
@@ -135,13 +172,10 @@ class MessageProcessor(
|
|||||||
log.warn("processOne unexpected failure msgId={} ec=INFRA msg={}", head.msgId, e.message ?: e.javaClass.simpleName)
|
log.warn("processOne unexpected failure msgId={} ec=INFRA msg={}", head.msgId, e.message ?: e.javaClass.simpleName)
|
||||||
procFailure.fail(head, ErrorClass.INFRA, e.message ?: e.javaClass.simpleName)
|
procFailure.fail(head, ErrorClass.INFRA, e.message ?: e.javaClass.simpleName)
|
||||||
}
|
}
|
||||||
// 终态已经写好了,马上试一次回填;写不进去也没关系,回填扫描会按退避继续重试
|
|
||||||
// (待办在写终态时就一起登记了)。还没处理完的消息不打标记。
|
|
||||||
if (terminal) backfill.attempt(head.msgId)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/** @return 这条消息是否已经落到终态(只有终态才允许回填信箱标记) */
|
/** @return 这条消息是否落到终态;回填由扫描驱动,调用方不再据此立即回填。 */
|
||||||
private fun processInternal(head: ProcState): Boolean {
|
private fun processInternal(head: ProcState): Boolean {
|
||||||
// 守卫:手工/遗留 FAILED 行若 attempts 已达上限,直接终态(防止退避到期后无限重试)
|
// 守卫:手工/遗留 FAILED 行若 attempts 已达上限,直接终态(防止退避到期后无限重试)
|
||||||
if (head.state == ProcStatus.FAILED && procFailure.scheduler.exhausted(head.attempts)) {
|
if (head.state == ProcStatus.FAILED && procFailure.scheduler.exhausted(head.attempts)) {
|
||||||
@@ -151,6 +185,7 @@ class MessageProcessor(
|
|||||||
errorClass = ErrorClass.EXHAUSTED,
|
errorClass = ErrorClass.EXHAUSTED,
|
||||||
lastError = head.lastError ?: "max-attempts",
|
lastError = head.lastError ?: "max-attempts",
|
||||||
attempts = head.attempts,
|
attempts = head.attempts,
|
||||||
|
now = clock.instant(),
|
||||||
)
|
)
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
@@ -179,7 +214,7 @@ class MessageProcessor(
|
|||||||
if (!procState.tryBindIdentity(head.msgId, identity)) {
|
if (!procState.tryBindIdentity(head.msgId, identity)) {
|
||||||
val owner = procState.ownerOfIdentity(identity) ?: -1L
|
val owner = procState.ownerOfIdentity(identity) ?: -1L
|
||||||
log.info("duplicate-of:{} -> SKIPPED msgId={}", owner, head.msgId)
|
log.info("duplicate-of:{} -> SKIPPED msgId={}", owner, head.msgId)
|
||||||
procState.markTerminal(head.msgId, ProcStatus.SKIPPED, lastError = "duplicate-of:$owner")
|
procState.markTerminal(head.msgId, ProcStatus.SKIPPED, lastError = "duplicate-of:$owner", now = clock.instant())
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -223,6 +258,7 @@ class MessageProcessor(
|
|||||||
head.msgId, ProcStatus.DEAD,
|
head.msgId, ProcStatus.DEAD,
|
||||||
errorClass = ErrorClass.PROTOCOL,
|
errorClass = ErrorClass.PROTOCOL,
|
||||||
lastError = result.reason.take(1000),
|
lastError = result.reason.take(1000),
|
||||||
|
now = clock.instant(),
|
||||||
)
|
)
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
@@ -233,7 +269,7 @@ class MessageProcessor(
|
|||||||
}
|
}
|
||||||
|
|
||||||
private fun deadMalformed(head: ProcState, detail: String): Boolean {
|
private fun deadMalformed(head: ProcState, detail: String): Boolean {
|
||||||
procState.markTerminal(head.msgId, ProcStatus.DEAD, errorClass = ErrorClass.MALFORMED, lastError = detail)
|
procState.markTerminal(head.msgId, ProcStatus.DEAD, errorClass = ErrorClass.MALFORMED, lastError = detail, now = clock.instant())
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -25,6 +25,7 @@ import com.gzzn.omms.msgexchange.infra.persistence.PipelineTransactionManager
|
|||||||
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.ProcStateRepository
|
||||||
import com.gzzn.omms.msgexchange.infra.persistence.SnapshotLogRepository
|
import com.gzzn.omms.msgexchange.infra.persistence.SnapshotLogRepository
|
||||||
import jakarta.inject.Singleton
|
import jakarta.inject.Singleton
|
||||||
|
import java.time.Clock
|
||||||
import java.time.Instant
|
import java.time.Instant
|
||||||
import java.time.LocalDate
|
import java.time.LocalDate
|
||||||
import java.time.ZoneId
|
import java.time.ZoneId
|
||||||
@@ -66,9 +67,11 @@ class ScheduleProcessor(
|
|||||||
private val snapshotLog: SnapshotLogRepository,
|
private val snapshotLog: SnapshotLogRepository,
|
||||||
operationDayProps: OperationDayProps,
|
operationDayProps: OperationDayProps,
|
||||||
private val mapper: ObjectMapper,
|
private val mapper: ObjectMapper,
|
||||||
|
private val clock: Clock,
|
||||||
) {
|
) {
|
||||||
|
private val zone: ZoneId = runCatching { ZoneId.of(operationDayProps.zone) }.getOrElse { ZoneId.of("Asia/Shanghai") }
|
||||||
private val opDay = OperationDayCalculator(
|
private val opDay = OperationDayCalculator(
|
||||||
zone = runCatching { ZoneId.of(operationDayProps.zone) }.getOrElse { ZoneId.of("Asia/Shanghai") },
|
zone = zone,
|
||||||
cutoffHour = operationDayProps.cutoffHour,
|
cutoffHour = operationDayProps.cutoffHour,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -98,7 +101,7 @@ class ScheduleProcessor(
|
|||||||
// 报文合法但没有记录:不写航班,但终态与回填意图仍在锁事务内一起提交
|
// 报文合法但没有记录:不写航班,但终态与回填意图仍在锁事务内一起提交
|
||||||
txManager.inTransaction {
|
txManager.inTransaction {
|
||||||
lock.lock()
|
lock.lock()
|
||||||
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED)
|
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED, now = clock.instant())
|
||||||
}
|
}
|
||||||
logSnapshot(head, body, SnapshotResult.COMMITTED, upserted = 0, setOf(SnapshotFlag.EMPTY), started)
|
logSnapshot(head, body, SnapshotResult.COMMITTED, upserted = 0, setOf(SnapshotFlag.EMPTY), started)
|
||||||
return ApplyResult.Succeeded
|
return ApplyResult.Succeeded
|
||||||
@@ -123,8 +126,10 @@ class ScheduleProcessor(
|
|||||||
|
|
||||||
var written = 0
|
var written = 0
|
||||||
val events = mutableListOf<MsgEvent>()
|
val events = mutableListOf<MsgEvent>()
|
||||||
|
// 一次建索引,避免在航班级循环里反复线性扫描(大日计划下是 O(N²))。
|
||||||
|
val recordsByFlid = body.records.associateBy { it.flid }
|
||||||
ok.perRecordDay.forEach { (flid, day) ->
|
ok.perRecordDay.forEach { (flid, day) ->
|
||||||
val record = body.records.first { it.flid == flid }
|
val record = recordsByFlid.getValue(flid)
|
||||||
val existingMain = mains[flid]
|
val existingMain = mains[flid]
|
||||||
val keepDeleted = existingMain?.state == FlightState.DELETED
|
val keepDeleted = existingMain?.state == FlightState.DELETED
|
||||||
if (keepDeleted) flags.add(SnapshotFlag.SCHD_REVIVE_CONFLICT) // 日计划不会让已删除的航班复活
|
if (keepDeleted) flags.add(SnapshotFlag.SCHD_REVIVE_CONFLICT) // 日计划不会让已删除的航班复活
|
||||||
@@ -135,7 +140,7 @@ class ScheduleProcessor(
|
|||||||
operationDay = day,
|
operationDay = day,
|
||||||
keepDeleted = keepDeleted,
|
keepDeleted = keepDeleted,
|
||||||
)
|
)
|
||||||
when (flightState.persistFullState(next, msgId = head.msgId, now = Instant.now())) {
|
when (flightState.persistFullState(next, msgId = head.msgId, now = clock.instant())) {
|
||||||
PersistOutcome.DAY_GUARD_VIOLATION ->
|
PersistOutcome.DAY_GUARD_VIOLATION ->
|
||||||
throw ProtocolViolation("operation-day guard violated flid=$flid")
|
throw ProtocolViolation("operation-day guard violated flid=$flid")
|
||||||
else -> written++
|
else -> written++
|
||||||
@@ -144,7 +149,7 @@ class ScheduleProcessor(
|
|||||||
}
|
}
|
||||||
if (events.isNotEmpty()) msgEvents.insertAll(events)
|
if (events.isNotEmpty()) msgEvents.insertAll(events)
|
||||||
// 终态与回填待办跟业务数据同事务提交:要么全成,要么全回滚
|
// 终态与回填待办跟业务数据同事务提交:要么全成,要么全回滚
|
||||||
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED)
|
procState.markTerminal(head.msgId, ProcStatus.SUCCEEDED, now = clock.instant())
|
||||||
written
|
written
|
||||||
}
|
}
|
||||||
logSnapshot(head, body, SnapshotResult.COMMITTED, upserted, flags, started)
|
logSnapshot(head, body, SnapshotResult.COMMITTED, upserted, flags, started)
|
||||||
@@ -192,9 +197,9 @@ class ScheduleProcessor(
|
|||||||
snapshotLog.append(
|
snapshotLog.append(
|
||||||
SnapshotLogEntry(
|
SnapshotLogEntry(
|
||||||
msgId = head.msgId,
|
msgId = head.msgId,
|
||||||
recvAt = Instant.now(),
|
recvAt = clock.instant(),
|
||||||
scopeStart = days.minOrNull() ?: LocalDate.now(),
|
scopeStart = days.minOrNull() ?: LocalDate.now(zone),
|
||||||
scopeEnd = days.maxOrNull() ?: LocalDate.now(),
|
scopeEnd = days.maxOrNull() ?: LocalDate.now(zone),
|
||||||
recs = body.records.size,
|
recs = body.records.size,
|
||||||
upserted = upserted,
|
upserted = upserted,
|
||||||
durationMs = (System.nanoTime() - startedNanos) / 1_000_000,
|
durationMs = (System.nanoTime() - startedNanos) / 1_000_000,
|
||||||
|
|||||||
@@ -16,12 +16,17 @@ msgx:
|
|||||||
poll-interval: 1s # KEEP 现役 1s 轮询节奏
|
poll-interval: 1s # KEEP 现役 1s 轮询节奏
|
||||||
claim-batch: 50
|
claim-batch: 50
|
||||||
max-attempts: 5 # 处理/投递同值
|
max-attempts: 5 # 处理/投递同值
|
||||||
backoff-ms: [1000, 2000, 4000, 8000, 16000] # 指数退避,单次封顶 60s
|
backoff-ms: [1000, 2000, 4000, 8000] # 指数退避;档位数必须 = max-attempts - 1(启动自检)
|
||||||
backoff-cap-ms: 60000
|
backoff-cap-ms: 60000
|
||||||
head-deadline: 10m # 队头滞留上界 = 最坏 HOL 时长(毒丸升级)
|
head-deadline: 10m # 队头滞留上界 = 最坏 HOL 时长(毒丸升级)
|
||||||
max-commit-delay: 5m # §5.1 空洞老化:W+1 空洞超过该时延判定为永久(Q2 最大提交时延)
|
max-commit-delay: 5m # §5.1 空洞老化:W+1 空洞超过该时延判定为永久(Q2 最大提交时延)
|
||||||
overdue-backfill: 30d # §5.2 超期补写期限 R:≥ 人工重放期限 + 人工处置期限(Q6)
|
overdue-backfill: 30d # §5.2 超期补写期限 R:≥ 人工重放期限 + 人工处置期限(Q6)
|
||||||
backfill-batch: 100 # 回填扫描单批条数
|
backfill-batch: 100 # 回填扫描单批条数
|
||||||
|
backfill-max-attempts: 100 # 回填自动重试上限;达上限停止自动重试(可人工恢复),避免占满批次
|
||||||
|
# 一次性切流播种:默认(注释掉)不播种。min=读现存全部 | zero=从 0 按空洞规则 | max=跳过可见存量 | <id>
|
||||||
|
# cutover-watermark: min
|
||||||
|
late-detect-period: 60s # 迟到到达检测(只读,阶段 0):复查被放行的空洞 ID 是否后来真的出现;0=关闭
|
||||||
|
late-detect-batch: 200 # 每轮最多复查多少个空洞 ID
|
||||||
autostart: false # U07:启动即拉起 Pump/Dispatcher 循环;需真实仓储或 msgx.stubs=true 才开启(dev 见 application-dev.yml)
|
autostart: false # U07:启动即拉起 Pump/Dispatcher 循环;需真实仓储或 msgx.stubs=true 才开启(dev 见 application-dev.yml)
|
||||||
schd:
|
schd:
|
||||||
flush-period: 3s # KEEP 现役推送节律
|
flush-period: 3s # KEEP 现役推送节律
|
||||||
@@ -55,6 +60,16 @@ datasources:
|
|||||||
username: ${MSGX_PG_USER}
|
username: ${MSGX_PG_USER}
|
||||||
password: ${MSGX_PG_PASSWORD}
|
password: ${MSGX_PG_PASSWORD}
|
||||||
driver-class-name: org.postgresql.Driver
|
driver-class-name: org.postgresql.Driver
|
||||||
|
# 有界外部调用:取连接/校验/空闲/寿命都有上限,避免线程无界等待。
|
||||||
|
# 注意 Micronaut 的 Hikari 配置项是**毫秒数(long)**,不是 Duration 字面量。
|
||||||
|
connection-timeout: 5000
|
||||||
|
validation-timeout: 3000
|
||||||
|
idle-timeout: 300000
|
||||||
|
max-lifetime: 1800000
|
||||||
|
# 驱动级超时(秒):connectTimeout/socketTimeout,防止网络黑洞挂住调用线程。
|
||||||
|
data-source-properties:
|
||||||
|
connectTimeout: 3
|
||||||
|
socketTimeout: 30
|
||||||
|
|
||||||
flyway:
|
flyway:
|
||||||
datasources:
|
datasources:
|
||||||
@@ -73,6 +88,11 @@ mailbox:
|
|||||||
username: ${MSGX_MAILBOX_USER}
|
username: ${MSGX_MAILBOX_USER}
|
||||||
password: ${MSGX_MAILBOX_PASSWORD}
|
password: ${MSGX_MAILBOX_PASSWORD}
|
||||||
driver-class-name: com.mysql.cj.jdbc.Driver
|
driver-class-name: com.mysql.cj.jdbc.Driver
|
||||||
|
# 有界外部调用(必填语义,默认即有限):Connector/J 默认 socketTimeout=0 = 无限等待
|
||||||
|
connect-timeout-ms: 3000
|
||||||
|
socket-timeout-ms: 30000
|
||||||
|
pool-connection-timeout-ms: 5000
|
||||||
|
pool-validation-timeout-ms: 3000
|
||||||
|
|
||||||
kafka:
|
kafka:
|
||||||
bootstrap:
|
bootstrap:
|
||||||
|
|||||||
@@ -0,0 +1,33 @@
|
|||||||
|
-- =====================================================================
|
||||||
|
-- V4:回填闭环
|
||||||
|
-- ---------------------------------------------------------------------
|
||||||
|
-- 只动自有 PostgreSQL;共享 MySQL 不建表、不改结构。
|
||||||
|
--
|
||||||
|
-- 这次迁移解决两个问题:
|
||||||
|
--
|
||||||
|
-- 1) "放弃回填"需要显式语义(新增 BACKFILL_ABANDONED_AT / BACKFILL_ABANDONED_REASON)
|
||||||
|
-- 原先"信箱行不存在"(MISSING)被当成可重试失败:记录会每 30 秒重试且永不收敛,
|
||||||
|
-- unmarkedTerminal 指标只涨不落。现在把两类情况分开:
|
||||||
|
-- · 运行时查询确认行不存在(确定性结论,重试不改变结果)→ 立即放弃自动重试;
|
||||||
|
-- · 超时/连接失败(暂时性)→ 继续按退避重试,达到上限后放弃自动重试。
|
||||||
|
-- **放弃 ≠ 标记已确认**:BACKFILL_AT 仍为空,因此不满足"边界内全部行已打标"的
|
||||||
|
-- 清除前提,库方不应据此清除。放弃行留有原因字段,并支持人工恢复(清标记后重排一次)。
|
||||||
|
--
|
||||||
|
-- 2) 扫描公平性(重建索引)
|
||||||
|
-- 原扫描按 MSG_ID 升序取批:最旧的一批永久失败行会持续占满批次,后面的记录永远
|
||||||
|
-- 轮不到(全局回填饥饿)。现在按 (BACKFILL_ATTEMPTS, MSG_ID) 轮转,并把已放弃的
|
||||||
|
-- 行排除在扫描之外,保证新行一定能拿到名额。
|
||||||
|
--
|
||||||
|
-- 明确不做:**不按年龄做任何存量推断**。"终态 + 从未尝试 + 接收时间早于保留窗"不能
|
||||||
|
-- 证明信箱行已删除,V2 用 UPDATED_AT 兜底也不构成删除证据;按年龄批量置 abandoned
|
||||||
|
-- 会把仍存在且未标记的旧行永久排除出扫描,反而阻断"边界内全部行已打标"。
|
||||||
|
-- 存量行一律保留原语义,交由运行时 MISSING 或人工对账处置。
|
||||||
|
-- =====================================================================
|
||||||
|
|
||||||
|
ALTER TABLE PROC_STATE ADD COLUMN BACKFILL_ABANDONED_AT TIMESTAMP(6) WITH TIME ZONE;
|
||||||
|
ALTER TABLE PROC_STATE ADD COLUMN BACKFILL_ABANDONED_REASON VARCHAR(64);
|
||||||
|
|
||||||
|
-- 回填扫描:只覆盖仍需自动回填的行,并支持 (attempts, msg_id) 的公平轮转顺序。
|
||||||
|
DROP INDEX IF EXISTS idx_proc_backfill_due;
|
||||||
|
CREATE INDEX idx_proc_backfill_due ON PROC_STATE (BACKFILL_ATTEMPTS, MSG_ID)
|
||||||
|
WHERE BACKFILL_AT IS NULL AND BACKFILL_ABANDONED_AT IS NULL;
|
||||||
@@ -0,0 +1,21 @@
|
|||||||
|
-- =====================================================================
|
||||||
|
-- V5:切流水位播种(一次性、显式)
|
||||||
|
-- ---------------------------------------------------------------------
|
||||||
|
-- 只动自有 PostgreSQL;共享 MySQL 不建表、不改结构。
|
||||||
|
--
|
||||||
|
-- 背景:`max-commit-delay` 的空洞老化只在"水位后面出现缺号"时起作用。若第一次对着一个
|
||||||
|
-- 已有数据的信箱启动(尤其最老分区已被 DROP、MIN(ID) 远大于 1),水位从 0 起会把 ID=1
|
||||||
|
-- 判成空洞,白等一个老化窗口才前进;而 W=0 又意味着会把保留期内全部存量重新入队。
|
||||||
|
-- 两种后果都不是代码能替业务决定的,因此这里只提供**显式的、一次性的**播种机制。
|
||||||
|
--
|
||||||
|
-- 语义:
|
||||||
|
-- · 默认(未配置 msgx.pipeline.cutover-watermark)不播种,保持既有行为;
|
||||||
|
-- · 四种模式严格区分:min = W:MIN(ID)-1(读当前全部现存行)、zero = W:0(按空洞规则从 0 扫)、
|
||||||
|
-- max = W:MAX(ID)(跳过当前可见存量)、<id> = 显式边界;
|
||||||
|
-- · 代码**不做默认选择**,也不会自动退化成 max;
|
||||||
|
-- · 升级实例(已有水位或已有处理记录)**拒绝重新播种**,重新切流必须是显式操作;
|
||||||
|
-- · SEEDED_AT 记录"已播种"这一事实;注意它**为 NULL 不等于"从未消费"**——
|
||||||
|
-- 已有库新增列后同样是 NULL,因此判据必须叠加"确实没有消费过"。
|
||||||
|
-- =====================================================================
|
||||||
|
|
||||||
|
ALTER TABLE INBOX_CURSOR ADD COLUMN SEEDED_AT TIMESTAMP(6) WITH TIME ZONE;
|
||||||
@@ -14,9 +14,15 @@ import io.micronaut.context.ApplicationContext
|
|||||||
import io.micronaut.test.extensions.junit5.annotation.MicronautTest
|
import io.micronaut.test.extensions.junit5.annotation.MicronautTest
|
||||||
import jakarta.inject.Inject
|
import jakarta.inject.Inject
|
||||||
import org.junit.jupiter.api.Assertions.assertEquals
|
import org.junit.jupiter.api.Assertions.assertEquals
|
||||||
|
import org.junit.jupiter.api.Assertions.assertFalse
|
||||||
import org.junit.jupiter.api.Assertions.assertNotNull
|
import org.junit.jupiter.api.Assertions.assertNotNull
|
||||||
|
import org.junit.jupiter.api.Assertions.assertNull
|
||||||
|
import org.junit.jupiter.api.Assertions.assertSame
|
||||||
import org.junit.jupiter.api.Assertions.assertTrue
|
import org.junit.jupiter.api.Assertions.assertTrue
|
||||||
import org.junit.jupiter.api.Test
|
import org.junit.jupiter.api.Test
|
||||||
|
import java.time.Instant
|
||||||
|
import java.util.concurrent.CountDownLatch
|
||||||
|
import java.util.concurrent.TimeUnit
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 不接任何外部中间件,用内存实现把整条链跑通:
|
* 不接任何外部中间件,用内存实现把整条链跑通:
|
||||||
@@ -59,6 +65,18 @@ class PipelineSmokeTest {
|
|||||||
assertNotNull(controller)
|
assertNotNull(controller)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 装配不变量:回填与人工重放必须共用**同一个** [com.gzzn.omms.msgexchange.processing.MessageLifecycleGate]。
|
||||||
|
* 若两者各持一把锁,互斥失效,"旧回填给已重新入队的消息写标记"的窗口会重新打开。
|
||||||
|
* 这条断言取代了以前"靠 Kotlin 默认参数值兜底"的隐患。
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
fun `backfill and replay share the same lifecycle gate`() {
|
||||||
|
val backfill = ctx.getBean(com.gzzn.omms.msgexchange.processing.BackfillService::class.java)
|
||||||
|
val replay = ctx.getBean(com.gzzn.omms.msgexchange.infra.retry.ReplayService::class.java)
|
||||||
|
assertSame(backfill.lifecycleGate, replay.lifecycleGate)
|
||||||
|
}
|
||||||
|
|
||||||
companion object {
|
companion object {
|
||||||
/** 报文头合法,但类型没有对应处理器,会走"暂不支持、先重试"这条路 */
|
/** 报文头合法,但类型没有对应处理器,会走"暂不支持、先重试"这条路 */
|
||||||
val UNSUPPORTED_XML = """
|
val UNSUPPORTED_XML = """
|
||||||
@@ -73,6 +91,8 @@ class PipelineSmokeTest {
|
|||||||
val receipt = controller.send(UNSUPPORTED_XML)
|
val receipt = controller.send(UNSUPPORTED_XML)
|
||||||
assertNotNull(receipt.body()) // 受理 ID
|
assertNotNull(receipt.body()) // 受理 ID
|
||||||
val id = receipt.body()!!.toLong()
|
val id = receipt.body()!!.toLong()
|
||||||
|
// 兼容入口只保证"已落信 + 已入队",**不推进水位**;必须先被收报发现(W 追平)才可领取。
|
||||||
|
ctx.getBean(com.gzzn.omms.msgexchange.ingress.InboxPoller::class.java).pollOnce()
|
||||||
|
|
||||||
pump.tick() // 解码成功但无 DELY Handler → FAILED(UNSUPPORTED)
|
pump.tick() // 解码成功但无 DELY Handler → FAILED(UNSUPPORTED)
|
||||||
|
|
||||||
@@ -89,6 +109,7 @@ class PipelineSmokeTest {
|
|||||||
fun `replay reopens failed UNSUPPORTED row to PENDING`() {
|
fun `replay reopens failed UNSUPPORTED row to PENDING`() {
|
||||||
val receipt = controller.send(UNSUPPORTED_XML)
|
val receipt = controller.send(UNSUPPORTED_XML)
|
||||||
val id = receipt.body()!!.toLong()
|
val id = receipt.body()!!.toLong()
|
||||||
|
ctx.getBean(com.gzzn.omms.msgexchange.ingress.InboxPoller::class.java).pollOnce()
|
||||||
pump.tick()
|
pump.tick()
|
||||||
val stub = ctx.getBean(StubProcState::class.java)
|
val stub = ctx.getBean(StubProcState::class.java)
|
||||||
assertEquals(ProcStatus.FAILED, stub.snapshotOf(id)!!.state)
|
assertEquals(ProcStatus.FAILED, stub.snapshotOf(id)!!.state)
|
||||||
@@ -103,22 +124,31 @@ class PipelineSmokeTest {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 回归用例:死信在进入终态时同时记下回填待办,并马上把标记写回信箱。
|
* 回归用例:死信在进入终态时同时记下回填意图(同一条 UPDATE),由回填扫描把标记写回信箱。
|
||||||
* 少了这一步,这些行永远占着每批的名额,攒够一批就再也发现不了新消息了。
|
* 少了这一步,这些行永远占着每批的名额,攒够一批就再也发现不了新消息了。
|
||||||
|
*
|
||||||
|
* 注意:主泵**不再**在终态后立即回填(跨库写不能占用 FIFO 关键路径),
|
||||||
|
* 因此这里显式触发一次扫描来验证回填链路。
|
||||||
*/
|
*/
|
||||||
@Test
|
@Test
|
||||||
fun `dead letter reaches terminal state, gets marked and cannot block later discovery`() {
|
fun `dead letter reaches terminal state, gets marked by sweep and cannot block later discovery`() {
|
||||||
val inbox = ctx.getBean(com.gzzn.omms.msgexchange.infra.stub.StubInbox::class.java)
|
val inbox = ctx.getBean(com.gzzn.omms.msgexchange.infra.stub.StubInbox::class.java)
|
||||||
val proc = ctx.getBean(StubProcState::class.java)
|
val proc = ctx.getBean(StubProcState::class.java)
|
||||||
val dead = inbox.simulateExternalWrite("<MSG/>")
|
val dead = inbox.simulateExternalWrite("<MSG/>")
|
||||||
ctx.getBean(com.gzzn.omms.msgexchange.ingress.InboxPoller::class.java).pollOnce()
|
ctx.getBean(com.gzzn.omms.msgexchange.ingress.InboxPoller::class.java).pollOnce()
|
||||||
|
|
||||||
pump.tick() // 解码 MALFORMED → DEAD + 回填意图(同一条 UPDATE)→ 提交后立即回填
|
pump.tick() // 解码 MALFORMED → DEAD + 回填意图(同一条 UPDATE)
|
||||||
|
|
||||||
val row = proc.find(dead)!!
|
val row = proc.find(dead)!!
|
||||||
assertEquals(ProcStatus.DEAD, row.state)
|
assertEquals(ProcStatus.DEAD, row.state)
|
||||||
assertEquals(ErrorClass.MALFORMED, row.errorClass)
|
assertEquals(ErrorClass.MALFORMED, row.errorClass)
|
||||||
assertNotNull(row.backfillAt)
|
assertNotNull(row.backfillNextAt) // 终态已登记回填意图
|
||||||
|
assertFalse(inbox.isMarked(dead)) // 主泵不再内联回填
|
||||||
|
assertNull(row.backfillAt)
|
||||||
|
|
||||||
|
ctx.getBean(com.gzzn.omms.msgexchange.processing.BackfillService::class.java).sweep()
|
||||||
|
|
||||||
|
assertNotNull(proc.find(dead)!!.backfillAt)
|
||||||
assertTrue(inbox.isMarked(dead))
|
assertTrue(inbox.isMarked(dead))
|
||||||
|
|
||||||
val fresh = inbox.simulateExternalWrite("<MSG/>")
|
val fresh = inbox.simulateExternalWrite("<MSG/>")
|
||||||
@@ -126,6 +156,105 @@ class PipelineSmokeTest {
|
|||||||
assertNotNull(proc.find(fresh)) // 死信不阻断后续发现
|
assertNotNull(proc.find(fresh)) // 死信不阻断后续发现
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 不变量:非业务型终态(MALFORMED / PROTOCOL / SKIPPED / EXHAUSTED)**不触碰航班表、也不登记待发事件**——
|
||||||
|
* 它们只写 `PROC_STATE` 一条记录(终态与回填意图是同一条 UPDATE)。
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
fun `non-business terminal states touch neither flight tables nor outbox`() {
|
||||||
|
val inbox = ctx.getBean(com.gzzn.omms.msgexchange.infra.stub.StubInbox::class.java)
|
||||||
|
val proc = ctx.getBean(StubProcState::class.java)
|
||||||
|
val flights = ctx.getBean(com.gzzn.omms.msgexchange.infra.stub.StubFlightState::class.java)
|
||||||
|
val events = ctx.getBean(StubMsgEvents::class.java)
|
||||||
|
val id = inbox.simulateExternalWrite("<MSG/>") // 非法报文(无 META)→ MALFORMED
|
||||||
|
ctx.getBean(com.gzzn.omms.msgexchange.ingress.InboxPoller::class.java).pollOnce()
|
||||||
|
|
||||||
|
pump.tick()
|
||||||
|
|
||||||
|
assertEquals(ProcStatus.DEAD, proc.find(id)!!.state)
|
||||||
|
assertEquals(ErrorClass.MALFORMED, proc.find(id)!!.errorClass)
|
||||||
|
assertTrue(flights.mains.isEmpty()) // 没碰航班表
|
||||||
|
assertTrue(events.rows.isEmpty()) // 没登记待发事件
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 【缺口基线 · G5】毒丸升级路径**不在** `MessageLifecycleGate` 内:即使门被别的线程持有,
|
||||||
|
* 它也会照常把队头置为 `DEAD`。这不是期望行为(与人工重放并发时存在窗口),
|
||||||
|
* 而是当前实现的已知缺口;把门覆盖到毒丸路径后,本用例必须反转成"先等待门"。
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
fun `baseline - poison escalation writes DEAD without taking the lifecycle gate`() {
|
||||||
|
val proc = ctx.getBean(StubProcState::class.java)
|
||||||
|
val props = ctx.getBean(com.gzzn.omms.msgexchange.config.PipelineProps::class.java)
|
||||||
|
val backfill = ctx.getBean(com.gzzn.omms.msgexchange.processing.BackfillService::class.java)
|
||||||
|
// 队头必须在水位以内才可领取:直接播种 PROC_STATE 的用例要显式把水位推上去。
|
||||||
|
ctx.getBean(com.gzzn.omms.msgexchange.infra.stub.StubInboxCursor::class.java)
|
||||||
|
.save(com.gzzn.omms.msgexchange.infra.persistence.InboxCursorRepository.Cursor(committedUpTo = 9001L))
|
||||||
|
proc.insertIfAbsent(9001L, Instant.now())
|
||||||
|
proc.update(
|
||||||
|
9001L, ProcStatus.FAILED,
|
||||||
|
attempts = props.pipeline.maxAttempts,
|
||||||
|
lastError = "boom",
|
||||||
|
)
|
||||||
|
|
||||||
|
val held = CountDownLatch(1)
|
||||||
|
val release = CountDownLatch(1)
|
||||||
|
val holder = Thread.ofPlatform().daemon(true).start {
|
||||||
|
backfill.lifecycleGate.exclusive {
|
||||||
|
held.countDown()
|
||||||
|
release.await()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
assertTrue(held.await(1, TimeUnit.SECONDS), "持有者必须先拿到 gate")
|
||||||
|
val ticker = Thread.ofPlatform().start { pump.tick() }
|
||||||
|
ticker.join(2_000)
|
||||||
|
assertFalse(ticker.isAlive, "毒丸路径不应等待 lifecycle gate(已知 G5 缺口)")
|
||||||
|
assertEquals(ProcStatus.DEAD, proc.find(9001L)!!.state)
|
||||||
|
assertEquals(ErrorClass.EXHAUSTED, proc.find(9001L)!!.errorClass)
|
||||||
|
} finally {
|
||||||
|
release.countDown()
|
||||||
|
holder.join(1_000)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* FIFO 修复(G2)端到端验收:兼容入口写入的高 ID **不会被提前领取**,
|
||||||
|
* 必须等水位追平(较小 ID 补齐并入队)之后才按顺序处理。
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
fun `compat injected high id is not claimed until the watermark catches up`() {
|
||||||
|
val inbox = ctx.getBean(com.gzzn.omms.msgexchange.infra.stub.StubInbox::class.java)
|
||||||
|
val proc = ctx.getBean(StubProcState::class.java)
|
||||||
|
val cursor = ctx.getBean(com.gzzn.omms.msgexchange.infra.stub.StubInboxCursor::class.java)
|
||||||
|
val poller = ctx.getBean(com.gzzn.omms.msgexchange.ingress.InboxPoller::class.java)
|
||||||
|
|
||||||
|
// 信箱:1 存在、2 是空洞(被删)、3 存在 → 水位停在 1
|
||||||
|
inbox.simulateExternalWrite("<MSG/>")
|
||||||
|
val hole = inbox.simulateExternalWrite("<MSG/>")
|
||||||
|
inbox.simulateExternalWrite("<MSG/>")
|
||||||
|
inbox.removeRow(hole)
|
||||||
|
poller.pollOnce(Instant.now())
|
||||||
|
assertEquals(1L, cursor.cursor.committedUpTo)
|
||||||
|
|
||||||
|
// 兼容入口写入高 ID:直接进 PG,但水位没追平(这正是原来的越序路径)
|
||||||
|
val high = controller.send(UNSUPPORTED_XML).body()!!.toLong()
|
||||||
|
assertTrue(high > 3L)
|
||||||
|
|
||||||
|
pump.tick() // 先按 FIFO 处理 ID=1
|
||||||
|
assertEquals(ProcStatus.DEAD, proc.find(1L)!!.state)
|
||||||
|
pump.tick() // 队头变成高 ID,但它在水位之外 → 不领取
|
||||||
|
assertEquals(ProcStatus.PENDING, proc.find(high)!!.state)
|
||||||
|
|
||||||
|
// 空洞补齐 → 水位追平 → 才允许继续按顺序处理
|
||||||
|
inbox.restoreRow(hole, "<MSG/>", Instant.now())
|
||||||
|
poller.pollOnce(Instant.now())
|
||||||
|
assertTrue(cursor.cursor.committedUpTo >= high)
|
||||||
|
repeat(3) { pump.tick() } // 依次处理 2、3、高 ID
|
||||||
|
assertEquals(ProcStatus.FAILED, proc.find(high)!!.state)
|
||||||
|
assertEquals(ErrorClass.UNSUPPORTED, proc.find(high)!!.errorClass)
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
fun `dispatcher flushes schd batch through stub port`() {
|
fun `dispatcher flushes schd batch through stub port`() {
|
||||||
val events = ctx.getBean(StubMsgEvents::class.java)
|
val events = ctx.getBean(StubMsgEvents::class.java)
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
package com.gzzn.omms.msgexchange.config
|
package com.gzzn.omms.msgexchange.config
|
||||||
|
|
||||||
import org.junit.jupiter.api.Assertions.assertEquals
|
import org.junit.jupiter.api.Assertions.assertEquals
|
||||||
|
import org.junit.jupiter.api.Assertions.assertThrows
|
||||||
import org.junit.jupiter.api.Test
|
import org.junit.jupiter.api.Test
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -15,11 +16,36 @@ class PipelinePropsTest {
|
|||||||
fun `backoff follows table then caps`() {
|
fun `backoff follows table then caps`() {
|
||||||
assertEquals(1000, pipeline.backoffFor(1))
|
assertEquals(1000, pipeline.backoffFor(1))
|
||||||
assertEquals(2000, pipeline.backoffFor(2))
|
assertEquals(2000, pipeline.backoffFor(2))
|
||||||
assertEquals(16000, pipeline.backoffFor(5))
|
assertEquals(8000, pipeline.backoffFor(4)) // 最后一档:max-attempts - 1 = 4
|
||||||
assertEquals(60_000, pipeline.backoffFor(6)) // 表外 → 封顶
|
assertEquals(60_000, pipeline.backoffFor(5)) // 表外 → 封顶
|
||||||
assertEquals(60_000, pipeline.backoffFor(99))
|
assertEquals(60_000, pipeline.backoffFor(99))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 配置自检:档位数必须等于 max-attempts − 1,否则有一段退避**永远走不到**
|
||||||
|
* (这正是默认配置曾经的错误:5 次尝试配了 5 档,16 秒档不可达)。
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
fun `validate rejects a backoff table whose size does not match max-attempts`() {
|
||||||
|
pipeline.validate() // 默认配置必须自洽
|
||||||
|
|
||||||
|
pipeline.backoffMs = listOf(1000, 2000, 4000, 8000, 16000)
|
||||||
|
assertThrows(IllegalArgumentException::class.java) { pipeline.validate() }
|
||||||
|
}
|
||||||
|
|
||||||
|
/** 切流播种只接受 min|zero|max|<id>:非法值必须在启动时挡掉。 */
|
||||||
|
@Test
|
||||||
|
fun `validate rejects an unknown cutover watermark mode`() {
|
||||||
|
pipeline.cutoverWatermark = "bogus"
|
||||||
|
assertThrows(IllegalArgumentException::class.java) { pipeline.validate() }
|
||||||
|
|
||||||
|
pipeline.cutoverWatermark = "max"
|
||||||
|
pipeline.validate()
|
||||||
|
|
||||||
|
pipeline.cutoverWatermark = "12345"
|
||||||
|
pipeline.validate()
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
fun `non-positive attempt never throws and falls back to first slot`() {
|
fun `non-positive attempt never throws and falls back to first slot`() {
|
||||||
assertEquals(1000, pipeline.backoffFor(0))
|
assertEquals(1000, pipeline.backoffFor(0))
|
||||||
|
|||||||
@@ -48,7 +48,7 @@ class HealthIndicatorsTest {
|
|||||||
val oldest = MutableClock.BASE.minusSeconds(3600)
|
val oldest = MutableClock.BASE.minusSeconds(3600)
|
||||||
proc.insertIfAbsent(1L, oldest)
|
proc.insertIfAbsent(1L, oldest)
|
||||||
proc.insertIfAbsent(2L, MutableClock.BASE)
|
proc.insertIfAbsent(2L, MutableClock.BASE)
|
||||||
proc.markTerminal(2L, ProcStatus.SUCCEEDED)
|
proc.markTerminal(2L, ProcStatus.SUCCEEDED, now = MutableClock.BASE)
|
||||||
val newest = inbox.insertRaw("<MSG/>")
|
val newest = inbox.insertRaw("<MSG/>")
|
||||||
cursor.save(InboxCursorRepository.Cursor(committedUpTo = newest - 2))
|
cursor.save(InboxCursorRepository.Cursor(committedUpTo = newest - 2))
|
||||||
|
|
||||||
@@ -65,7 +65,7 @@ class HealthIndicatorsTest {
|
|||||||
|
|
||||||
@Test
|
@Test
|
||||||
fun `inbox lifecycle stays up when ports are not bound`() {
|
fun `inbox lifecycle stays up when ports are not bound`() {
|
||||||
val result = lifecycleHealth(procState = null, cursor = null, mailbox = null)
|
val result = lifecycleHealth(procState = null, cursor = null, mailbox = null, now = MutableClock.BASE)
|
||||||
|
|
||||||
assertEquals(HealthStatus.UP, result.status) // 可用性由依赖自身指示器承担
|
assertEquals(HealthStatus.UP, result.status) // 可用性由依赖自身指示器承担
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,63 @@
|
|||||||
|
package com.gzzn.omms.msgexchange.infra.metrics
|
||||||
|
|
||||||
|
import com.gzzn.omms.msgexchange.domain.ProcStatus
|
||||||
|
import com.gzzn.omms.msgexchange.infra.stub.StubProcState
|
||||||
|
import io.micrometer.core.instrument.MeterRegistry
|
||||||
|
import io.micronaut.context.ApplicationContext
|
||||||
|
import io.micronaut.context.annotation.Property
|
||||||
|
import io.micronaut.test.extensions.junit5.annotation.MicronautTest
|
||||||
|
import jakarta.inject.Inject
|
||||||
|
import org.junit.jupiter.api.Assertions.assertEquals
|
||||||
|
import org.junit.jupiter.api.BeforeEach
|
||||||
|
import org.junit.jupiter.api.Test
|
||||||
|
import java.time.Instant
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 指标接入回归:管道关键量必须真的注册进 `MeterRegistry` 且取数正确。
|
||||||
|
* 抓取端点由 Micronaut Micrometer 提供,这里只验证"注册 + 取值"。
|
||||||
|
*
|
||||||
|
* 缓存窗口在测试里置 0,保证断言看到的是当前 stub 状态而不是缓存快照。
|
||||||
|
*/
|
||||||
|
@MicronautTest
|
||||||
|
@Property(name = "msgx.health.backlog-cache-ttl-ms", value = "0")
|
||||||
|
class PipelineMetricsTest {
|
||||||
|
|
||||||
|
@Inject
|
||||||
|
lateinit var ctx: ApplicationContext
|
||||||
|
|
||||||
|
@Inject
|
||||||
|
lateinit var registry: MeterRegistry
|
||||||
|
|
||||||
|
@BeforeEach
|
||||||
|
fun clean() {
|
||||||
|
ctx.getBean(StubProcState::class.java).clear()
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `backlog gauges are registered and reflect the current pipeline state`() {
|
||||||
|
val proc = ctx.getBean(StubProcState::class.java)
|
||||||
|
val t0 = Instant.parse("2026-09-08T03:00:00Z")
|
||||||
|
proc.insertIfAbsent(1L, t0) // 未处理 → backlog
|
||||||
|
proc.insertIfAbsent(2L, t0)
|
||||||
|
proc.markTerminal(2L, ProcStatus.SUCCEEDED, now = t0) // 终态未打标 → unmarked
|
||||||
|
|
||||||
|
assertEquals(1.0, gauge("msgx.pipeline.backlog.unfinished"), 0.001)
|
||||||
|
assertEquals(1.0, gauge("msgx.pipeline.backfill.unmarked_terminal"), 0.001)
|
||||||
|
assertEquals(0.0, gauge("msgx.pipeline.backfill.abandoned"), 0.001)
|
||||||
|
// 信箱为空 → 没有"最新 ID"可比,滞后用 -1(无值)而不是伪造 0
|
||||||
|
assertEquals(-1.0, gauge("msgx.pipeline.watermark.lag"), 0.001)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `permanent hole releases are exposed as a monotonic total`() {
|
||||||
|
val counters = ctx.getBean(PipelineCounters::class.java)
|
||||||
|
val before = gauge("msgx.pipeline.hole.aged_out.total")
|
||||||
|
counters.holeAgedOutIncrement()
|
||||||
|
counters.holeAgedOutIncrement()
|
||||||
|
|
||||||
|
assertEquals(before + 2.0, gauge("msgx.pipeline.hole.aged_out.total"), 0.001)
|
||||||
|
}
|
||||||
|
|
||||||
|
private fun gauge(name: String): Double =
|
||||||
|
requireNotNull(registry.find(name).gauge()) { "gauge not registered: $name" }.value()
|
||||||
|
}
|
||||||
+13
-5
@@ -40,7 +40,7 @@ class FlywayMigrationTest {
|
|||||||
|
|
||||||
DriverManager.getConnection(url, user, pass).use { conn ->
|
DriverManager.getConnection(url, user, pass).use { conn ->
|
||||||
conn.createStatement().use { stmt ->
|
conn.createStatement().use { stmt ->
|
||||||
// 迁移记录:V1 基线 + V2 信箱生命周期 + V3 稳定处理起点
|
// 迁移记录:V1 基线 + V2 信箱生命周期 + V3 稳定处理起点 + V4 回填闭环 + V5 切流播种
|
||||||
stmt.executeQuery(
|
stmt.executeQuery(
|
||||||
"SELECT version, script, success FROM flyway_schema_history ORDER BY installed_rank ASC",
|
"SELECT version, script, success FROM flyway_schema_history ORDER BY installed_rank ASC",
|
||||||
).use { rs ->
|
).use { rs ->
|
||||||
@@ -48,13 +48,17 @@ class FlywayMigrationTest {
|
|||||||
while (rs.next()) {
|
while (rs.next()) {
|
||||||
records.add(Triple(rs.getString("version"), rs.getString("script"), rs.getBoolean("success")))
|
records.add(Triple(rs.getString("version"), rs.getString("script"), rs.getBoolean("success")))
|
||||||
}
|
}
|
||||||
assertTrue(records.size >= 3, "flyway_schema_history must record all migrations")
|
assertTrue(records.size >= 5, "flyway_schema_history must record all migrations")
|
||||||
assertEquals("1", records[0].first)
|
assertEquals("1", records[0].first)
|
||||||
assertEquals("V1__flight_state_baseline.sql", records[0].second)
|
assertEquals("V1__flight_state_baseline.sql", records[0].second)
|
||||||
assertEquals("2", records[1].first)
|
assertEquals("2", records[1].first)
|
||||||
assertEquals("V2__inbox_lifecycle.sql", records[1].second)
|
assertEquals("V2__inbox_lifecycle.sql", records[1].second)
|
||||||
assertEquals("3", records[2].first)
|
assertEquals("3", records[2].first)
|
||||||
assertEquals("V3__stable_processing_start.sql", records[2].second)
|
assertEquals("V3__stable_processing_start.sql", records[2].second)
|
||||||
|
assertEquals("4", records[3].first)
|
||||||
|
assertEquals("V4__backfill_closure.sql", records[3].second)
|
||||||
|
assertEquals("5", records[4].first)
|
||||||
|
assertEquals("V5__cutover_seed.sql", records[4].second)
|
||||||
assertTrue(records.all { it.third })
|
assertTrue(records.all { it.third })
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -79,16 +83,20 @@ class FlywayMigrationTest {
|
|||||||
assertEquals(setOf("flid", "operation_day", "state", "state_version", "last_msg_id"), cols)
|
assertEquals(setOf("flid", "operation_day", "state", "state_version", "last_msg_id"), cols)
|
||||||
}
|
}
|
||||||
|
|
||||||
// 回填相关的列都落在 PROC_STATE 上(收信时间用来判断超期,其余记录回填进度)
|
// 回填相关的列都落在 PROC_STATE 上(收信时间判断超期;abandoned 记录"停止自动重试")
|
||||||
stmt.executeQuery(
|
stmt.executeQuery(
|
||||||
"SELECT column_name FROM information_schema.columns WHERE table_name = 'proc_state' " +
|
"SELECT column_name FROM information_schema.columns WHERE table_name = 'proc_state' " +
|
||||||
"AND column_name IN ('received_at', 'backfill_at', 'backfill_next_at', " +
|
"AND column_name IN ('received_at', 'backfill_at', 'backfill_next_at', " +
|
||||||
"'backfill_attempts', 'backfill_error', 'processing_started_at')",
|
"'backfill_attempts', 'backfill_error', 'backfill_abandoned_at', " +
|
||||||
|
"'backfill_abandoned_reason', 'processing_started_at')",
|
||||||
).use { rs ->
|
).use { rs ->
|
||||||
val cols = mutableSetOf<String>()
|
val cols = mutableSetOf<String>()
|
||||||
while (rs.next()) cols.add(rs.getString("column_name"))
|
while (rs.next()) cols.add(rs.getString("column_name"))
|
||||||
assertEquals(
|
assertEquals(
|
||||||
setOf("received_at", "backfill_at", "backfill_next_at", "backfill_attempts", "backfill_error", "processing_started_at"),
|
setOf(
|
||||||
|
"received_at", "backfill_at", "backfill_next_at", "backfill_attempts", "backfill_error",
|
||||||
|
"backfill_abandoned_at", "backfill_abandoned_reason", "processing_started_at",
|
||||||
|
),
|
||||||
cols,
|
cols,
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|||||||
+5
-2
@@ -58,9 +58,9 @@ class InboxLifecycleJdbcSqlTest {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
proc = JdbcProcStateRepository(ds)
|
proc = JdbcProcStateRepository(ds, java.time.Clock.systemUTC())
|
||||||
mailbox = JdbcCminmsgInboxRepository(ds)
|
mailbox = JdbcCminmsgInboxRepository(ds)
|
||||||
cursor = JdbcInboxCursorRepository(ds)
|
cursor = JdbcInboxCursorRepository(ds, java.time.Clock.systemUTC())
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -223,6 +223,8 @@ class InboxLifecycleJdbcSqlTest {
|
|||||||
backfill_next_at TIMESTAMP WITH TIME ZONE,
|
backfill_next_at TIMESTAMP WITH TIME ZONE,
|
||||||
backfill_attempts INT NOT NULL DEFAULT 0,
|
backfill_attempts INT NOT NULL DEFAULT 0,
|
||||||
backfill_error VARCHAR(512),
|
backfill_error VARCHAR(512),
|
||||||
|
backfill_abandoned_at TIMESTAMP WITH TIME ZONE,
|
||||||
|
backfill_abandoned_reason VARCHAR(64),
|
||||||
processing_started_at TIMESTAMP WITH TIME ZONE,
|
processing_started_at TIMESTAMP WITH TIME ZONE,
|
||||||
updated_at TIMESTAMP WITH TIME ZONE NOT NULL,
|
updated_at TIMESTAMP WITH TIME ZONE NOT NULL,
|
||||||
CONSTRAINT uk_proc_identity UNIQUE (identity_key)
|
CONSTRAINT uk_proc_identity UNIQUE (identity_key)
|
||||||
@@ -234,6 +236,7 @@ class InboxLifecycleJdbcSqlTest {
|
|||||||
cursor_id INT PRIMARY KEY,
|
cursor_id INT PRIMARY KEY,
|
||||||
committed_up_to BIGINT NOT NULL,
|
committed_up_to BIGINT NOT NULL,
|
||||||
hole_since TIMESTAMP WITH TIME ZONE,
|
hole_since TIMESTAMP WITH TIME ZONE,
|
||||||
|
seeded_at TIMESTAMP WITH TIME ZONE,
|
||||||
updated_at TIMESTAMP WITH TIME ZONE NOT NULL
|
updated_at TIMESTAMP WITH TIME ZONE NOT NULL
|
||||||
)
|
)
|
||||||
"""
|
"""
|
||||||
|
|||||||
@@ -47,11 +47,17 @@ class ReplayServiceTest {
|
|||||||
|
|
||||||
override fun recordBackfillFailure(msgId: Long, error: String?, attempts: Int, nextAttemptAt: Instant, now: Instant) = Unit
|
override fun recordBackfillFailure(msgId: Long, error: String?, attempts: Int, nextAttemptAt: Instant, now: Instant) = Unit
|
||||||
|
|
||||||
|
override fun markBackfillAbandoned(msgId: Long, reason: String, now: Instant): Boolean = false
|
||||||
|
|
||||||
|
override fun reopenBackfill(msgId: Long, now: Instant): Boolean = false
|
||||||
|
|
||||||
override fun findBackfillDue(now: Instant, overdueBefore: Instant, limit: Int) =
|
override fun findBackfillDue(now: Instant, overdueBefore: Instant, limit: Int) =
|
||||||
emptyList<com.gzzn.omms.msgexchange.infra.persistence.BackfillDue>()
|
emptyList<com.gzzn.omms.msgexchange.infra.persistence.BackfillDue>()
|
||||||
|
|
||||||
override fun backlog() = com.gzzn.omms.msgexchange.infra.persistence.Backlog(0, null, 0)
|
override fun backlog() = com.gzzn.omms.msgexchange.infra.persistence.Backlog(0, null, 0)
|
||||||
|
|
||||||
|
override fun hasAny(): Boolean = rows.isNotEmpty()
|
||||||
|
|
||||||
override fun requeueByErrorClasses(errorClasses: List<ErrorClass>): Int {
|
override fun requeueByErrorClasses(errorClasses: List<ErrorClass>): Int {
|
||||||
requeueCalls += errorClasses
|
requeueCalls += errorClasses
|
||||||
var n = 0
|
var n = 0
|
||||||
@@ -74,7 +80,8 @@ class ReplayServiceTest {
|
|||||||
repo.seed(2, ProcStatus.DEAD, ErrorClass.MALFORMED)
|
repo.seed(2, ProcStatus.DEAD, ErrorClass.MALFORMED)
|
||||||
repo.seed(3, ProcStatus.FAILED, ErrorClass.UNSUPPORTED)
|
repo.seed(3, ProcStatus.FAILED, ErrorClass.UNSUPPORTED)
|
||||||
|
|
||||||
val n = ReplayService(repo).replay(listOf(ErrorClass.CODEC_ERROR, ErrorClass.MALFORMED, ErrorClass.UNSUPPORTED))
|
val n = ReplayService(repo, com.gzzn.omms.msgexchange.processing.MessageLifecycleGate())
|
||||||
|
.replay(listOf(ErrorClass.CODEC_ERROR, ErrorClass.MALFORMED, ErrorClass.UNSUPPORTED))
|
||||||
|
|
||||||
assertEquals(2, n)
|
assertEquals(2, n)
|
||||||
assertEquals(listOf(listOf(ErrorClass.CODEC_ERROR, ErrorClass.UNSUPPORTED)), repo.requeueCalls)
|
assertEquals(listOf(listOf(ErrorClass.CODEC_ERROR, ErrorClass.UNSUPPORTED)), repo.requeueCalls)
|
||||||
@@ -91,7 +98,8 @@ class ReplayServiceTest {
|
|||||||
val repo = FakeRepo()
|
val repo = FakeRepo()
|
||||||
repo.seed(2, ProcStatus.DEAD, ErrorClass.MALFORMED)
|
repo.seed(2, ProcStatus.DEAD, ErrorClass.MALFORMED)
|
||||||
|
|
||||||
val n = ReplayService(repo).replay(listOf(ErrorClass.MALFORMED))
|
val n = ReplayService(repo, com.gzzn.omms.msgexchange.processing.MessageLifecycleGate())
|
||||||
|
.replay(listOf(ErrorClass.MALFORMED))
|
||||||
|
|
||||||
assertEquals(0, n)
|
assertEquals(0, n)
|
||||||
assertNull(repo.requeueCalls.lastOrNull()) // 白名单过滤后为空 → 不触达仓储
|
assertNull(repo.requeueCalls.lastOrNull()) // 白名单过滤后为空 → 不触达仓储
|
||||||
@@ -105,7 +113,7 @@ class ReplayServiceTest {
|
|||||||
repo.seed(5, ProcStatus.DEAD, ErrorClass.EXHAUSTED)
|
repo.seed(5, ProcStatus.DEAD, ErrorClass.EXHAUSTED)
|
||||||
repo.seed(6, ProcStatus.DEAD, ErrorClass.MALFORMED)
|
repo.seed(6, ProcStatus.DEAD, ErrorClass.MALFORMED)
|
||||||
|
|
||||||
val n = ReplayService(repo).replayAll()
|
val n = ReplayService(repo, com.gzzn.omms.msgexchange.processing.MessageLifecycleGate()).replayAll()
|
||||||
|
|
||||||
assertEquals(2, n)
|
assertEquals(2, n)
|
||||||
assertEquals(ProcStatus.PENDING, repo.rows[4]!!.state)
|
assertEquals(ProcStatus.PENDING, repo.rows[4]!!.state)
|
||||||
|
|||||||
@@ -0,0 +1,165 @@
|
|||||||
|
package com.gzzn.omms.msgexchange.ingress
|
||||||
|
|
||||||
|
import com.gzzn.omms.msgexchange.config.PipelineProps
|
||||||
|
import com.gzzn.omms.msgexchange.infra.metrics.PipelineCounters
|
||||||
|
import com.gzzn.omms.msgexchange.infra.persistence.InboxCursorRepository
|
||||||
|
import com.gzzn.omms.msgexchange.infra.stub.StubInbox
|
||||||
|
import com.gzzn.omms.msgexchange.infra.stub.StubInboxCursor
|
||||||
|
import com.gzzn.omms.msgexchange.infra.stub.StubPipelineTx
|
||||||
|
import com.gzzn.omms.msgexchange.infra.stub.StubProcState
|
||||||
|
import org.junit.jupiter.api.Assertions.assertEquals
|
||||||
|
import org.junit.jupiter.api.Assertions.assertNotEquals
|
||||||
|
import org.junit.jupiter.api.Assertions.assertNotNull
|
||||||
|
import org.junit.jupiter.api.Assertions.assertNull
|
||||||
|
import org.junit.jupiter.api.Assertions.assertTrue
|
||||||
|
import org.junit.jupiter.api.BeforeEach
|
||||||
|
import org.junit.jupiter.api.Test
|
||||||
|
import java.time.Clock
|
||||||
|
import java.time.Instant
|
||||||
|
import java.time.ZoneOffset
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 切流水位播种(ACM2-35):**显式、一次性、升级安全**。
|
||||||
|
*
|
||||||
|
* 覆盖评审提出的四条硬要求:
|
||||||
|
* 1. 默认(未配置)不动作,代码不做默认选择;
|
||||||
|
* 2. 四种模式严格区分(`min` 读现存全部 / `zero` 从 0 按空洞规则 / `max` 跳过可见存量 / 显式 ID);
|
||||||
|
* 3. 升级实例(已有水位或已有处理记录)**拒绝重新播种**——`SEEDED_AT` 为 NULL 不等于"从未消费";
|
||||||
|
* 4. 播种至多一次,且不会被普通的水位推进抹掉。
|
||||||
|
*/
|
||||||
|
class CutoverSeedTest {
|
||||||
|
|
||||||
|
private val t0: Instant = Instant.parse("2026-09-08T03:00:00Z")
|
||||||
|
private val props = PipelineProps()
|
||||||
|
|
||||||
|
private lateinit var inbox: StubInbox
|
||||||
|
private lateinit var proc: StubProcState
|
||||||
|
private lateinit var cursor: StubInboxCursor
|
||||||
|
private lateinit var poller: InboxPoller
|
||||||
|
|
||||||
|
@BeforeEach
|
||||||
|
fun setUp() {
|
||||||
|
inbox = StubInbox().apply { clear() }
|
||||||
|
proc = StubProcState().apply { clear() }
|
||||||
|
cursor = StubInboxCursor().apply { clear() }
|
||||||
|
props.pipeline.cutoverWatermark = null
|
||||||
|
poller = InboxPoller(
|
||||||
|
inbox, proc, cursor, StubPipelineTx(), props,
|
||||||
|
Clock.fixed(t0, ZoneOffset.UTC), PipelineCounters(),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** 造一个 MIN(ID)=5 的信箱:1..4 已被库方清除(典型"最老分区已 DROP")。 */
|
||||||
|
private fun mailboxWithMinId5(): List<Long> {
|
||||||
|
val ids = (1..8).map { inbox.insertRaw("<MSG/>") }
|
||||||
|
ids.take(4).forEach { inbox.removeRow(it) }
|
||||||
|
return ids.drop(4)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `default does nothing - no seeding without explicit configuration`() {
|
||||||
|
val kept = mailboxWithMinId5()
|
||||||
|
|
||||||
|
assertEquals(0, poller.pollOnce(t0)) // W=0 → ID=1 判为空洞,不推进
|
||||||
|
|
||||||
|
assertNull(cursor.cursor.seededAt)
|
||||||
|
assertEquals(0L, cursor.cursor.committedUpTo)
|
||||||
|
assertNotNull(cursor.cursor.holeSince)
|
||||||
|
assertTrue(kept.all { proc.find(it) == null })
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `min mode reads all currently existing rows`() {
|
||||||
|
val kept = mailboxWithMinId5()
|
||||||
|
props.pipeline.cutoverWatermark = "min"
|
||||||
|
|
||||||
|
assertEquals(4, poller.pollOnce(t0)) // 播种 W=4,同一轮把 5..8 全部读入
|
||||||
|
|
||||||
|
assertNotNull(cursor.cursor.seededAt)
|
||||||
|
assertEquals(8L, cursor.cursor.committedUpTo)
|
||||||
|
assertTrue(kept.all { proc.find(it) != null })
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `max mode skips the currently visible backlog`() {
|
||||||
|
mailboxWithMinId5()
|
||||||
|
props.pipeline.cutoverWatermark = "max"
|
||||||
|
|
||||||
|
assertEquals(0, poller.pollOnce(t0))
|
||||||
|
|
||||||
|
assertNotNull(cursor.cursor.seededAt)
|
||||||
|
assertEquals(8L, cursor.cursor.committedUpTo) // 直接跳到 MAX(ID)
|
||||||
|
assertEquals(0, proc.rows.size)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `zero mode scans from zero and stops at the first hole`() {
|
||||||
|
mailboxWithMinId5()
|
||||||
|
props.pipeline.cutoverWatermark = "zero"
|
||||||
|
|
||||||
|
assertEquals(0, poller.pollOnce(t0))
|
||||||
|
|
||||||
|
assertNotNull(cursor.cursor.seededAt)
|
||||||
|
assertEquals(0L, cursor.cursor.committedUpTo) // 1..4 是空洞 → 按空洞规则停住
|
||||||
|
assertNotNull(cursor.cursor.holeSince)
|
||||||
|
assertEquals(0, proc.rows.size)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `explicit boundary mode is accepted`() {
|
||||||
|
mailboxWithMinId5()
|
||||||
|
props.pipeline.cutoverWatermark = "6"
|
||||||
|
|
||||||
|
poller.pollOnce(t0)
|
||||||
|
|
||||||
|
assertNotNull(cursor.cursor.seededAt)
|
||||||
|
assertEquals(8L, cursor.cursor.committedUpTo) // 从 6 起读 7..8(6 本身已在界内)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `upgraded instance with existing watermark refuses to re-seed`() {
|
||||||
|
mailboxWithMinId5()
|
||||||
|
cursor.save(InboxCursorRepository.Cursor(committedUpTo = 3L, holeSince = t0))
|
||||||
|
props.pipeline.cutoverWatermark = "max"
|
||||||
|
|
||||||
|
poller.pollOnce(t0)
|
||||||
|
|
||||||
|
assertNull(cursor.cursor.seededAt) // 没有重新播种
|
||||||
|
assertNotEquals(8L, cursor.cursor.committedUpTo)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `instance that already consumed messages refuses to seed even at W zero`() {
|
||||||
|
mailboxWithMinId5()
|
||||||
|
proc.insertIfAbsent(1L, t0) // 已有处理记录(即使水位还是 0)
|
||||||
|
props.pipeline.cutoverWatermark = "max"
|
||||||
|
|
||||||
|
poller.pollOnce(t0)
|
||||||
|
|
||||||
|
assertNull(cursor.cursor.seededAt)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `seeding happens at most once and survives normal watermark saves`() {
|
||||||
|
mailboxWithMinId5()
|
||||||
|
props.pipeline.cutoverWatermark = "max"
|
||||||
|
poller.pollOnce(t0)
|
||||||
|
val seededAt = cursor.cursor.seededAt
|
||||||
|
assertNotNull(seededAt)
|
||||||
|
|
||||||
|
cursor.save(InboxCursorRepository.Cursor(committedUpTo = 2L, holeSince = null))
|
||||||
|
poller.pollOnce(t0.plusSeconds(1))
|
||||||
|
|
||||||
|
assertEquals(seededAt, cursor.cursor.seededAt) // 标记仍在:普通 save 不抹播种事实
|
||||||
|
assertEquals(2L, cursor.cursor.committedUpTo)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `empty mailbox is left unseeded`() {
|
||||||
|
props.pipeline.cutoverWatermark = "max"
|
||||||
|
|
||||||
|
assertEquals(0, poller.pollOnce(t0))
|
||||||
|
|
||||||
|
assertNull(cursor.cursor.seededAt)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -11,6 +11,7 @@ import org.junit.jupiter.api.Assertions.assertEquals
|
|||||||
import org.junit.jupiter.api.Assertions.assertFalse
|
import org.junit.jupiter.api.Assertions.assertFalse
|
||||||
import org.junit.jupiter.api.Assertions.assertNotNull
|
import org.junit.jupiter.api.Assertions.assertNotNull
|
||||||
import org.junit.jupiter.api.Assertions.assertNull
|
import org.junit.jupiter.api.Assertions.assertNull
|
||||||
|
import org.junit.jupiter.api.Assertions.assertTrue
|
||||||
import org.junit.jupiter.api.BeforeEach
|
import org.junit.jupiter.api.BeforeEach
|
||||||
import org.junit.jupiter.api.Test
|
import org.junit.jupiter.api.Test
|
||||||
import java.time.Instant
|
import java.time.Instant
|
||||||
@@ -38,7 +39,11 @@ class InboxPollerTest {
|
|||||||
inbox = StubInbox().apply { clear() }
|
inbox = StubInbox().apply { clear() }
|
||||||
proc = StubProcState().apply { clear() }
|
proc = StubProcState().apply { clear() }
|
||||||
cursor = StubInboxCursor().apply { clear() }
|
cursor = StubInboxCursor().apply { clear() }
|
||||||
poller = InboxPoller(inbox, proc, cursor, StubPipelineTx(), props)
|
poller = InboxPoller(
|
||||||
|
inbox, proc, cursor, StubPipelineTx(), props,
|
||||||
|
java.time.Clock.fixed(t0, java.time.ZoneOffset.UTC),
|
||||||
|
com.gzzn.omms.msgexchange.infra.metrics.PipelineCounters(),
|
||||||
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -67,7 +72,7 @@ class InboxPollerTest {
|
|||||||
val dead = (1..3).map { inbox.simulateExternalWrite("<MSG/>") }
|
val dead = (1..3).map { inbox.simulateExternalWrite("<MSG/>") }
|
||||||
assertEquals(3, poller.pollOnce(t0))
|
assertEquals(3, poller.pollOnce(t0))
|
||||||
dead.forEach {
|
dead.forEach {
|
||||||
proc.markTerminal(it, ProcStatus.DEAD, errorClass = ErrorClass.MALFORMED, lastError = "raw-missing")
|
proc.markTerminal(it, ProcStatus.DEAD, errorClass = ErrorClass.MALFORMED, lastError = "raw-missing", now = t0)
|
||||||
}
|
}
|
||||||
|
|
||||||
val fresh = inbox.simulateExternalWrite("<MSG/>")
|
val fresh = inbox.simulateExternalWrite("<MSG/>")
|
||||||
@@ -112,7 +117,7 @@ class InboxPollerTest {
|
|||||||
|
|
||||||
@Test
|
@Test
|
||||||
fun `compat http path and poller do not double enqueue the same message`() {
|
fun `compat http path and poller do not double enqueue the same message`() {
|
||||||
val receipt = InboxService(inbox, proc).accept("<MSG/>")
|
val receipt = InboxService(inbox, proc, java.time.Clock.fixed(t0, java.time.ZoneOffset.UTC)).accept("<MSG/>")
|
||||||
|
|
||||||
assertEquals(0, poller.pollOnce(t0))
|
assertEquals(0, poller.pollOnce(t0))
|
||||||
assertEquals(receipt.msgId, cursor.cursor.committedUpTo) // 已在 PG:读取进度照常推进
|
assertEquals(receipt.msgId, cursor.cursor.committedUpTo) // 已在 PG:读取进度照常推进
|
||||||
@@ -144,4 +149,60 @@ class InboxPollerTest {
|
|||||||
assertNull(proc.find(fifth))
|
assertNull(proc.find(fifth))
|
||||||
assertEquals(first + 2, third)
|
assertEquals(first + 2, third)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 【缺口基线 · G1】快路径只读 `ID > W`、从不回头:水位越过某个 ID 之后,该 ID 即使后来
|
||||||
|
* 出现在信箱里也不会再被发现。这不是期望行为,而是"没有补偿扫描"的**已知缺口**;
|
||||||
|
* 本用例把它固定成基线,补偿扫描(ACM2-41 / G1)落地后必须反转成"能被发现并安全处置"。
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
fun `baseline - an id that appears after the watermark passed it is never discovered`() {
|
||||||
|
inbox.simulateExternalWrite("<MSG/>") // 1
|
||||||
|
val hole = inbox.simulateExternalWrite("<MSG/>") // 2
|
||||||
|
val third = inbox.simulateExternalWrite("<MSG/>") // 3
|
||||||
|
inbox.removeRow(hole)
|
||||||
|
|
||||||
|
poller.pollOnce(t0) // W 停在 1,空洞在 2
|
||||||
|
val agedOut = t0.plus(props.pipeline.maxCommitDelay)
|
||||||
|
poller.pollOnce(agedOut) // 空洞判永久 → 放行
|
||||||
|
poller.pollOnce(agedOut) // 发现 3
|
||||||
|
assertEquals(third, cursor.cursor.committedUpTo)
|
||||||
|
|
||||||
|
// 迟到的 2 现在才出现:水位已经越过它,快路径再也不会读它。
|
||||||
|
inbox.restoreRow(hole, "<MSG/>", agedOut)
|
||||||
|
|
||||||
|
assertEquals(0, poller.pollOnce(agedOut.plusSeconds(1)))
|
||||||
|
assertNull(proc.find(hole))
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 兼容入口会把消息登记在水位**之外**(`msgId > W`),因此它天然成为"最小未完成行"。
|
||||||
|
* 领取侧的守卫在主泵:只领 `msgId <= W`(端到端验收见 `PipelineSmokeTest` 的
|
||||||
|
* "compat injected high id is not claimed until the watermark catches up")。
|
||||||
|
* 本用例只固定收报侧的事实:**登记发生,但水位不动**。
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
fun `compat accept registers a row above the watermark without advancing it`() {
|
||||||
|
props.pipeline.claimBatch = 10
|
||||||
|
val one = inbox.simulateExternalWrite("<MSG/>") // 1
|
||||||
|
val hole = inbox.simulateExternalWrite("<MSG/>") // 2(空洞)
|
||||||
|
val third = inbox.simulateExternalWrite("<MSG/>") // 3
|
||||||
|
inbox.removeRow(hole)
|
||||||
|
|
||||||
|
assertEquals(1, poller.pollOnce(t0)) // W=1,遇空洞即停
|
||||||
|
assertEquals(one, cursor.cursor.committedUpTo)
|
||||||
|
assertNull(proc.find(third)) // 3 还没入队
|
||||||
|
|
||||||
|
val receipt = InboxService(inbox, proc, java.time.Clock.fixed(t0, java.time.ZoneOffset.UTC))
|
||||||
|
.accept("<MSG/>") // 兼容入口:高 ID 直接进 PG
|
||||||
|
val high = receipt.msgId
|
||||||
|
assertTrue(high > third)
|
||||||
|
|
||||||
|
assertEquals(one, cursor.cursor.committedUpTo) // 水位不动(不参与水位)
|
||||||
|
// 主泵处理掉队头 1 之后,最小未完成行就是兼容入口写进来的高 ID……
|
||||||
|
proc.markTerminal(one, ProcStatus.SUCCEEDED, now = t0)
|
||||||
|
assertEquals(high, proc.headUnfinished()!!.msgId)
|
||||||
|
assertTrue(high > cursor.cursor.committedUpTo) // ……但它超出水位,主泵不会领取
|
||||||
|
assertNull(proc.find(third))
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,122 @@
|
|||||||
|
package com.gzzn.omms.msgexchange.ingress
|
||||||
|
|
||||||
|
import com.gzzn.omms.msgexchange.config.PipelineProps
|
||||||
|
import com.gzzn.omms.msgexchange.infra.metrics.PipelineCounters
|
||||||
|
import com.gzzn.omms.msgexchange.infra.stub.StubInbox
|
||||||
|
import com.gzzn.omms.msgexchange.infra.stub.StubInboxCursor
|
||||||
|
import com.gzzn.omms.msgexchange.infra.stub.StubPipelineTx
|
||||||
|
import com.gzzn.omms.msgexchange.infra.stub.StubProcState
|
||||||
|
import org.junit.jupiter.api.Assertions.assertEquals
|
||||||
|
import org.junit.jupiter.api.Assertions.assertNotNull
|
||||||
|
import org.junit.jupiter.api.Assertions.assertNull
|
||||||
|
import org.junit.jupiter.api.BeforeEach
|
||||||
|
import org.junit.jupiter.api.Test
|
||||||
|
import java.time.Clock
|
||||||
|
import java.time.Duration
|
||||||
|
import java.time.Instant
|
||||||
|
import java.time.ZoneOffset
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 迟到到达检测(ACM2-41 阶段 0):**只读观测**。
|
||||||
|
*
|
||||||
|
* 语义:被判定为永久空洞并放行的 ID,如果后来真的出现在信箱里,就是"上游提交晚于水位推进"。
|
||||||
|
* 阶段 0 只计数与告警,**不入队、不改变处理语义**(补入队属阶段 1,需先与库方定案)。
|
||||||
|
*/
|
||||||
|
class LateArrivalDetectTest {
|
||||||
|
|
||||||
|
private val t0: Instant = Instant.parse("2026-09-08T03:00:00Z")
|
||||||
|
private val props = PipelineProps()
|
||||||
|
|
||||||
|
private lateinit var inbox: StubInbox
|
||||||
|
private lateinit var proc: StubProcState
|
||||||
|
private lateinit var cursor: StubInboxCursor
|
||||||
|
private lateinit var counters: PipelineCounters
|
||||||
|
private lateinit var poller: InboxPoller
|
||||||
|
|
||||||
|
@BeforeEach
|
||||||
|
fun setUp() {
|
||||||
|
inbox = StubInbox().apply { clear() }
|
||||||
|
proc = StubProcState().apply { clear() }
|
||||||
|
cursor = StubInboxCursor().apply { clear() }
|
||||||
|
counters = PipelineCounters()
|
||||||
|
poller = InboxPoller(inbox, proc, cursor, StubPipelineTx(), props, Clock.fixed(t0, ZoneOffset.UTC), counters)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** 造出"1 存在、2 是空洞、3 存在",并把空洞等到超期放行。返回迟到的那个 ID。 */
|
||||||
|
private fun ageOutHoleAt2(): Long {
|
||||||
|
inbox.simulateExternalWrite("<MSG/>") // 1
|
||||||
|
val hole = inbox.simulateExternalWrite("<MSG/>") // 2
|
||||||
|
inbox.simulateExternalWrite("<MSG/>") // 3
|
||||||
|
inbox.removeRow(hole)
|
||||||
|
poller.pollOnce(t0) // W=1,空洞在 2
|
||||||
|
poller.pollOnce(t0.plus(props.pipeline.maxCommitDelay)) // 空洞判永久 → 放行
|
||||||
|
return hole
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `a hole that really appears later is detected and counted`() {
|
||||||
|
val hole = ageOutHoleAt2()
|
||||||
|
inbox.restoreRow(hole, "<MSG/>", t0.plus(props.pipeline.maxCommitDelay))
|
||||||
|
|
||||||
|
// 过了检测周期再轮询一次
|
||||||
|
poller.pollOnce(t0.plus(props.pipeline.maxCommitDelay).plus(props.pipeline.lateDetectPeriod))
|
||||||
|
|
||||||
|
assertEquals(1L, counters.lateArrivalDetectedCount())
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `detection does not enqueue the late message - phase 0 is observation only`() {
|
||||||
|
val hole = ageOutHoleAt2()
|
||||||
|
inbox.restoreRow(hole, "<MSG/>", t0.plus(props.pipeline.maxCommitDelay))
|
||||||
|
|
||||||
|
poller.pollOnce(t0.plus(props.pipeline.maxCommitDelay).plus(props.pipeline.lateDetectPeriod))
|
||||||
|
|
||||||
|
assertEquals(1L, counters.lateArrivalDetectedCount())
|
||||||
|
assertNull(proc.find(hole)) // 仍然不会被补入队
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `a hole that stays absent is not counted`() {
|
||||||
|
ageOutHoleAt2()
|
||||||
|
|
||||||
|
poller.pollOnce(t0.plus(props.pipeline.maxCommitDelay).plus(props.pipeline.lateDetectPeriod))
|
||||||
|
|
||||||
|
assertEquals(0L, counters.lateArrivalDetectedCount())
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `detection does not run before the configured period elapses`() {
|
||||||
|
val hole = ageOutHoleAt2()
|
||||||
|
inbox.restoreRow(hole, "<MSG/>", t0)
|
||||||
|
|
||||||
|
// 只过了一个 max-commit-delay,未到检测周期
|
||||||
|
poller.pollOnce(t0.plus(props.pipeline.maxCommitDelay).plusSeconds(1))
|
||||||
|
|
||||||
|
assertEquals(0L, counters.lateArrivalDetectedCount())
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `detection can be switched off`() {
|
||||||
|
props.pipeline.lateDetectPeriod = Duration.ZERO
|
||||||
|
val hole = ageOutHoleAt2()
|
||||||
|
inbox.restoreRow(hole, "<MSG/>", t0)
|
||||||
|
|
||||||
|
poller.pollOnce(t0.plus(props.pipeline.maxCommitDelay).plus(Duration.ofHours(1)))
|
||||||
|
|
||||||
|
assertEquals(0L, counters.lateArrivalDetectedCount())
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `the same late id is counted once even after the hole is aged out again`() {
|
||||||
|
val hole = ageOutHoleAt2()
|
||||||
|
inbox.restoreRow(hole, "<MSG/>", t0.plus(props.pipeline.maxCommitDelay))
|
||||||
|
val later = t0.plus(props.pipeline.maxCommitDelay).plus(props.pipeline.lateDetectPeriod)
|
||||||
|
poller.pollOnce(later)
|
||||||
|
assertEquals(1L, counters.lateArrivalDetectedCount())
|
||||||
|
|
||||||
|
poller.pollOnce(later.plus(props.pipeline.lateDetectPeriod))
|
||||||
|
|
||||||
|
assertEquals(1L, counters.lateArrivalDetectedCount()) // 已命中的 ID 不再重复计数
|
||||||
|
assertNotNull(cursor.cursor)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -41,21 +41,24 @@ class BackfillServiceTest {
|
|||||||
private val props = PipelineProps()
|
private val props = PipelineProps()
|
||||||
|
|
||||||
/** 可以人为制造故障的信箱,用来验证回填失败时怎么处理。 */
|
/** 可以人为制造故障的信箱,用来验证回填失败时怎么处理。 */
|
||||||
private class FakeMailbox(var fail: Boolean = false) : CminmsgInboxRepository {
|
private class FakeMailbox(var fail: Boolean = false, var missing: Boolean = false) : CminmsgInboxRepository {
|
||||||
val marked = linkedSetOf<Long>()
|
val marked = linkedSetOf<Long>()
|
||||||
override fun insertRaw(rawXml: String): Long = 1L
|
override fun insertRaw(rawXml: String): Long = 1L
|
||||||
override fun rawOf(msgId: Long): String? = null
|
override fun rawOf(msgId: Long): String? = null
|
||||||
override fun receivedAtOf(msgId: Long): Instant? = null
|
override fun receivedAtOf(msgId: Long): Instant? = null
|
||||||
override fun readRange(fromExclusive: Long, limit: Int): List<MailboxRow> = emptyList()
|
override fun readRange(fromExclusive: Long, limit: Int): List<MailboxRow> = emptyList()
|
||||||
override fun maxId(): Long? = null
|
override fun maxId(): Long? = null
|
||||||
|
override fun minId(): Long? = null
|
||||||
|
override fun existingIds(msgIds: Collection<Long>): Set<Long> = emptySet()
|
||||||
override fun markProcessedIfUnmarked(msgId: Long, value: String): MailboxMarkResult {
|
override fun markProcessedIfUnmarked(msgId: Long, value: String): MailboxMarkResult {
|
||||||
if (fail) throw IllegalStateException("mysql-down")
|
if (fail) throw IllegalStateException("mysql-down")
|
||||||
|
if (missing) return MailboxMarkResult.MISSING
|
||||||
return if (marked.add(msgId)) MailboxMarkResult.MARKED else MailboxMarkResult.ALREADY_MARKED
|
return if (marked.add(msgId)) MailboxMarkResult.MARKED else MailboxMarkResult.ALREADY_MARKED
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
private fun service(proc: StubProcState, mailbox: CminmsgInboxRepository, now: Instant = t0) =
|
private fun service(proc: StubProcState, mailbox: CminmsgInboxRepository, now: Instant = t0) =
|
||||||
BackfillService(proc, mailbox, MailboxProps(), props, Clock.fixed(now, ZoneOffset.UTC))
|
BackfillService(proc, mailbox, MailboxProps(), props, Clock.fixed(now, ZoneOffset.UTC), MessageLifecycleGate())
|
||||||
|
|
||||||
/** 终态 + 回填意图(固定时刻,避免依赖真实时钟)。 */
|
/** 终态 + 回填意图(固定时刻,避免依赖真实时钟)。 */
|
||||||
private fun succeeded(proc: StubProcState, id: Long) {
|
private fun succeeded(proc: StubProcState, id: Long) {
|
||||||
@@ -122,8 +125,13 @@ class BackfillServiceTest {
|
|||||||
assertNull(row.backfillAt)
|
assertNull(row.backfillAt)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 评审修正 R3:信箱行不存在是**确定性结论**,不再当作"可重试失败"——
|
||||||
|
* 重试不会改变结果,只会每 30 秒重试一次并永久占满扫描批次。
|
||||||
|
* 但仍必须区分"停止自动重试"与"标记已确认":backfillAt 保持为空。
|
||||||
|
*/
|
||||||
@Test
|
@Test
|
||||||
fun `missing mailbox row remains an unconfirmed backfill failure`() {
|
fun `missing mailbox row is abandoned instead of retried forever`() {
|
||||||
val proc = StubProcState()
|
val proc = StubProcState()
|
||||||
val inbox = StubInbox().apply { clear() }
|
val inbox = StubInbox().apply { clear() }
|
||||||
val id = inbox.insertRaw("<MSG/>")
|
val id = inbox.insertRaw("<MSG/>")
|
||||||
@@ -133,10 +141,12 @@ class BackfillServiceTest {
|
|||||||
service(proc, inbox).attempt(id)
|
service(proc, inbox).attempt(id)
|
||||||
|
|
||||||
val row = proc.find(id)!!
|
val row = proc.find(id)!!
|
||||||
assertNull(row.backfillAt)
|
assertNull(row.backfillAt) // 放弃 ≠ 已确认
|
||||||
assertEquals(1, row.backfillAttempts)
|
assertEquals(BackfillService.ABANDON_MISSING_ROW, row.backfillAbandonedReason)
|
||||||
assertEquals("mailbox-row-missing", row.backfillError)
|
assertNotNull(row.backfillAbandonedAt)
|
||||||
assertNotNull(row.backfillNextAt)
|
assertNull(row.backfillNextAt) // 不再排下一次重试
|
||||||
|
assertNull(row.backfillError) // 不再是"失败重试",而是放弃
|
||||||
|
assertEquals(0, service(proc, inbox).sweep(t0)) // 也不再被扫描
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -229,6 +239,8 @@ class BackfillServiceTest {
|
|||||||
override fun receivedAtOf(msgId: Long) = t0
|
override fun receivedAtOf(msgId: Long) = t0
|
||||||
override fun readRange(fromExclusive: Long, limit: Int) = emptyList<MailboxRow>()
|
override fun readRange(fromExclusive: Long, limit: Int) = emptyList<MailboxRow>()
|
||||||
override fun maxId(): Long? = 1L
|
override fun maxId(): Long? = 1L
|
||||||
|
override fun minId(): Long? = 1L
|
||||||
|
override fun existingIds(msgIds: Collection<Long>): Set<Long> = emptySet()
|
||||||
override fun markProcessedIfUnmarked(msgId: Long, value: String): MailboxMarkResult {
|
override fun markProcessedIfUnmarked(msgId: Long, value: String): MailboxMarkResult {
|
||||||
entered.countDown()
|
entered.countDown()
|
||||||
release.await()
|
release.await()
|
||||||
@@ -262,4 +274,90 @@ class BackfillServiceTest {
|
|||||||
pool.shutdownNow()
|
pool.shutdownNow()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ------------------------------------------------------------------
|
||||||
|
// 回填闭环(评审修正 R3 + 饥饿回归)
|
||||||
|
// ------------------------------------------------------------------
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `missing mailbox row is abandoned immediately and is never treated as marked`() {
|
||||||
|
val proc = StubProcState()
|
||||||
|
val mailbox = FakeMailbox(missing = true)
|
||||||
|
succeeded(proc, 911L)
|
||||||
|
|
||||||
|
service(proc, mailbox).attempt(911L)
|
||||||
|
|
||||||
|
val row = proc.find(911L)!!
|
||||||
|
assertEquals(BackfillService.ABANDON_MISSING_ROW, row.backfillAbandonedReason)
|
||||||
|
assertNotNull(row.backfillAbandonedAt)
|
||||||
|
// 放弃 ≠ 标记已确认:清除前提(边界内全部行已打标)因此仍然不成立。
|
||||||
|
assertNull(row.backfillAt)
|
||||||
|
assertNull(row.backfillNextAt)
|
||||||
|
// 已放弃的行不再进入扫描:不会每 30 秒无限重试。
|
||||||
|
assertEquals(0, service(proc, mailbox).sweep(t0))
|
||||||
|
// 但它**不是**永久失去补偿:人工恢复后可以重新排队。
|
||||||
|
assertTrue(service(proc, mailbox).reopen(911L))
|
||||||
|
assertNull(proc.find(911L)!!.backfillAbandonedAt)
|
||||||
|
assertNotNull(proc.find(911L)!!.backfillNextAt)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `transient failures keep retrying until the attempt cap and stay recoverable`() {
|
||||||
|
props.pipeline.backfillMaxAttempts = 3
|
||||||
|
val proc = StubProcState()
|
||||||
|
val mailbox = FakeMailbox(fail = true)
|
||||||
|
succeeded(proc, 912L)
|
||||||
|
val svc = service(proc, mailbox)
|
||||||
|
|
||||||
|
svc.attempt(912L) // 1 次:暂时性故障,只退避
|
||||||
|
svc.attempt(912L) // 2 次
|
||||||
|
assertNull(proc.find(912L)!!.backfillAbandonedAt)
|
||||||
|
|
||||||
|
svc.attempt(912L) // 3 次:达到上限 → 停止自动重试
|
||||||
|
val row = proc.find(912L)!!
|
||||||
|
assertEquals(BackfillService.ABANDON_MAX_ATTEMPTS, row.backfillAbandonedReason)
|
||||||
|
assertNull(row.backfillAt)
|
||||||
|
|
||||||
|
assertTrue(svc.reopen(912L)) // 人工恢复入口存在且有效
|
||||||
|
assertNull(proc.find(912L)!!.backfillAbandonedAt)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `permanently failing oldest rows do not starve later rows (fair rotation)`() {
|
||||||
|
props.pipeline.backfillBatch = 3
|
||||||
|
val proc = StubProcState()
|
||||||
|
val overdueBefore = t0.plusSeconds(3600)
|
||||||
|
// 1..3:模拟"最旧且已尝试多次"的永久失败行,保持 due(nextAt <= now)
|
||||||
|
(1L..3L).forEach { id ->
|
||||||
|
proc.insertIfAbsent(id, t0)
|
||||||
|
proc.markTerminal(id, ProcStatus.SUCCEEDED, now = t0)
|
||||||
|
proc.recordBackfillFailure(id, "still-failing", attempts = 50, nextAttemptAt = t0, now = t0)
|
||||||
|
}
|
||||||
|
// 200:新行,从未尝试
|
||||||
|
proc.insertIfAbsent(200L, t0)
|
||||||
|
proc.markTerminal(200L, ProcStatus.SUCCEEDED, now = t0)
|
||||||
|
|
||||||
|
val due = proc.findBackfillDue(t0, overdueBefore, limit = 3)
|
||||||
|
|
||||||
|
assertEquals(3, due.size)
|
||||||
|
// 公平轮转:尝试次数少的先被扫描,因此新行不会被最旧的一批永久失败行饿死。
|
||||||
|
assertEquals(200L, due.first().msgId)
|
||||||
|
assertTrue(due.any { it.msgId == 200L })
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 语义固定(G4):`RECEIVED_AT` 为 NULL 时"超期"分支不成立,`R` 兜底**不生效**,
|
||||||
|
* 该行只能靠退避重试。把这条钉住,避免以后误以为 `R` 一定能兜底。
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
fun `null received time disables the overdue shortcut so only backoff applies`() {
|
||||||
|
val proc = StubProcState()
|
||||||
|
proc.insertIfAbsent(921L, null) // 上游未提供接收时间
|
||||||
|
proc.markTerminal(921L, ProcStatus.SUCCEEDED, now = t0)
|
||||||
|
proc.recordBackfillFailure(921L, "mysql-down", attempts = 1, nextAttemptAt = t0.plusSeconds(3600), now = t0)
|
||||||
|
|
||||||
|
val due = proc.findBackfillDue(now = t0, overdueBefore = t0.plusSeconds(10_000), limit = 10)
|
||||||
|
|
||||||
|
assertTrue(due.isEmpty()) // 超期分支无效 + 退避未到期
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -63,7 +63,7 @@ class FdelAndAdftProcessorTest {
|
|||||||
val proc = StubProcState()
|
val proc = StubProcState()
|
||||||
proc.insertIfAbsent(msgId, null)
|
proc.insertIfAbsent(msgId, null)
|
||||||
|
|
||||||
val result = FdelProcessor(StubPipelineTx(), StubPipelineLock(), f, events, proc, ObjectMapper())
|
val result = FdelProcessor(StubPipelineTx(), StubPipelineLock(), f, events, proc, ObjectMapper(), java.time.Clock.systemUTC())
|
||||||
.apply(head(), msg(), FlopPayload("121"))
|
.apply(head(), msg(), FlopPayload("121"))
|
||||||
|
|
||||||
assertEquals(ApplyResult.Succeeded, result)
|
assertEquals(ApplyResult.Succeeded, result)
|
||||||
@@ -83,7 +83,7 @@ class FdelAndAdftProcessorTest {
|
|||||||
fun `repeated FDEL is idempotent without version bump or duplicate event`() {
|
fun `repeated FDEL is idempotent without version bump or duplicate event`() {
|
||||||
val f = flights()
|
val f = flights()
|
||||||
val events = StubMsgEvents()
|
val events = StubMsgEvents()
|
||||||
val proc = FdelProcessor(StubPipelineTx(), StubPipelineLock(), f, events, StubProcState(), ObjectMapper())
|
val proc = FdelProcessor(StubPipelineTx(), StubPipelineLock(), f, events, StubProcState(), ObjectMapper(), java.time.Clock.systemUTC())
|
||||||
|
|
||||||
proc.apply(head(), msg(), FlopPayload("121"))
|
proc.apply(head(), msg(), FlopPayload("121"))
|
||||||
val version = f.findMainRow("121")!!.stateVersion
|
val version = f.findMainRow("121")!!.stateVersion
|
||||||
@@ -99,7 +99,7 @@ class FdelAndAdftProcessorTest {
|
|||||||
val f = StubFlightState()
|
val f = StubFlightState()
|
||||||
val events = StubMsgEvents()
|
val events = StubMsgEvents()
|
||||||
|
|
||||||
val result = FdelProcessor(StubPipelineTx(), StubPipelineLock(), f, events, StubProcState(), ObjectMapper())
|
val result = FdelProcessor(StubPipelineTx(), StubPipelineLock(), f, events, StubProcState(), ObjectMapper(), java.time.Clock.systemUTC())
|
||||||
.apply(head(), msg(), FlopPayload("999"))
|
.apply(head(), msg(), FlopPayload("999"))
|
||||||
|
|
||||||
assertEquals(ApplyResult.Succeeded, result) // 航班不存在(迟到或多余)也算成功
|
assertEquals(ApplyResult.Succeeded, result) // 航班不存在(迟到或多余)也算成功
|
||||||
@@ -113,7 +113,7 @@ class FdelAndAdftProcessorTest {
|
|||||||
val events = StubMsgEvents()
|
val events = StubMsgEvents()
|
||||||
val adft = AdftProcessor(
|
val adft = AdftProcessor(
|
||||||
StubPipelineTx(), StubPipelineLock(), f, events, StubProcState(),
|
StubPipelineTx(), StubPipelineLock(), f, events, StubProcState(),
|
||||||
OperationDayProps().apply { zone = "Asia/Shanghai" }, ObjectMapper(),
|
OperationDayProps().apply { zone = "Asia/Shanghai" }, ObjectMapper(), java.time.Clock.systemUTC(),
|
||||||
)
|
)
|
||||||
val record = com.gzzn.omms.msgexchange.domain.flight.ScheduleRecord(
|
val record = com.gzzn.omms.msgexchange.domain.flight.ScheduleRecord(
|
||||||
"121", mapOf("SODT" to "15DEC261723", "FLNO" to "CA002"),
|
"121", mapOf("SODT" to "15DEC261723", "FLNO" to "CA002"),
|
||||||
@@ -134,7 +134,7 @@ class FdelAndAdftProcessorTest {
|
|||||||
val events = StubMsgEvents()
|
val events = StubMsgEvents()
|
||||||
val adft = AdftProcessor(
|
val adft = AdftProcessor(
|
||||||
StubPipelineTx(), StubPipelineLock(), f, events, StubProcState(),
|
StubPipelineTx(), StubPipelineLock(), f, events, StubProcState(),
|
||||||
OperationDayProps().apply { zone = "Asia/Shanghai" }, ObjectMapper(),
|
OperationDayProps().apply { zone = "Asia/Shanghai" }, ObjectMapper(), java.time.Clock.systemUTC(),
|
||||||
)
|
)
|
||||||
|
|
||||||
val result = adft.apply(
|
val result = adft.apply(
|
||||||
@@ -160,7 +160,7 @@ class FdelAndAdftProcessorTest {
|
|||||||
)
|
)
|
||||||
val adft = AdftProcessor(
|
val adft = AdftProcessor(
|
||||||
StubPipelineTx(), StubPipelineLock(), f, StubMsgEvents(), StubProcState(),
|
StubPipelineTx(), StubPipelineLock(), f, StubMsgEvents(), StubProcState(),
|
||||||
OperationDayProps().apply { zone = "Asia/Shanghai" }, ObjectMapper(),
|
OperationDayProps().apply { zone = "Asia/Shanghai" }, ObjectMapper(), java.time.Clock.systemUTC(),
|
||||||
)
|
)
|
||||||
|
|
||||||
adft.apply(head(), msg(), com.gzzn.omms.msgexchange.domain.flight.ScheduleRecord("555", mapOf("FLNO" to "XX200")))
|
adft.apply(head(), msg(), com.gzzn.omms.msgexchange.domain.flight.ScheduleRecord("555", mapOf("FLNO" to "XX200")))
|
||||||
|
|||||||
@@ -83,6 +83,7 @@ class ScheduleProcessorTest {
|
|||||||
snapshotLog = log,
|
snapshotLog = log,
|
||||||
operationDayProps = OperationDayProps().apply { zone = "Asia/Shanghai"; cutoffHour = 0 },
|
operationDayProps = OperationDayProps().apply { zone = "Asia/Shanghai"; cutoffHour = 0 },
|
||||||
mapper = ObjectMapper(),
|
mapper = ObjectMapper(),
|
||||||
|
clock = java.time.Clock.systemUTC(),
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -129,7 +130,7 @@ class ScheduleProcessorTest {
|
|||||||
fun `replay of succeeded message records idempotent success without writes`() {
|
fun `replay of succeeded message records idempotent success without writes`() {
|
||||||
val proc = StubProcState()
|
val proc = StubProcState()
|
||||||
proc.insertIfAbsent(msgId, null)
|
proc.insertIfAbsent(msgId, null)
|
||||||
proc.markTerminal(msgId, ProcStatus.SUCCEEDED)
|
proc.markTerminal(msgId, ProcStatus.SUCCEEDED, now = java.time.Instant.now())
|
||||||
val flights = StubFlightState()
|
val flights = StubFlightState()
|
||||||
val log = StubSnapshotLog()
|
val log = StubSnapshotLog()
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user